4 Commits
Author SHA1 Message Date
Bruno Fargnoli b0e84c0f29 Fixed node "Trellis2 - String To File3D" 2026-09-25 17:00:38 +02:00
Bruno Fargnoli 14597418bb Updated ExportMesh node 2026-09-08 13:08:20 +02:00
Bruno Fargnoli 09b7d9a6ce Added node "Mesh Texturing Pixal3D MultiView"
node "Export Mesh" exports a model_3d, compatible with the new "Preview 3D" ComfyUI node.
2026-09-08 12:19:44 +02:00
Bruno Fargnoli 87491a6420 Added a node for Pixal3D MultiView used in 3D Gen Studio 2026-09-07 17:58:22 +02:00
5 changed files with 1366 additions and 10 deletions
+1
View File
@@ -14,6 +14,7 @@
| Date | Description |
| --- | --- |
| **2026-09-08** | Pixal3D: Added node "Mesh Texturing Pixal3D MultiView"<br>"Export Mesh" node exports model_3d compatible with new "Preview 3D" node |
| **2026-09-07** | Add support for Pixal3D MultiView |
| **2026-07-31** | Added new nodes "Smooth Mesh with PyMeshlab" and "Smooth Trimesh with PyMeshlab" |
| **2026-06-02** | Added new node "Render MultiView (Nvdiffrast)"<br>Thanks GiusTex |
@@ -0,0 +1,792 @@
{
"id": "759342bb-744a-4afa-90ae-92c19b9a88be",
"revision": 0,
"last_node_id": 32,
"last_link_id": 29,
"nodes": [
{
"id": 10,
"type": "Trellis2LoadMesh",
"pos": [
-3946.555858396243,
4830.563545196109
],
"size": [
300.095703125,
82
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "trimesh",
"type": "TRIMESH",
"links": [
22
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2LoadMesh"
},
"widgets_values": [
"C:\\Travaux\\Rimuru_3D.glb",
false
],
"widgets_values_named": {
"glb_path": "C:\\Travaux\\Rimuru_3D.glb",
"only_vertices_and_faces": false
}
},
{
"id": 17,
"type": "BatchImagesNode",
"pos": [
-4169.328444961045,
4947.077808694696
],
"size": [
140,
106
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "images.image0",
"type": "IMAGE",
"link": 13
},
{
"name": "images.image1",
"shape": 7,
"type": "IMAGE",
"link": 14
},
{
"name": "images.image2",
"shape": 7,
"type": "IMAGE",
"link": 15
},
{
"name": "images.image3",
"shape": 7,
"type": "IMAGE",
"link": null
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
16
]
}
],
"properties": {
"cnr_id": "comfy-core",
"ver": "0.34.0",
"Node name for S&R": "BatchImagesNode"
}
},
{
"id": 13,
"type": "Trellis2Pixal3DMultiViewConfig",
"pos": [
-3921.8074167155683,
4988.107064124985
],
"size": [
347.66015625,
290
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 16
},
{
"name": "masks",
"shape": 7,
"type": "MASK",
"link": null
},
{
"name": "moge_camera_config",
"shape": 7,
"type": "MOGE_CAM_CONFIG",
"link": null
}
],
"outputs": [
{
"name": "pixal3d_mv_views",
"type": "PIXAL3D_MV_VIEWS",
"links": [
23
]
},
{
"name": "moge_camera_config",
"type": "MOGE_CAM_CONFIG",
"links": [
25
]
},
{
"name": "preview",
"type": "IMAGE",
"links": []
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2Pixal3DMultiViewConfig"
},
"widgets_values": [
"0,90,180",
"0,0,0",
0.1,
"rad",
1,
0,
false,
false,
"auto"
],
"widgets_values_named": {
"azimuths": "0,90,180",
"elevations": "0,0,0",
"fov": 0.1,
"fov_unit": "rad",
"mesh_scale": 1,
"distance": 0,
"remove_background": false,
"invert_mask": false,
"framing": "auto"
}
},
{
"id": 14,
"type": "Trellis2LoadImageWithTransparency",
"pos": [
-4935.05186993715,
4777.234091036305
],
"size": [
314.6997010026156,
390.17747690166925
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": []
},
{
"name": "mask",
"type": "MASK",
"links": null
},
{
"name": "image_with_alpha",
"type": "IMAGE",
"links": [
13,
24
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2LoadImageWithTransparency"
},
"widgets_values": [
"Rimuru_Front.png",
"image"
],
"widgets_values_named": {
"image": "Rimuru_Front.png",
"upload": "image"
}
},
{
"id": 12,
"type": "Trellis2LoadModel",
"pos": [
-3957.156502525757,
4410.740682598691
],
"size": [
283.349609375,
250
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "pipeline",
"type": "TRELLIS2PIPELINE",
"links": [
21
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2LoadModel"
},
"widgets_values": [
"TencentARC/Pixal3D",
"flash_attn",
"cuda",
true,
false,
"flex_gemm",
"flash_attn",
false,
true
],
"widgets_values_named": {
"modelname": "TencentARC/Pixal3D",
"backend": "flash_attn",
"device": "cuda",
"low_vram": true,
"keep_models_loaded": false,
"conv_backend": "flex_gemm",
"sparse_backend": "flash_attn",
"use_reconviagen": false,
"pixal3d_multiview": true
}
},
{
"id": 16,
"type": "Trellis2LoadImageWithTransparency",
"pos": [
-4928.340096252822,
5236.061381907661
],
"size": [
314.6997010026156,
390.17747690166925
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": []
},
{
"name": "mask",
"type": "MASK",
"links": null
},
{
"name": "image_with_alpha",
"type": "IMAGE",
"links": [
14
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2LoadImageWithTransparency"
},
"widgets_values": [
"Image_1024_00227_.png",
"image"
],
"widgets_values_named": {
"image": "Image_1024_00227_.png",
"upload": "image"
}
},
{
"id": 15,
"type": "Trellis2LoadImageWithTransparency",
"pos": [
-4525.987077331534,
5234.150601332578
],
"size": [
314.6997010026156,
390.17747690166925
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": null
},
{
"name": "mask",
"type": "MASK",
"links": null
},
{
"name": "image_with_alpha",
"type": "IMAGE",
"links": [
15
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2LoadImageWithTransparency"
},
"widgets_values": [
"Image_1024_00226_.png",
"image"
],
"widgets_values_named": {
"image": "Image_1024_00226_.png",
"upload": "image"
}
},
{
"id": 19,
"type": "Trellis2ExportMesh",
"pos": [
-2916.8244182148565,
4634.524873512722
],
"size": [
270,
122
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "trimesh",
"type": "TRIMESH",
"link": 26
}
],
"outputs": [
{
"name": "glb_path",
"type": "STRING",
"links": []
},
{
"name": "relative_path",
"type": "STRING",
"links": null
},
{
"name": "model_3d",
"type": "FILE_3D",
"links": [
29
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2ExportMesh"
},
"widgets_values": [
"Rimuru_P3D_1024",
"glb"
],
"widgets_values_named": {
"filename_prefix": "Rimuru_P3D_1024",
"file_format": "glb"
}
},
{
"id": 27,
"type": "Trellis2MeshTexturingPixal3DMultiView",
"pos": [
-3472.1825617262434,
4637.888545018394
],
"size": [
419.15234375,
642
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "pipeline",
"type": "TRELLIS2PIPELINE",
"link": 21
},
{
"name": "trimesh",
"type": "TRIMESH",
"link": 22
},
{
"name": "pixal3d_mv_views",
"shape": 7,
"type": "PIXAL3D_MV_VIEWS",
"link": 23
},
{
"name": "image",
"shape": 7,
"type": "IMAGE",
"link": 24
},
{
"name": "moge_camera_config",
"shape": 7,
"type": "MOGE_CAM_CONFIG",
"link": 25
}
],
"outputs": [
{
"name": "trimesh",
"type": "TRIMESH",
"links": [
26
]
},
{
"name": "base_color_texture",
"type": "IMAGE",
"links": null
},
{
"name": "metallic_roughness_texture",
"type": "IMAGE",
"links": null
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
"ver": "87491a6420cfdf65a79f58782163cc0148efd513",
"Node name for S&R": "Trellis2MeshTexturingPixal3DMultiView"
},
"widgets_values": [
12345,
"fixed",
12,
1,
0.5,
3,
1024,
2048,
"OPAQUE",
false,
0.6,
0.9,
false,
false,
60,
"auto",
"euler",
"telea",
false,
0,
5,
1
],
"widgets_values_named": {
"seed": 12345,
"control_after_generate": "fixed",
"texture_steps": 12,
"texture_guidance_strength": 1,
"texture_guidance_rescale": 0.5,
"texture_rescale_t": 3,
"resolution": 1024,
"texture_size": 2048,
"texture_alpha_mode": "OPAQUE",
"double_side_material": false,
"texture_guidance_interval_start": 0.6,
"texture_guidance_interval_end": 0.9,
"bake_on_vertices": false,
"use_custom_normals": false,
"mesh_cluster_threshold_cone_half_angle_rad": 60,
"mesh_orientation": "auto",
"sampler": "euler",
"inpainting": "telea",
"verbose": false,
"dino_lock": 0,
"dino_substeps": 5,
"dino_foundation_cap": 1
}
},
{
"id": 30,
"type": "Preview3DAdvanced",
"pos": [
-2917.1508627052895,
4811.286329952889
],
"size": [
792.9444713556263,
966.4697042965545
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "model_3d",
"type": "FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
"link": 29
},
{
"name": "model_3d_info",
"shape": 7,
"type": "LOAD3D_MODEL_INFO",
"link": null
},
{
"name": "camera_info",
"shape": 7,
"type": "LOAD3D_CAMERA",
"link": null
}
],
"outputs": [
{
"name": "model_3d",
"type": "FILE_3D",
"links": null
},
{
"name": "model_3d_info",
"type": "LOAD3D_MODEL_INFO",
"links": null
},
{
"name": "camera_info",
"type": "LOAD3D_CAMERA",
"links": null
},
{
"name": "width",
"type": "INT",
"links": null
},
{
"name": "height",
"type": "INT",
"links": null
}
],
"properties": {
"cnr_id": "comfy-core",
"ver": "0.34.0",
"Node name for S&R": "Preview3DAdvanced",
"Camera Config": {
"cameraType": "perspective",
"fov": 35,
"state": {
"position": {
"x": -0.665585719053298,
"y": 0.5121245853121424,
"z": 0.6877965355795258
},
"target": {
"x": 0,
"y": 0,
"z": 0
},
"zoom": 1,
"cameraType": "perspective"
}
},
"Last Time Model File": "preview3d_advanced_184e39189db24f4ca7cbee796f783ab3.glb [temp]",
"Light Config": {
"intensity": 3,
"hdri": {
"enabled": false,
"hdriPath": "",
"showAsBackground": false,
"intensity": 1
}
},
"Scene Config": {
"showGrid": true,
"backgroundColor": "#282828",
"backgroundImage": "",
"backgroundRenderMode": "tiled",
"models": []
},
"Model Config": {
"upDirection": "original",
"materialMode": "original",
"showSkeleton": false,
"gizmo": {
"enabled": false,
"mode": "translate",
"position": {
"x": 0,
"y": 0,
"z": 0
},
"rotation": {
"x": 0,
"y": 0,
"z": 0
},
"scale": {
"x": 1,
"y": 1,
"z": 1
}
}
}
},
"widgets_values": [
"",
1024,
1024
],
"widgets_values_named": {
"viewport_state": "",
"width": 1024,
"height": 1024
}
}
],
"links": [
[
13,
14,
2,
17,
0,
"IMAGE"
],
[
14,
16,
2,
17,
1,
"IMAGE"
],
[
15,
15,
2,
17,
2,
"IMAGE"
],
[
16,
17,
0,
13,
0,
"IMAGE"
],
[
21,
12,
0,
27,
0,
"TRELLIS2PIPELINE"
],
[
22,
10,
0,
27,
1,
"TRIMESH"
],
[
23,
13,
0,
27,
2,
"PIXAL3D_MV_VIEWS"
],
[
24,
14,
2,
27,
3,
"IMAGE"
],
[
25,
13,
1,
27,
4,
"MOGE_CAM_CONFIG"
],
[
26,
27,
0,
19,
0,
"TRIMESH"
],
[
29,
19,
2,
30,
0,
"FILE_3D"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.5054470284993056,
"offset": [
5262.785779634466,
-4220.7667180975695
]
},
"frontendVersion": "1.51.10",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+271 -8
View File
@@ -942,8 +942,8 @@ class Trellis2ExportMesh:
}
}
RETURN_TYPES = ("STRING","STRING",)
RETURN_NAMES = ("glb_path","relative_path",)
RETURN_TYPES = ("STRING","STRING","FILE_3D",)
RETURN_NAMES = ("glb_path","relative_path","model_3d",)
FUNCTION = "process"
CATEGORY = "Trellis2Wrapper"
OUTPUT_NODE = True
@@ -964,7 +964,14 @@ class Trellis2ExportMesh:
relative_path = Path(subfolder) / f'{filename}_{counter:05}_.{file_format}'
return (str(output_glb_path), str(relative_path), )
from comfy_api.latest import Types
file_3d = Types.File3D(str(output_glb_path))
saved_name = f'{filename}_{counter:05}_.{file_format}'
return {
"ui": {"3d": [{"filename": saved_name, "subfolder": subfolder, "type": "output"}]},
"result": (str(output_glb_path), str(relative_path), file_3d,),
}
class Trellis2PostProcessMesh:
@classmethod
@@ -2850,7 +2857,7 @@ class Trellis2MeshTexturing:
verbose, dino_lock, dino_substeps, dino_foundation_cap, moge_camera_config = None):
if pipeline.isPixal3D:
raise Exception('Pixal3D does not support Mesh Texturing')
raise Exception('Pixal3D does not support this node: its denoisers take projected image features, not the global DINO conditioning built here. Use Trellis2 - Mesh Texturing Pixal3D MultiView instead.')
images = tensor_batch_to_pil_list(image, max_views=max_views)
image_in = images[0] if len(images) == 1 else images
@@ -3001,14 +3008,160 @@ class Trellis2MeshTexturingMultiView:
baseColorTexture = pil2tensor(baseColorTexture_np)
metallicRoughnessTexture = pil2tensor(metallicRoughnessTexture_np)
return (textured_mesh, baseColorTexture, metallicRoughnessTexture, )
return (textured_mesh, baseColorTexture, metallicRoughnessTexture, )
class Trellis2MeshTexturingPixal3DMultiView:
"""
Texture an existing mesh with Pixal3D, conditioned on posed multi-view images.
Runs only the texture stage of the Pixal3D cascade: the mesh is encoded with the
shape VAE and takes the place the generated shape latent normally holds, then the
tex denoiser is conditioned on the views projected into that same grid.
Pixal3D projects the views geometrically instead of blending global DINO features
the way Trellis2 - Mesh Texturing Multi-View does, so there is nothing to blend
(no front_axis / blend_temperature) and any number of views fuse by averaging --
but the mesh has to be posed and scaled the way the cameras describe. The mesh is
normalised to a tight bbox, so build the views with framing = auto (Trellis2 -
Pixal3D MultiView Config): a distance that disagrees with how the views are framed
makes the surface sample the background, which washes out the texture.
The mesh must also be axis-aligned the way the views are, i.e. azimuth 0 has to be
its front. Wire `image` + `moge_camera_config` instead of `pixal3d_mv_views` to run
the single-view Pixal3D weights.
"""
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"pipeline": ("TRELLIS2PIPELINE",),
"trimesh": ("TRIMESH",),
"seed": ("INT", {"default": 0, "min": 0, "max": 0x7fffffff}),
"texture_steps": ("INT",{"default":12, "min":1, "max":100},),
"texture_guidance_strength": ("FLOAT",{"default":1.00,"min":0.00,"max":99.99,"step":0.01}),
"texture_guidance_rescale": ("FLOAT",{"default":0.00,"min":0.00,"max":1.00,"step":0.01}),
"texture_rescale_t": ("FLOAT",{"default":3.00,"min":0.00,"max":9.99,"step":0.01}),
"resolution": ([1024,1536],{"default":1024,"tooltip":"Pixal3D has no 512 texture denoiser. The projected cond grid is 96^3 x 2048 at 1536 against 64^3 at 1024: measured peak allocation ~32 GB vs ~20 GB, so both lean on shared system memory on a 16 GB card and 1536 leans much harder."}),
"texture_size": ("INT",{"default":4096,"min":512,"max":16384}),
"texture_alpha_mode": (["OPAQUE","MASK","BLEND"],{"default":"OPAQUE"}),
"double_side_material": ("BOOLEAN",{"default":False}),
"texture_guidance_interval_start": ("FLOAT",{"default":0.60,"min":0.00,"max":1.00,"step":0.01}),
"texture_guidance_interval_end": ("FLOAT",{"default":0.90,"min":0.00,"max":1.00,"step":0.01}),
"bake_on_vertices": ("BOOLEAN",{"default":False}),
"use_custom_normals": ("BOOLEAN",{"default":False}),
"mesh_cluster_threshold_cone_half_angle_rad": ("FLOAT",{"default":60.0,"min":0.0,"max":359.9}),
"mesh_orientation": (["auto","none","90 degrees","-90 degrees"],{"default":"auto","tooltip":"Rotates the mesh into the frame the cameras describe, then rotates the result back. auto fits it by silhouette against the views. A Y-up glb (Load Mesh) needs \"90 degrees\"; a mesh straight out of Mesh With Voxel To Trimesh is already there, so \"none\". Getting this wrong is what makes the texture come out near-black."}),
"sampler": (["euler", "heun", "rk4", "rk5"], {"default": "euler"}),
"inpainting": (["telea","ns"],{"default":"telea"}),
"verbose": ("BOOLEAN",{"default":False}),
"dino_lock": ("FLOAT",{"default":0.00,"min":0.00,"max":1.00,"step":0.01}),
"dino_substeps": ("INT",{"default":4,"min":1,"max":99,"step":1}),
"dino_foundation_cap": ("FLOAT",{"default":1.00,"min":0.01,"max":1.00,"step":0.01}),
},
"optional": {
"pixal3d_mv_views": ("PIXAL3D_MV_VIEWS",),
"image": ("IMAGE",),
"moge_camera_config": ("MOGE_CAM_CONFIG",),
}
}
RETURN_TYPES = ("TRIMESH","IMAGE","IMAGE",)
RETURN_NAMES = ("trimesh","base_color_texture","metallic_roughness_texture",)
FUNCTION = "process"
CATEGORY = "Trellis2Wrapper"
OUTPUT_NODE = True
def process(self,
pipeline,
trimesh,
seed,
texture_steps,
texture_guidance_strength,
texture_guidance_rescale,
texture_rescale_t,
resolution,
texture_size,
texture_alpha_mode,
double_side_material,
texture_guidance_interval_start,
texture_guidance_interval_end,
bake_on_vertices,
use_custom_normals,
mesh_cluster_threshold_cone_half_angle_rad,
mesh_orientation,
sampler,
inpainting,
verbose,
dino_lock,
dino_substeps,
dino_foundation_cap,
pixal3d_mv_views = None,
image = None,
moge_camera_config = None):
if not pipeline.isPixal3D:
raise Exception('This node needs the Pixal3D pipeline. Select TencentARC/Pixal3D in '
'Trellis2 - LoadModel, or use Trellis2 - Mesh Texturing Multi-View '
'for TRELLIS.2.')
if pixal3d_mv_views is None and image is None:
raise Exception('Wire pixal3d_mv_views (multi-view weights), or image + '
'moge_camera_config for the single-view ones.')
camera_params = None
image_in = None
if pixal3d_mv_views is not None:
check_pixal3d_mv_pipeline(pipeline)
if image is not None:
print('[Pixal3D MV] pixal3d_mv_views is wired; ignoring the image input.')
else:
if moge_camera_config is None:
raise Exception('moge_camera_config is required when texturing from a single image')
images = tensor_batch_to_pil_list(image, max_views=16)
image_in = images[0] if len(images) == 1 else images
camera_params = moge_camera_config
reset_cuda()
texture_guidance_interval = [texture_guidance_interval_start,texture_guidance_interval_end]
tex_slat_sampler_params = {"steps":texture_steps,"guidance_strength":texture_guidance_strength,"guidance_rescale":texture_guidance_rescale,"guidance_interval":texture_guidance_interval,"rescale_t":texture_rescale_t}
textured_mesh, baseColorTexture_np, metallicRoughnessTexture_np = pipeline.texture_mesh_pixal3d(
mesh = trimesh,
views = pixal3d_mv_views,
image = image_in,
camera_params = camera_params,
seed = seed,
tex_slat_sampler_params = tex_slat_sampler_params,
resolution = resolution,
texture_size = texture_size,
texture_alpha_mode = texture_alpha_mode,
double_side_material = double_side_material,
bake_on_vertices = bake_on_vertices,
use_custom_normals = use_custom_normals,
mesh_cluster_threshold_cone_half_angle_rad = mesh_cluster_threshold_cone_half_angle_rad,
sampler = sampler,
inpainting = inpainting,
verbose = verbose,
dino_lock = dino_lock,
dino_substeps = dino_substeps,
dino_foundation_cap = dino_foundation_cap,
mesh_orientation = mesh_orientation
)
baseColorTexture = pil2tensor(baseColorTexture_np)
metallicRoughnessTexture = pil2tensor(metallicRoughnessTexture_np)
return (textured_mesh, baseColorTexture, metallicRoughnessTexture, )
class Trellis2LoadMesh:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"glb_path": ("STRING", {"default": "", "tooltip": "The glb path with mesh to load."}),
"glb_path": ("STRING", {"default": "", "tooltip": "The glb path with mesh to load."}),
"only_vertices_and_faces": ("BOOLEAN",{"default":False}),
}
}
RETURN_TYPES = ("TRIMESH",)
@@ -3019,12 +3172,15 @@ class Trellis2LoadMesh:
CATEGORY = "Trellis2Wrapper"
DESCRIPTION = "Loads a glb model from the given path."
def load(self, glb_path):
def load(self, glb_path, only_vertices_and_faces = False):
if not os.path.exists(glb_path):
glb_path = os.path.join(folder_paths.get_input_directory(), glb_path)
trimesh = Trimesh.load(glb_path, force="mesh")
if only_vertices_and_faces:
trimesh = Trimesh.Trimesh(vertices=trimesh.vertices,faces=trimesh.faces)
return (trimesh,)
class Trellis2PreProcessImage:
@@ -8247,6 +8403,107 @@ class Trellis2Pixal3DLoadMultiViewFolder:
'mesh_scale': float(views['mesh_scale'])}
return (views, cam_config, pixal3d_views_to_preview(views),)
class Trellis2SelectImagesForPixal3DMultiView:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"firstimage": ("STRING",{"default":""}),
"preprocess": ("BOOLEAN",{"default":False}),
"padding": ("INT",{"default":0,"min":0,"max":1024}),
"remove_background": ("BOOLEAN",{"default":False}),
"max_size": ("INT",{"default":2048,"min":512,"max":8192,"step":128}),
"azimuths": ("STRING",{"default":"0"}),
"elevations": ("STRING", {"default":"0"}),
},
"optional":{
"secondimage": ("STRING",{"default":""}),
"thirdimage": ("STRING",{"default":""}),
"fourthimage": ("STRING",{"default":""})
}
}
RETURN_TYPES = ("IMAGE", "STRING", "STRING",)
RETURN_NAMES = ("images", "azimuths", "elevations",)
FUNCTION = "process"
CATEGORY = "Trellis2Wrapper"
def load_view(self, path):
"""Resolve a path (absolute, or relative to the ComfyUI input dir) to an IMAGE tensor."""
if path is None or str(path).strip() == '':
return None
path = str(path).strip()
if not os.path.exists(path):
path = os.path.join(folder_paths.get_input_directory(), path)
if not os.path.exists(path):
return None
image = Image.open(path)
image = image.convert("RGBA" if 'A' in image.getbands() else "RGB")
return pil2tensor(image)
def process(self, firstimage, preprocess, padding, remove_background, max_size, azimuths, elevations, secondimage = None, thirdimage = None, fourthimage = None):
first_image = self.load_view(firstimage)
second_image = self.load_view(secondimage)
third_image = self.load_view(thirdimage)
fourth_image = self.load_view(fourthimage)
if first_image is None:
raise ValueError(f"Trellis2SelectImagesForPixal3DMultiView: could not find firstimage image '{firstimage}'")
allimages = []
if preprocess:
t2preprocess = Trellis2PreProcessImage()
# process() is a ComfyUI node function: it returns a 1-tuple, so unwrap it.
if first_image is not None:
first_image = t2preprocess.process(first_image, padding, remove_background, max_size)[0]
if second_image is not None:
second_image = t2preprocess.process(second_image, padding, remove_background, max_size)[0]
if third_image is not None:
third_image = t2preprocess.process(third_image, padding, remove_background, max_size)[0]
if fourth_image is not None:
fourth_image = t2preprocess.process(fourth_image, padding, remove_background, max_size)[0]
if first_image is not None:
allimages.append(first_image)
if second_image is not None:
allimages.append(second_image)
if third_image is not None:
allimages.append(third_image)
if fourth_image is not None:
allimages.append(fourth_image)
output_images = torch.cat(allimages, dim=0)
return (output_images, azimuths, elevations, )
class Trellis2StringToFile3D:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"glb_path": ("STRING",{"default":""}),
},
}
RETURN_TYPES = ("FILE_3D",)
RETURN_NAMES = ("model_3d",)
FUNCTION = "process"
CATEGORY = "Trellis2Wrapper"
def process(self, glb_path):
if not os.path.exists(glb_path):
glb_path = os.path.join(folder_paths.get_input_directory(), glb_path)
from comfy_api.latest import Types
file_3d = Types.File3D(glb_path)
return (file_3d, )
NODE_CLASS_MAPPINGS = {
"Trellis2LoadModel": Trellis2LoadModel,
@@ -8326,6 +8583,9 @@ NODE_CLASS_MAPPINGS = {
"Trellis2SmoothTrimeshWithPyMeshlab": Trellis2SmoothTrimeshWithPyMeshlab,
"Trellis2Pixal3DMultiViewConfig": Trellis2Pixal3DMultiViewConfig,
"Trellis2Pixal3DLoadMultiViewFolder": Trellis2Pixal3DLoadMultiViewFolder,
"Trellis2SelectImagesForPixal3DMultiView": Trellis2SelectImagesForPixal3DMultiView,
"Trellis2MeshTexturingPixal3DMultiView": Trellis2MeshTexturingPixal3DMultiView,
"Trellis2StringToFile3D": Trellis2StringToFile3D,
}
@@ -8407,4 +8667,7 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"Trellis2SmoothTrimeshWithPyMeshlab": "Trellis2 - Smooth Trimesh With PyMeshlab",
"Trellis2Pixal3DMultiViewConfig": "Trellis2 - Pixal3D MultiView Config",
"Trellis2Pixal3DLoadMultiViewFolder": "Trellis2 - Pixal3D Load MultiView Folder",
"Trellis2SelectImagesForPixal3DMultiView": "Trellis2 - Select Images For Pixal3D MultiView",
"Trellis2MeshTexturingPixal3DMultiView": "Trellis2 - Mesh Texturing Pixal3D MultiView",
"Trellis2StringToFile3D": "Trellis2 - String To File3D",
}
+287 -2
View File
@@ -572,13 +572,42 @@ class Trellis2ImageTo3DPipeline(Pipeline):
self.models['tex_slat_flow_model_1024'] = None
self._cleanup_cuda()
def _shape_slat_encoder_path(self) -> str:
"""
Resolve the shape VAE encoder, falling back to the TRELLIS.2-4B copy.
Pixal3D ships the shape decoder but no encoder, so encoding an existing mesh
(Mesh Texturing) has nothing to load from its own ckpts. Its shape VAE is the
TRELLIS.2-4B one -- shape_dec_next_dc_f16c32_fp16 is byte-identical in both
repos -- so the TRELLIS.2-4B encoder is the matching half and puts the mesh in
the very latent space Pixal3D's denoisers were trained on.
"""
local = f"{self.path}/ckpts/shape_enc_next_dc_f16c32_fp16"
if os.path.exists(f"{local}.safetensors"):
return local
trellis2_dir = os.path.join(folder_paths.models_dir, 'microsoft', 'TRELLIS.2-4B')
fallback = os.path.join(trellis2_dir, 'ckpts', 'shape_enc_next_dc_f16c32_fp16')
if not os.path.exists(f"{fallback}.safetensors"):
print('Shape Slat Encoder not found. Downloading it from microsoft/TRELLIS.2-4B ...')
from huggingface_hub import hf_hub_download
os.environ.setdefault("HF_HUB_ENABLE_HF_TRANSFER", "1")
for ext in ('json', 'safetensors'):
hf_hub_download(
repo_id='microsoft/TRELLIS.2-4B',
filename=f'ckpts/shape_enc_next_dc_f16c32_fp16.{ext}',
local_dir=trellis2_dir,
local_dir_use_symlinks=False,
)
return fallback
def load_shape_slat_encoder(self):
if self.models['shape_slat_encoder'] is None:
print('Loading Shape Slat Encoder model ...')
if getattr(self, 'use_fp8', False):
self.models['shape_slat_encoder'] = models.from_pretrained(f"{self.path}/ckpts_fp8/shape_enc_next_dc_f16c32_fp8")
else:
self.models['shape_slat_encoder'] = models.from_pretrained(f"{self.path}/ckpts/shape_enc_next_dc_f16c32_fp16")
self.models['shape_slat_encoder'] = models.from_pretrained(self._shape_slat_encoder_path())
self.models['shape_slat_encoder'].eval()
self.models['shape_slat_encoder'].to(self._device)
if hasattr(self.models['shape_slat_encoder'], 'low_vram'):
@@ -3563,7 +3592,263 @@ class Trellis2ImageTo3DPipeline(Pipeline):
out_mesh, baseColorTexture, metallicRoughnessTexture = self.postprocess_mesh(mesh, pbr_voxel, resolution, texture_size, texture_alpha_mode, double_side_material, bake_on_vertices, use_custom_normals, mesh_cluster_threshold_cone_half_angle_rad, inpainting)
return out_mesh, baseColorTexture, metallicRoughnessTexture
# Reorientations that map an incoming mesh into the frame preprocess_mesh expects,
# named and defined exactly like Trellis2MeshWithVoxelToTrimesh's reorient_vertices.
# The pipeline's own TRIMESH frame is Z-up (that node's "90 degrees" output), so a
# plain Y-up glb needs "90 degrees" applied again to get there.
MESH_ORIENTATIONS = {
'none': np.eye(4),
'90 degrees': np.array([[1., 0., 0., 0.], # (x, y, z) -> (x, z, -y)
[0., 0., 1., 0.],
[0., -1., 0., 0.],
[0., 0., 0., 1.]]),
'-90 degrees': np.array([[1., 0., 0., 0.], # (x, y, z) -> (x, -z, y)
[0., 0., -1., 0.],
[0., 1., 0., 0.],
[0., 0., 0., 1.]]),
}
def _view_silhouettes(self, mesh: trimesh.Trimesh, views: dict) -> List[np.ndarray]:
"""
Rasterise the mesh through each view's conditioning camera.
Uses calc_mat (F @ inv(C_0) @ C_i) and the same intrinsics as
project_points_to_image_batch, so the silhouettes show what the conditioning
actually reads -- not merely what the input poses say.
"""
from ..trainers.flow_matching.mixins.image_conditioned_proj import (
ProjGridMV, compute_relative_calc_mat)
from ..utils import mv_camera
res = mv_camera.ALPHA_SIZE
device = self.device
F_mat = ProjGridMV(grid_resolution=2, image_resolution=64).front_view_transform_matrix
calc_mat = compute_relative_calc_mat(
views['transform_matrix'].to(device).float(),
views['camera_distance'].to(device).float(),
F_mat.to(device).float(),
)[0] # [V, 4, 4]
cam_angle = views['camera_angle_x'][0].float()
mesh_scale = float(views['mesh_scale'])
# preprocess_mesh puts the mesh in [-0.5, 0.5]^3; ProjGrid's world box is the
# [-1, 1] grid over 2*mesh_scale, so world = preprocessed / mesh_scale.
pre = self.preprocess_mesh(mesh)
verts = torch.tensor(np.asarray(pre.vertices) / mesh_scale,
dtype=torch.float32, device=device)
# ... then into the Blender frame ProjGrid projects from.
rot = torch.tensor([[1., 0., 0.], [0., 0., -1.], [0., 1., 0.]], device=device)
verts = verts @ rot.T
faces = torch.tensor(np.asarray(pre.faces), dtype=torch.int32, device=device).contiguous()
try:
glctx = dr.RasterizeCudaContext()
except Exception:
glctx = dr.RasterizeGLContext()
out = []
near, far = 0.05, 100.0
for i in range(calc_mat.shape[0]):
w2c = torch.linalg.inv(calc_mat[i])
vc = verts @ w2c[:3, :3].T + w2c[:3, 3]
f = 1.0 / torch.tan(cam_angle[i] / 2).to(device)
clip = torch.stack([
f * vc[:, 0], f * vc[:, 1],
-(far + near) / (far - near) * vc[:, 2] - 2 * far * near / (far - near),
-vc[:, 2],
], dim=-1)[None].contiguous()
rast, _ = dr.rasterize(glctx, clip, faces, resolution=[res, res])
# nvdiffrast's row 0 is the bottom row; the masks have row 0 at the top.
out.append(torch.flip(rast[0, ..., 3] > 0, dims=[0]))
del glctx
return out
@torch.no_grad()
def fit_mesh_orientation(self, mesh: trimesh.Trimesh, views: dict,
candidates: List[str] = None) -> Tuple[str, float, dict]:
"""
Pick the reorientation whose silhouette best matches the views.
The projected conditioning has no way to notice that a mesh is posed
differently from the cameras: it just reads whatever the surface projects onto,
so a mesh in the wrong frame samples background everywhere and the texture
comes out near-black rather than failing. Scoring the silhouette against the
view masks catches that before 3 minutes of sampling.
Returns (name, mean IoU, {name: mean IoU}).
"""
if views.get('alphas') is None:
raise ValueError('views has no alpha masks; rebuild it with mv_camera.build_views')
if candidates is None:
candidates = list(self.MESH_ORIENTATIONS)
masks = (views['alphas'][0].to(self.device) > 0.8)
scores = {}
for name in candidates:
probe = mesh.copy()
probe.apply_transform(self.MESH_ORIENTATIONS[name])
ious = []
for sil, m in zip(self._view_silhouettes(probe, views), masks):
inter = (sil & m).sum().item()
union = (sil | m).sum().item()
ious.append(inter / union if union else 0.0)
scores[name] = float(np.mean(ious))
print(f'[Pixal3D MV] orientation {name!r}: silhouette IoU '
f'{", ".join(f"{v*100:.0f}%" for v in ious)} (mean {scores[name]*100:.0f}%)')
best = max(scores, key=scores.get)
return best, scores[best], scores
@torch.inference_mode()
def texture_mesh_pixal3d(
self,
mesh: trimesh.Trimesh,
views: dict = None,
image: Image.Image = None,
camera_params: dict = None,
seed: int = 42,
tex_slat_sampler_params: dict = {},
resolution: int = 1024,
texture_size: int = 2048,
texture_alpha_mode = 'OPAQUE',
double_side_material = True,
bake_on_vertices = False,
use_custom_normals = False,
mesh_cluster_threshold_cone_half_angle_rad = 60.0,
sampler: str = 'euler',
inpainting: str = 'telea',
verbose: bool = False,
dino_lock: float = 0.0,
dino_substeps: int = 4,
dino_foundation_cap: float = 0.92,
mesh_orientation: str = 'auto',
):
"""
Texture an existing mesh with Pixal3D, i.e. run only the tex stage of its
cascade over a shape latent that came from a mesh instead of the shape stages.
The mesh takes the place the generated shape latent normally holds: encode it
with the shape VAE, then condition the tex denoiser on the same projected
image features the cascade uses (get_proj_cond_shape*), at the grid resolution
implied by `resolution`. `sample_tex_slat` concatenates the shape latent, so
the tex denoiser sees exactly what it does mid-cascade.
Unlike texture_mesh_multiview (global DINO features blended per view by a
heuristic), the conditioning here is a geometric un-projection: the views need
no blending, any number of them fuse by averaging, but the mesh has to be
posed and scaled the way the cameras describe. preprocess_mesh normalises to a
tight bbox in [-0.5, 0.5]^3, so if the views frame the object differently the
grid samples off-surface and the texture washes out -- build the views with
framing = auto (Trellis2 - Pixal3D MultiView Config), which fits the camera
distance to the silhouette.
Args:
views: the bundle from mv_camera.build_views(); needs the multi-view
weights (pipeline_mv.json). Takes precedence over `image`.
image: single view, for the non-multi-view Pixal3D weights. Needs
`camera_params` (camera_angle_x / distance / mesh_scale).
mesh_orientation: how to rotate the mesh into the frame the cameras
describe -- 'auto' fits it by silhouette (needs `views`), or name one
of MESH_ORIENTATIONS. The output is rotated back, so the textured mesh
comes out in the frame it went in.
"""
if views is None and image is None:
raise ValueError('texture_mesh_pixal3d needs either views or an image')
if views is None and camera_params is None:
raise ValueError('texture_mesh_pixal3d needs camera_params alongside a single image')
if resolution % 16 != 0:
raise ValueError(f'resolution must be a multiple of 16, got {resolution}')
self.switch_samplers(sampler)
if mesh_orientation == 'auto':
if views is None:
# Nothing to score a silhouette against.
print("[Pixal3D] mesh_orientation 'auto' has no view masks to fit against on "
"the single-image path; assuming 'none'. A y-up glb needs '90 degrees' "
"-- set it explicitly if the texture comes out near-black.")
mesh_orientation = 'none'
else:
mesh_orientation, iou, _ = self.fit_mesh_orientation(mesh, views)
print(f'[Pixal3D] mesh_orientation auto -> {mesh_orientation!r} (IoU {iou*100:.0f}%)')
if iou < 0.5:
print('[Pixal3D] Warning: even the best orientation matches the views '
f'only {iou*100:.0f}%. Check that the azimuths/elevations describe '
'these views, that the mesh is the object in them, and that '
'framing = auto.')
if mesh_orientation not in self.MESH_ORIENTATIONS:
raise ValueError(f'unknown mesh_orientation {mesh_orientation!r}; '
f'expected auto or one of {list(self.MESH_ORIENTATIONS)}')
reorient = self.MESH_ORIENTATIONS[mesh_orientation]
if mesh_orientation != 'none':
mesh = mesh.copy()
mesh.apply_transform(reorient)
mesh = self.preprocess_mesh(mesh)
seed_all(seed)
shape_slat = self.encode_shape_slat(mesh, resolution)
# The cascade drives the cond grid off the resolution it upsampled to; here
# that is just the grid the encoder produced, i.e. resolution // 16.
tex_grid_res = resolution // 16
if views is not None:
image_cond_model = self.load_pixal3d_mv_image_cond_tex_1024()
cond = self.get_proj_cond_shape_mv(
image_cond_model, views, shape_slat.coords,
grid_resolution_override=tex_grid_res,
)
del image_cond_model
if not self.keep_models_loaded:
self.unload_pixal3d_mv_image_cond_tex_1024()
else:
images = list(image) if isinstance(image, (list, tuple)) else [image]
image_cond_model = self.load_pixal3d_image_cond_tex_1024()
cond = self.get_proj_cond_shape(
image_cond_model, images, shape_slat.coords,
camera_angle_x=camera_params['camera_angle_x'],
distance=camera_params['distance'],
mesh_scale=camera_params.get('mesh_scale', 1.0),
grid_resolution_override=tex_grid_res,
)
del image_cond_model
if not self.keep_models_loaded:
self.unload_pixal3d_image_cond_tex_1024()
torch.cuda.empty_cache()
# Pixal3D has no 512 tex denoiser; the 1024 one is rope-based and is what the
# cascade already runs at every resolution it upsamples to.
self.load_tex_slat_flow_model_1024()
tex_slat = self.sample_tex_slat(
cond, self.models['tex_slat_flow_model_1024'],
shape_slat, tex_slat_sampler_params,
verbose = verbose,
dino_lock = dino_lock,
dino_substeps = dino_substeps,
dino_foundation_cap = dino_foundation_cap
)
if not self.keep_models_loaded:
self.unload_tex_slat_flow_model_1024()
del cond
torch.cuda.empty_cache()
pbr_voxel = self.decode_tex_slat(tex_slat)
torch.cuda.empty_cache()
out_mesh, baseColorTexture, metallicRoughnessTexture = self.postprocess_mesh(mesh, pbr_voxel, resolution, texture_size, texture_alpha_mode, double_side_material, bake_on_vertices, use_custom_normals, mesh_cluster_threshold_cone_half_angle_rad, inpainting)
# preprocess/postprocess_mesh are inverses, so the mesh comes back in the frame
# it went into them; undo the reorientation too, to hand back the caller's frame.
if mesh_orientation != 'none':
out_mesh.apply_transform(np.linalg.inv(reorient))
return out_mesh, baseColorTexture, metallicRoughnessTexture
def get_coords_from_trimesh(self, mesh, resolution):
vertices = torch.from_numpy(mesh.vertices).float()
faces = torch.from_numpy(mesh.faces).long()
+15
View File
@@ -57,6 +57,9 @@ _WORLD_UP = (0.0, 0.0, 1.0)
# object until it fills the frame.
PIXAL3D_RIG_MARGIN = 1.1
# Resolution the per-view masks are kept at, for silhouette checks only.
ALPHA_SIZE = 512
def camera_distance_for_extent(camera_angle_x: float, half_extent_px: float,
mesh_scale: float = 1.0, image_resolution: int = 512) -> float:
@@ -223,6 +226,17 @@ def build_views(
for size in image_sizes # [1, V, 3, S, S]
}
# The conditioning images are alpha-premultiplied, so a dark object is
# indistinguishable from background in them. Keep the masks separately -- that is
# what a silhouette check (fit_mesh_orientation) needs to tell whether a mesh is
# posed the way these cameras describe.
alphas = torch.stack([
torch.tensor(np.array(im.convert('RGBA')
.resize((ALPHA_SIZE, ALPHA_SIZE), Image.Resampling.LANCZOS)
.getchannel(3))).float() / 255.0
for im in images
], dim=0)[None] # [1, V, S, S]
if view_names is None:
view_names = [f"view{i:02d}" for i in range(V)]
@@ -232,6 +246,7 @@ def build_views(
return {
'images': bundle_images,
'alphas': alphas,
'camera_angle_x': cax,
'camera_distance': camera_distance,
'transform_matrix': tm,