diff --git a/README.md b/README.md
index fc77c1e..c355fbc 100644
--- a/README.md
+++ b/README.md
@@ -14,6 +14,7 @@
| Date | Description |
| --- | --- |
+| **2026-04-04** | Added node "Sparse Generator with ReconViaGen" |
| **2026-04-01** | Added node "Voxel to Mesh"
It replaces Remeshing to make watertight mesh |
| **2026-03-21** | Added node "Projection HighPoly to LowPoly"
Added node "Render MultiView" |
| **2026-03-17** | Added Inpainting Choice NS and TELEA |
diff --git a/example_workflows/Advanced.json b/example_workflows/Advanced.json
index 20e4ce3..1f51ae1 100644
--- a/example_workflows/Advanced.json
+++ b/example_workflows/Advanced.json
@@ -1,22 +1,136 @@
{
"id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
"revision": 0,
- "last_node_id": 55,
- "last_link_id": 117,
+ "last_node_id": 207,
+ "last_link_id": 416,
"nodes": [
{
- "id": 10,
- "type": "Preview3D",
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
"pos": [
- 1739.3745174243877,
- 813.1451251994486
+ -377.72410571388025,
+ 328.4870830026766
],
"size": [
- 1216.03125,
- 1078.3125
+ 717.521556382955,
+ 915.5313594325766
],
"flags": {},
- "order": 6,
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -373.0898096413628,
+ 163.12227474974895
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 409
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -368.2858630795249,
+ 34.401714106564214
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 410
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 376.48890096546415,
+ 331.2515281089069
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 12,
"mode": 0,
"inputs": [
{
@@ -33,111 +147,37 @@
},
{
"name": "model_file",
- "type": "STRING",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
"widget": {
"name": "model_file"
},
- "link": 22
+ "link": 398
}
],
"outputs": [],
"properties": {
"cnr_id": "comfy-core",
- "ver": "0.4.0",
- "Node name for S&R": "Preview3D",
- "widget_ue_connectable": {},
- "Last Time Model File": "ArmoredWarrior_00002_.glb",
- "Scene Config": {
- "showGrid": true,
- "backgroundColor": "#282828",
- "backgroundImage": "",
- "backgroundRenderMode": "tiled"
- },
- "Camera Config": {
- "cameraType": "perspective",
- "fov": 35,
- "state": {
- "position": {
- "x": 1.4101977762939673,
- "y": 2.397771625867536,
- "z": 8.997201229712031
- },
- "target": {
- "x": 1.4732208254010746e-177,
- "y": 2.5,
- "z": 6.912872131937994e-178
- },
- "zoom": 1,
- "cameraType": "perspective"
- }
- },
- "Light Config": {
- "intensity": 3
- }
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
},
"widgets_values": [
- "ArmoredWarrior_00002_.glb",
+ "",
""
]
},
- {
- "id": 6,
- "type": "Trellis2LoadImageWithTransparency",
- "pos": [
- 138.06663994864374,
- 1231.972182577106
- ],
- "size": [
- 470.78125,
- 439.8125
- ],
- "flags": {},
- "order": 0,
- "mode": 0,
- "inputs": [],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": null
- },
- {
- "name": "mask",
- "type": "MASK",
- "links": null
- },
- {
- "name": "image_with_alpha",
- "type": "IMAGE",
- "links": [
- 116
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
- "Node name for S&R": "Trellis2LoadImageWithTransparency",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- "Image_2048_00008_.png",
- "image"
- ]
- },
{
"id": 39,
"type": "Trellis2LoadModel",
"pos": [
- 243.7073730589844,
- 824.811803650631
+ 404.6117778691903,
+ -564.6350558283436
],
"size": [
- 362.0625,
- 207.328125
+ 301.71875,
+ 202
],
"flags": {},
- "order": 1,
+ "order": 3,
"mode": 0,
"inputs": [],
"outputs": [
@@ -145,7 +185,7 @@
"name": "pipeline",
"type": "TRELLIS2PIPELINE",
"links": [
- 87
+ 411
]
}
],
@@ -156,79 +196,76 @@
"widget_ue_connectable": {}
},
"widgets_values": [
- "TRELLIS.2-4B",
+ "microsoft/TRELLIS.2-4B",
"flash_attn",
"cuda",
true,
- false
+ false,
+ "flex_gemm",
+ "flash_attn"
]
},
{
- "id": 19,
- "type": "Trellis2ExportMesh",
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
"pos": [
- 1583.1015090615297,
- 594.4780695351682
+ 413.7442818307099,
+ -185.99180300678682
],
"size": [
- 324,
- 146
+ 281.8229166666667,
+ 106
],
"flags": {},
- "order": 5,
+ "order": 4,
"mode": 0,
"inputs": [
{
- "name": "trimesh",
- "type": "TRIMESH",
- "link": 115
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
}
],
"outputs": [
{
- "name": "glb_path",
- "type": "STRING",
+ "name": "image",
+ "type": "IMAGE",
"links": [
- 22
+ 412
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
- "Node name for S&R": "Trellis2ExportMesh",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
"widget_ue_connectable": {}
},
"widgets_values": [
- "Trellis2Mesh",
- "glb",
- true
+ 10,
+ false,
+ 1024
]
},
{
- "id": 45,
- "type": "Trellis2MeshWithVoxelAdvancedGenerator",
+ "id": 196,
+ "type": "Trellis2FillHolesWithCuMesh",
"pos": [
- 687.5583313121791,
- 819.8533042235456
+ 1148.203832543889,
+ -691.4362805609004
],
"size": [
- 495.8125,
- 866
+ 312.4361328125,
+ 58
],
"flags": {},
- "order": 3,
+ "order": 6,
"mode": 0,
"inputs": [
{
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
- "link": 87
- },
- {
- "name": "image",
- "type": "IMAGE",
- "link": 117
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 413
}
],
"outputs": [
@@ -236,41 +273,175 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 113
+ 392
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 197,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1143.7616665123248,
+ -576.3871604957823
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 392
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 393
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 198,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1146.6254930330851,
+ -381.0428638307529
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 393
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 409
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 394
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 205,
+ "type": "Trellis2MeshWithVoxelAdvancedGenerator",
+ "pos": [
+ 717.5499413662296,
+ -588.6883997158563
+ ],
+ "size": [
+ 413.1841796875,
+ 702
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 411
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 412
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 413
]
},
{
"name": "bvh",
"type": "BVH",
"links": [
- 114
+ 415
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a28d6cf93b0cf24aa9e7d3dccfd14f406227fbd0",
- "Node name for S&R": "Trellis2MeshWithVoxelAdvancedGenerator",
- "widget_ue_connectable": {}
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelAdvancedGenerator"
},
"widgets_values": [
12345,
- "randomize",
+ "fixed",
"1024_cascade",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
12,
- 6.5,
- 0.2,
- 4,
- 12,
- 6.5,
- 0.2,
- 4,
+ 7.5,
+ 0.01,
+ 3,
12,
3,
- 0.2,
+ 0.01,
3,
999999,
- 4,
+ 1,
32,
true,
0.1,
@@ -279,33 +450,129 @@
1,
0,
0.9,
- true
+ true,
+ "euler"
]
},
{
- "id": 54,
- "type": "Trellis2PostProcessAndUnWrapAndRasterizer",
+ "id": 201,
+ "type": "Trellis2ExportMesh",
"pos": [
- 1262.5083613749216,
- 819.4992892507565
+ 1597.089782460921,
+ -140.8020848589493
],
"size": [
- 419.125,
- 651.328125
+ 270,
+ 102
],
"flags": {},
- "order": 4,
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 416
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 410
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 398
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ },
+ {
+ "id": 199,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1151.769446503904,
+ -245.92060963339878
+ ],
+ "size": [
+ 302.4315956250001,
+ 47.18107670916754
+ ],
+ "flags": {},
+ "order": 9,
"mode": 0,
"inputs": [
{
"name": "mesh",
"type": "MESHWITHVOXEL",
- "link": 113
+ "link": 394
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 414
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 207,
+ "type": "Trellis2UnWrapAndRasterizer",
+ "pos": [
+ 1153.8343873953147,
+ -150.96148118668327
+ ],
+ "size": [
+ 419.15234375,
+ 314
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 414
},
{
"name": "bvh",
"type": "BVH",
- "link": 114
+ "link": 415
}
],
"outputs": [
@@ -313,7 +580,7 @@
"name": "trimesh",
"type": "TRIMESH",
"links": [
- 115
+ 416
]
},
{
@@ -329,144 +596,170 @@
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72",
- "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer",
- "widget_ue_connectable": {}
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2UnWrapAndRasterizer"
},
"widgets_values": [
60,
0,
1,
1,
- 4096,
- true,
- 1,
- 0,
- 2000000,
- "Cumesh",
- true,
+ 2048,
"OPAQUE",
- "1024",
- false,
- true,
false,
false,
- true
- ]
- },
- {
- "id": 55,
- "type": "Trellis2PreProcessImage",
- "pos": [
- 310.038958285416,
- 1092.8065963153765
- ],
- "size": [
- 281.78125,
- 79.328125
- ],
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "link": 116
- }
- ],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": [
- 117
- ]
- }
- ],
- "properties": {
- "widget_ue_connectable": {},
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72",
- "Node name for S&R": "Trellis2PreProcessImage"
- },
- "widgets_values": [
- 25
+ false,
+ "telea"
]
}
],
"links": [
[
- 22,
- 19,
- 0,
- 10,
+ 241,
+ 69,
2,
- "STRING"
+ 119,
+ 0,
+ "IMAGE"
],
[
- 87,
- 39,
+ 392,
+ 196,
0,
- 45,
- 0,
- "TRELLIS2PIPELINE"
- ],
- [
- 113,
- 45,
- 0,
- 54,
+ 197,
0,
"MESHWITHVOXEL"
],
[
- 114,
- 45,
+ 393,
+ 197,
+ 0,
+ 198,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 394,
+ 198,
+ 0,
+ 199,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 398,
+ 201,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 409,
+ 203,
+ 0,
+ 198,
1,
- 54,
+ "INT"
+ ],
+ [
+ 410,
+ 204,
+ 0,
+ 201,
+ 1,
+ "STRING"
+ ],
+ [
+ 411,
+ 39,
+ 0,
+ 205,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 412,
+ 119,
+ 0,
+ 205,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 413,
+ 205,
+ 0,
+ 196,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 414,
+ 199,
+ 0,
+ 207,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 415,
+ 205,
+ 1,
+ 207,
1,
"BVH"
],
[
- 115,
- 54,
+ 416,
+ 207,
0,
- 19,
+ 201,
0,
"TRIMESH"
- ],
- [
- 116,
- 6,
- 2,
- 55,
- 0,
- "IMAGE"
- ],
- [
- 117,
- 55,
- 0,
- 45,
- 1,
- "IMAGE"
]
],
- "groups": [],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -387.72410571388025,
+ -53.37077389343578,
+ 737.521556382955,
+ 1307.389216328689
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 394.6117778691903,
+ -772.2962805609001,
+ 1523.2043605195558,
+ 1035.4596396805657
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
"config": {},
"extra": {
- "workflowRendererVersion": "Vue",
+ "workflowRendererVersion": "LG",
"ue_links": [],
"ds": {
- "scale": 0.520986848192445,
+ "scale": 0.5644739300537777,
"offset": [
- 91.64617544164359,
- -275.12756326798984
+ 717.24240821608,
+ 973.4146936919584
]
},
"links_added_by_ue": [],
- "frontendVersion": "1.37.11",
+ "frontendVersion": "1.42.8",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
diff --git a/example_workflows/Advanced_CustomSteps.json b/example_workflows/Advanced_CustomSteps.json
new file mode 100644
index 0000000..d5be7db
--- /dev/null
+++ b/example_workflows/Advanced_CustomSteps.json
@@ -0,0 +1,1249 @@
+{
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
+ "revision": 0,
+ "last_node_id": 223,
+ "last_link_id": 456,
+ "nodes": [
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -377.72410571388025,
+ 328.4870830026766
+ ],
+ "size": [
+ 717.521556382955,
+ 915.5313594325766
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 213,
+ "type": "Trellis2ShapeGenerator",
+ "pos": [
+ 783.1875248410059,
+ -209.1034640968784
+ ],
+ "size": [
+ 335.24609375,
+ 266
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 428
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 429
+ },
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "link": 430
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 433
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 434
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 431
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeGenerator"
+ },
+ "widgets_values": [
+ 512,
+ 25,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 212,
+ "type": "Trellis2SparseGenerator",
+ "pos": [
+ 755.5154833297573,
+ -586.166449201357
+ ],
+ "size": [
+ 395.65625,
+ 314
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 426
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 427
+ }
+ ],
+ "outputs": [
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "links": [
+ 430
+ ]
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "links": [
+ 435
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 428
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2SparseGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ "euler",
+ 32,
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 385.7151992875252,
+ -750.0584765375115
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 444
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
+ ]
+ },
+ {
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 424.3737559582122,
+ -335.9840037159542
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 445
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 208,
+ "type": "Trellis2ImageCondGenerator",
+ "pos": [
+ 779.7493550843124,
+ -745.7690930815886
+ ],
+ "size": [
+ 308.0748046875,
+ 98
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 444
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 445
+ }
+ ],
+ "outputs": [
+ {
+ "name": "cond_512",
+ "type": "IMAGE_COND",
+ "links": [
+ 427,
+ 429
+ ]
+ },
+ {
+ "name": "cond_1024",
+ "type": "IMAGE_COND",
+ "links": [
+ 432,
+ 447
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 426
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ImageCondGenerator"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 214,
+ "type": "Trellis2ShapeCascadeGenerator",
+ "pos": [
+ 1204.298470222733,
+ -593.6739172678739
+ ],
+ "size": [
+ 335.9791015625,
+ 338
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 431
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 432
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 433
+ },
+ {
+ "name": "from_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "from_resolution"
+ },
+ "link": 434
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "sparse_structure_resolution"
+ },
+ "link": 435
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 423,
+ 448
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 424
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 446
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeCascadeGenerator"
+ },
+ "widgets_values": [
+ 0,
+ 1024,
+ 0,
+ 999999,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -373.0898096413628,
+ 163.12227474974895
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 451
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 216,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1703.393750669334,
+ -500.59691468161645
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 437
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 451
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 436
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 220,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 1670.7679554530735,
+ -854.0252749431562
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 443
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 425
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 211,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1658.76843203833,
+ -712.5207015189599
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 425
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 437
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 215,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1682.2010450446755,
+ -361.2081091911047
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 436
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 452
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 209,
+ "type": "Trellis2DecodeLatents",
+ "pos": [
+ 1214.7402059785773,
+ 139.23732698343224
+ ],
+ "size": [
+ 270,
+ 122
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 450
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 423
+ },
+ {
+ "name": "texture_slat",
+ "shape": 7,
+ "type": "TEXTURE_SLAT",
+ "link": 449
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "widget": {
+ "name": "resolution"
+ },
+ "link": 424
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 443
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": [
+ 453
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2DecodeLatents"
+ },
+ "widgets_values": [
+ 0,
+ true
+ ]
+ },
+ {
+ "id": 222,
+ "type": "Trellis2UnWrapAndRasterizer",
+ "pos": [
+ 1627.2488876663126,
+ -234.41356008052338
+ ],
+ "size": [
+ 419.15234375,
+ 314
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 452
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "link": 453
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 454
+ ]
+ },
+ {
+ "name": "base_color_texture",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "metallic_roughness_texture",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2UnWrapAndRasterizer"
+ },
+ "widgets_values": [
+ 60,
+ 0,
+ 1,
+ 1,
+ 2048,
+ "OPAQUE",
+ false,
+ false,
+ false,
+ "telea"
+ ]
+ },
+ {
+ "id": 223,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ 1641.6918045364303,
+ 146.40641977787487
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 454
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 456
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 455
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 405.0700414873716,
+ 334.1097337450922
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 17,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 455
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
+ },
+ "widgets_values": [
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -368.2858630795249,
+ 34.401714106564214
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 456
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 221,
+ "type": "Trellis2TexSlatGenerator",
+ "pos": [
+ 1188.8316361245397,
+ -197.22556623205915
+ ],
+ "size": [
+ 340.390625,
+ 266
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 446
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 447
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 448
+ }
+ ],
+ "outputs": [
+ {
+ "name": "texture_slat",
+ "type": "TEXTURE_SLAT",
+ "links": [
+ 449
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 450
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2TexSlatGenerator"
+ },
+ "widgets_values": [
+ 1024,
+ 12,
+ 3,
+ 0.01,
+ 3,
+ "euler",
+ 0,
+ 0.9
+ ]
+ }
+ ],
+ "links": [
+ [
+ 241,
+ 69,
+ 2,
+ 119,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 423,
+ 214,
+ 0,
+ 209,
+ 1,
+ "SHAPE_SLAT"
+ ],
+ [
+ 424,
+ 214,
+ 1,
+ 209,
+ 3,
+ "INT"
+ ],
+ [
+ 425,
+ 220,
+ 0,
+ 211,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 426,
+ 208,
+ 2,
+ 212,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 427,
+ 208,
+ 0,
+ 212,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 428,
+ 212,
+ 2,
+ 213,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 429,
+ 208,
+ 0,
+ 213,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 430,
+ 212,
+ 0,
+ 213,
+ 2,
+ "COORDS"
+ ],
+ [
+ 431,
+ 213,
+ 2,
+ 214,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 432,
+ 208,
+ 1,
+ 214,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 433,
+ 213,
+ 0,
+ 214,
+ 2,
+ "SHAPE_SLAT"
+ ],
+ [
+ 434,
+ 213,
+ 1,
+ 214,
+ 3,
+ "INT"
+ ],
+ [
+ 435,
+ 212,
+ 1,
+ 214,
+ 4,
+ "INT"
+ ],
+ [
+ 436,
+ 216,
+ 0,
+ 215,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 437,
+ 211,
+ 0,
+ 216,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 443,
+ 209,
+ 0,
+ 220,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 444,
+ 39,
+ 0,
+ 208,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 445,
+ 119,
+ 0,
+ 208,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 446,
+ 214,
+ 2,
+ 221,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 447,
+ 208,
+ 1,
+ 221,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 448,
+ 214,
+ 0,
+ 221,
+ 2,
+ "SHAPE_SLAT"
+ ],
+ [
+ 449,
+ 221,
+ 0,
+ 209,
+ 2,
+ "TEXTURE_SLAT"
+ ],
+ [
+ 450,
+ 221,
+ 1,
+ 209,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 451,
+ 203,
+ 0,
+ 216,
+ 1,
+ "INT"
+ ],
+ [
+ 452,
+ 215,
+ 0,
+ 222,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 453,
+ 209,
+ 1,
+ 222,
+ 1,
+ "BVH"
+ ],
+ [
+ 454,
+ 222,
+ 0,
+ 223,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 455,
+ 223,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 456,
+ 204,
+ 0,
+ 223,
+ 1,
+ "STRING"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -387.72410571388025,
+ -53.37077389343578,
+ 737.521556382955,
+ 1307.389216328689
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 375.7151992875252,
+ -957.7197012700683,
+ 1860.9819191012211,
+ 1225.6071509713981
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "config": {},
+ "extra": {
+ "workflowRendererVersion": "LG",
+ "ue_links": [],
+ "ds": {
+ "scale": 0.6209213230591558,
+ "offset": [
+ 341.8543937245997,
+ 855.1515920610424
+ ]
+ },
+ "links_added_by_ue": [],
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/Advanced_CustomSteps_MeshOnly.json b/example_workflows/Advanced_CustomSteps_MeshOnly.json
new file mode 100644
index 0000000..844feeb
--- /dev/null
+++ b/example_workflows/Advanced_CustomSteps_MeshOnly.json
@@ -0,0 +1,1118 @@
+{
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
+ "revision": 0,
+ "last_node_id": 224,
+ "last_link_id": 459,
+ "nodes": [
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -377.72410571388025,
+ 328.4870830026766
+ ],
+ "size": [
+ 717.521556382955,
+ 915.5313594325766
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 213,
+ "type": "Trellis2ShapeGenerator",
+ "pos": [
+ 783.1875248410059,
+ -209.1034640968784
+ ],
+ "size": [
+ 335.24609375,
+ 266
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 428
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 429
+ },
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "link": 430
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 433
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 434
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 431
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeGenerator"
+ },
+ "widgets_values": [
+ 512,
+ 25,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 212,
+ "type": "Trellis2SparseGenerator",
+ "pos": [
+ 755.5154833297573,
+ -586.166449201357
+ ],
+ "size": [
+ 395.65625,
+ 314
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 426
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 427
+ }
+ ],
+ "outputs": [
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "links": [
+ 430
+ ]
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "links": [
+ 435
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 428
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2SparseGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ "euler",
+ 32,
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 385.7151992875252,
+ -750.0584765375115
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 444
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
+ ]
+ },
+ {
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 424.3737559582122,
+ -335.9840037159542
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 445
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 208,
+ "type": "Trellis2ImageCondGenerator",
+ "pos": [
+ 779.7493550843124,
+ -745.7690930815886
+ ],
+ "size": [
+ 308.0748046875,
+ 98
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 444
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 445
+ }
+ ],
+ "outputs": [
+ {
+ "name": "cond_512",
+ "type": "IMAGE_COND",
+ "links": [
+ 427,
+ 429
+ ]
+ },
+ {
+ "name": "cond_1024",
+ "type": "IMAGE_COND",
+ "links": [
+ 432
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 426
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ImageCondGenerator"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 214,
+ "type": "Trellis2ShapeCascadeGenerator",
+ "pos": [
+ 1204.298470222733,
+ -593.6739172678739
+ ],
+ "size": [
+ 335.9791015625,
+ 338
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 431
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 432
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 433
+ },
+ {
+ "name": "from_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "from_resolution"
+ },
+ "link": 434
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "sparse_structure_resolution"
+ },
+ "link": 435
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 423
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 424
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 457
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeCascadeGenerator"
+ },
+ "widgets_values": [
+ 0,
+ 1024,
+ 0,
+ 999999,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -373.0898096413628,
+ 163.12227474974895
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 451
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 220,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 1670.7679554530735,
+ -854.0252749431562
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 443
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 425
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 211,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1658.76843203833,
+ -712.5207015189599
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 425
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 437
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 405.0700414873716,
+ 334.1097337450922
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 455
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
+ },
+ "widgets_values": [
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -368.2858630795249,
+ 34.401714106564214
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 456
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 209,
+ "type": "Trellis2DecodeLatents",
+ "pos": [
+ 1222.2559520778204,
+ -168.90688691732427
+ ],
+ "size": [
+ 270,
+ 122
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 457
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 423
+ },
+ {
+ "name": "texture_slat",
+ "shape": 7,
+ "type": "TEXTURE_SLAT",
+ "link": null
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "widget": {
+ "name": "resolution"
+ },
+ "link": 424
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 443
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": []
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2DecodeLatents"
+ },
+ "widgets_values": [
+ 0,
+ true
+ ]
+ },
+ {
+ "id": 224,
+ "type": "Trellis2MeshWithVoxelToTrimesh",
+ "pos": [
+ 1658.7016089575882,
+ -75.3276646144487
+ ],
+ "size": [
+ 349.41171875,
+ 58
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 458
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 459
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ },
+ "widgets_values": [
+ "90 degrees"
+ ]
+ },
+ {
+ "id": 215,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1668.2431606477028,
+ -224.8516122407264
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 436
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 458
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 216,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1657.2258628678205,
+ -436.17651468161654
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 437
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 451
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 436
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 223,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ 1660.8568735364302,
+ 90.69364157484758
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 459
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 456
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 455
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ }
+ ],
+ "links": [
+ [
+ 241,
+ 69,
+ 2,
+ 119,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 423,
+ 214,
+ 0,
+ 209,
+ 1,
+ "SHAPE_SLAT"
+ ],
+ [
+ 424,
+ 214,
+ 1,
+ 209,
+ 3,
+ "INT"
+ ],
+ [
+ 425,
+ 220,
+ 0,
+ 211,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 426,
+ 208,
+ 2,
+ 212,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 427,
+ 208,
+ 0,
+ 212,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 428,
+ 212,
+ 2,
+ 213,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 429,
+ 208,
+ 0,
+ 213,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 430,
+ 212,
+ 0,
+ 213,
+ 2,
+ "COORDS"
+ ],
+ [
+ 431,
+ 213,
+ 2,
+ 214,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 432,
+ 208,
+ 1,
+ 214,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 433,
+ 213,
+ 0,
+ 214,
+ 2,
+ "SHAPE_SLAT"
+ ],
+ [
+ 434,
+ 213,
+ 1,
+ 214,
+ 3,
+ "INT"
+ ],
+ [
+ 435,
+ 212,
+ 1,
+ 214,
+ 4,
+ "INT"
+ ],
+ [
+ 436,
+ 216,
+ 0,
+ 215,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 437,
+ 211,
+ 0,
+ 216,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 443,
+ 209,
+ 0,
+ 220,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 444,
+ 39,
+ 0,
+ 208,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 445,
+ 119,
+ 0,
+ 208,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 451,
+ 203,
+ 0,
+ 216,
+ 1,
+ "INT"
+ ],
+ [
+ 455,
+ 223,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 456,
+ 204,
+ 0,
+ 223,
+ 1,
+ "STRING"
+ ],
+ [
+ 457,
+ 214,
+ 2,
+ 209,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 458,
+ 215,
+ 0,
+ 224,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 459,
+ 224,
+ 0,
+ 223,
+ 0,
+ "TRIMESH"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -387.72410571388025,
+ -53.37077389343578,
+ 737.521556382955,
+ 1307.389216328689
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 375.7151992875252,
+ -957.7197012700683,
+ 1860.9819191012211,
+ 1225.6071509713981
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "config": {},
+ "extra": {
+ "workflowRendererVersion": "LG",
+ "ue_links": [],
+ "ds": {
+ "scale": 0.5131581182307072,
+ "offset": [
+ 698.5290381042139,
+ 1125.1909919152047
+ ]
+ },
+ "links_added_by_ue": [],
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/MultiViews.json b/example_workflows/MultiViews.json
index 4882eb8..a40a9d3 100644
--- a/example_workflows/MultiViews.json
+++ b/example_workflows/MultiViews.json
@@ -1,51 +1,28 @@
{
- "id": "cb2e6635-37a0-47a0-bf98-1cfad1b842c2",
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
"revision": 0,
- "last_node_id": 16,
- "last_link_id": 19,
+ "last_node_id": 243,
+ "last_link_id": 494,
"nodes": [
{
- "id": 1,
- "type": "Trellis2MeshWithVoxelMultiViewGenerator",
+ "id": 227,
+ "type": "Trellis2ReconstructMeshWithQuad",
"pos": [
- 1160.6611250957358,
- 1551.885281893709
+ 1176.3670349681408,
+ -707.7757475998008
],
"size": [
- 612.71875,
- 972.65625
+ 331.5878996659427,
+ 142.11360851901668
],
"flags": {},
- "order": 5,
+ "order": 13,
"mode": 0,
"inputs": [
{
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
- "link": 8
- },
- {
- "name": "front_image",
- "type": "IMAGE",
- "link": 2
- },
- {
- "name": "back_image",
- "shape": 7,
- "type": "IMAGE",
- "link": 10
- },
- {
- "name": "left_image",
- "shape": 7,
- "type": "IMAGE",
- "link": null
- },
- {
- "name": "right_image",
- "shape": 7,
- "type": "IMAGE",
- "link": null
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 470
}
],
"outputs": [
@@ -53,63 +30,120 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 16
- ]
- },
- {
- "name": "bvh",
- "type": "BVH",
- "links": [
- 17
+ 472
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
"widget_ue_connectable": {}
},
"widgets_values": [
- 12345,
- "fixed",
- "1024_cascade",
- 25,
- 6.5,
- 0.2,
- 4,
- 25,
- 6.5,
- 0.2,
- 4,
- 25,
- 3,
- 0.2,
- 3,
- 999999,
- 32,
- true,
- 0.1,
1,
- 0.1,
- 1,
- 0,
- 0.9,
+ 1024,
true,
- "z",
- 2
+ true
]
},
{
- "id": 3,
- "type": "Trellis2LoadImageWithTransparency",
+ "id": 231,
+ "type": "Trellis2FillHolesWithCuMesh",
"pos": [
- 189.80670439097503,
- 1011.8182611411379
+ 1189.269979801219,
+ -832.1907924780478
],
"size": [
- 453.4375,
- 837.1875
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 487
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 470
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 229,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1181.0098963429245,
+ -503.10569576014126
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 472
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 475
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 471
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -840.0731525127071,
+ -237.35696845568904
+ ],
+ "size": [
+ 270,
+ 82
],
"flags": {},
"order": 0,
@@ -117,85 +151,34 @@
"inputs": [],
"outputs": [
{
- "name": "image",
- "type": "IMAGE",
- "links": null
- },
- {
- "name": "mask",
- "type": "MASK",
- "links": null
- },
- {
- "name": "image_with_alpha",
- "type": "IMAGE",
+ "name": "INT",
+ "type": "INT",
"links": [
- 1
+ 475
]
}
],
+ "title": "Target Face Number",
"properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2LoadImageWithTransparency",
- "widget_ue_connectable": {}
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
},
"widgets_values": [
- "Image_500_00001_.png",
- "image"
+ 300000,
+ "fixed"
]
},
{
- "id": 4,
- "type": "Trellis2PreProcessImage",
+ "id": 204,
+ "type": "PrimitiveString",
"pos": [
- 739.4753259488509,
- 1571.2899959426204
+ -848.2606713054529,
+ -370.80161968053943
],
"size": [
- 349.09375,
- 107.328125
- ],
- "flags": {},
- "order": 3,
- "mode": 0,
- "inputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "link": 1
- }
- ],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": [
- 2
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2PreProcessImage",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- 25,
- false
- ]
- },
- {
- "id": 10,
- "type": "Trellis2LoadModel",
- "pos": [
- 720.4210852610131,
- 1268.6993659791224
- ],
- "size": [
- 355.5,
- 207.328125
+ 270,
+ 58
],
"flags": {},
"order": 1,
@@ -203,204 +186,50 @@
"inputs": [],
"outputs": [
{
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
+ "name": "STRING",
+ "type": "STRING",
"links": [
- 8
+ 479
]
}
],
+ "title": "Name",
"properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2LoadModel",
- "widget_ue_connectable": {}
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
},
"widgets_values": [
- "TRELLIS.2-4B",
- "flash_attn",
- "cuda",
- true,
- false
+ "Tank"
]
},
{
- "id": 11,
- "type": "Trellis2LoadImageWithTransparency",
- "pos": [
- 196.16440063611725,
- 1894.1248321979997
- ],
- "size": [
- 443.6875,
- 849.5
- ],
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": null
- },
- {
- "name": "mask",
- "type": "MASK",
- "links": null
- },
- {
- "name": "image_with_alpha",
- "type": "IMAGE",
- "links": [
- 9
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2LoadImageWithTransparency",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- "Image_480_00001_.png",
- "image"
- ]
- },
- {
- "id": 12,
- "type": "Trellis2PreProcessImage",
- "pos": [
- 739.1482171781413,
- 1738.5360038621288
- ],
- "size": [
- 354.203125,
- 107.328125
- ],
- "flags": {},
- "order": 4,
- "mode": 0,
- "inputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "link": 9
- }
- ],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": [
- 10
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "Node name for S&R": "Trellis2PreProcessImage",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- 25,
- false
- ]
- },
- {
- "id": 14,
- "type": "Trellis2PostProcessAndUnWrapAndRasterizer",
- "pos": [
- 1823.7677651122783,
- 1551.9434773806804
- ],
- "size": [
- 561.796875,
- 674.578125
- ],
- "flags": {},
- "order": 6,
- "mode": 0,
- "inputs": [
- {
- "name": "mesh",
- "type": "MESHWITHVOXEL",
- "link": 16
- },
- {
- "name": "bvh",
- "type": "BVH",
- "link": 17
- }
- ],
- "outputs": [
- {
- "name": "trimesh",
- "type": "TRIMESH",
- "links": [
- 18
- ]
- },
- {
- "name": "base_color_texture",
- "type": "IMAGE",
- "links": null
- },
- {
- "name": "metallic_roughness_texture",
- "type": "IMAGE",
- "links": null
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer"
- },
- "widgets_values": [
- 60,
- 0,
- 1,
- 1,
- 4096,
- true,
- 1,
- 0,
- 500000,
- "Cumesh",
- true,
- "OPAQUE",
- "1024",
- false,
- true,
- false,
- false,
- true
- ]
- },
- {
- "id": 15,
+ "id": 233,
"type": "Trellis2ExportMesh",
"pos": [
- 2476.893251656926,
- 1550.6756032446528
+ 1191.3255568422467,
+ -77.49101003118169
],
"size": [
- 329.8125,
- 148.75
+ 270,
+ 102
],
"flags": {},
- "order": 7,
+ "order": 17,
"mode": 0,
"inputs": [
{
"name": "trimesh",
"type": "TRIMESH",
- "link": 18
+ "link": 494
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 479
}
],
"outputs": [
@@ -408,35 +237,254 @@
"name": "glb_path",
"type": "STRING",
"links": [
- 19
+ 480
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 391.3481025606576,
+ -777.5583507484255
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 483
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2ExportMesh"
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
},
"widgets_values": [
- "Trellis2MV",
- "glb",
- true
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
]
},
{
- "id": 16,
- "type": "Preview3D",
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
"pos": [
- 2420.5741031607986,
- 1763.4314920895954
+ 379.7979564010223,
+ -519.6169765183606
],
"size": [
- 842.90625,
- 949.015625
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 484
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 239,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 386.64212656289783,
+ -361.6545116326391
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 485
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 486
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 240,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 384.7958133512735,
+ -204.2542485606769
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 488
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 489
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 241,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 384.795813351274,
+ -50.03583437610661
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 490
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 491
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 360.99306605686377,
+ 121.97109784511967
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
],
"flags": {},
- "order": 8,
+ "order": 18,
"mode": 0,
"inputs": [
{
@@ -457,110 +505,601 @@
"widget": {
"name": "model_file"
},
- "link": 19
+ "link": 480
}
],
"outputs": [],
"properties": {
"cnr_id": "comfy-core",
- "ver": "0.13.0",
- "widget_ue_connectable": {},
+ "ver": "0.18.1",
"Node name for S&R": "Preview3D"
},
"widgets_values": [
"",
""
]
+ },
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -847.5380090037122,
+ -82.9715880857893
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "title": "Front Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 236,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -258.0644919609741,
+ -81.59760778432084
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 485
+ ]
+ }
+ ],
+ "title": "Back Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 237,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -834.3440812871166,
+ 489.10851591788764
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 488
+ ]
+ }
+ ],
+ "title": "Left Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 238,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -256.71749538654217,
+ 489.176741691081
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 490
+ ]
+ }
+ ],
+ "title": "Right Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 228,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1184.3251213630779,
+ -362.82432619683135
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 471
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 492
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 235,
+ "type": "Trellis2MeshWithVoxelMultiViewGenerator",
+ "pos": [
+ 730.5993372134623,
+ -806.3203136622369
+ ],
+ "size": [
+ 416.855078125,
+ 786
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 483
+ },
+ {
+ "name": "front_image",
+ "type": "IMAGE",
+ "link": 484
+ },
+ {
+ "name": "back_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 486
+ },
+ {
+ "name": "left_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 489
+ },
+ {
+ "name": "right_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 491
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 487
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": [
+ 493
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ "1024_cascade",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ 12,
+ 3,
+ 0.01,
+ 3,
+ 999999,
+ 32,
+ true,
+ 0.1,
+ 1,
+ 0.1,
+ 1,
+ 0,
+ 0.9,
+ true,
+ "z",
+ 1,
+ "heun"
+ ]
+ },
+ {
+ "id": 242,
+ "type": "Trellis2UnWrapAndRasterizer",
+ "pos": [
+ 1496.5934369390998,
+ -290.13985378139296
+ ],
+ "size": [
+ 419.15234375,
+ 314
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 492
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "link": 493
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 494
+ ]
+ },
+ {
+ "name": "base_color_texture",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "metallic_roughness_texture",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2UnWrapAndRasterizer"
+ },
+ "widgets_values": [
+ 60,
+ 0,
+ 1,
+ 1,
+ 2048,
+ "OPAQUE",
+ false,
+ false,
+ false,
+ "telea"
+ ]
}
],
"links": [
[
- 1,
- 3,
+ 241,
+ 69,
2,
- 4,
+ 119,
0,
"IMAGE"
],
[
- 2,
- 4,
+ 470,
+ 231,
0,
- 1,
- 1,
- "IMAGE"
- ],
- [
- 8,
- 10,
- 0,
- 1,
- 0,
- "TRELLIS2PIPELINE"
- ],
- [
- 9,
- 11,
- 2,
- 12,
- 0,
- "IMAGE"
- ],
- [
- 10,
- 12,
- 0,
- 1,
- 2,
- "IMAGE"
- ],
- [
- 16,
- 1,
- 0,
- 14,
+ 227,
0,
"MESHWITHVOXEL"
],
[
- 17,
+ 471,
+ 229,
+ 0,
+ 228,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 472,
+ 227,
+ 0,
+ 229,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 475,
+ 203,
+ 0,
+ 229,
1,
+ "INT"
+ ],
+ [
+ 479,
+ 204,
+ 0,
+ 233,
1,
- 14,
+ "STRING"
+ ],
+ [
+ 480,
+ 233,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 483,
+ 39,
+ 0,
+ 235,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 484,
+ 119,
+ 0,
+ 235,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 485,
+ 236,
+ 2,
+ 239,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 486,
+ 239,
+ 0,
+ 235,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 487,
+ 235,
+ 0,
+ 231,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 488,
+ 237,
+ 2,
+ 240,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 489,
+ 240,
+ 0,
+ 235,
+ 3,
+ "IMAGE"
+ ],
+ [
+ 490,
+ 238,
+ 2,
+ 241,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 491,
+ 241,
+ 0,
+ 235,
+ 4,
+ "IMAGE"
+ ],
+ [
+ 492,
+ 228,
+ 0,
+ 242,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 493,
+ 235,
+ 1,
+ 242,
1,
"BVH"
],
[
- 18,
- 14,
+ 494,
+ 242,
0,
- 15,
+ 233,
0,
"TRIMESH"
- ],
- [
- 19,
- 15,
- 0,
- 16,
- 2,
- "STRING"
]
],
- "groups": [],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -874.7851579398082,
+ -458.57410768053984,
+ 1193.2585052367917,
+ 1492.7062709864938
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 360.7008822276498,
+ -937.193468065821,
+ 1585.9419620118415,
+ 1014.8225405559199
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
"config": {},
"extra": {
+ "workflowRendererVersion": "LG",
"ue_links": [],
"ds": {
- "scale": 0.45020114458154903,
+ "scale": 0.5644739300537791,
"offset": [
- 307.1190043799142,
- -866.3604306862078
+ 1098.7675999602163,
+ 1065.1675662022888
]
},
- "workflowRendererVersion": "Vue",
"links_added_by_ue": [],
- "frontendVersion": "1.38.13",
+ "frontendVersion": "1.42.8",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
diff --git a/example_workflows/MultiViews_MeshOnly.json b/example_workflows/MultiViews_MeshOnly.json
index 308afc3..202b912 100644
--- a/example_workflows/MultiViews_MeshOnly.json
+++ b/example_workflows/MultiViews_MeshOnly.json
@@ -1,156 +1,28 @@
{
- "id": "440427ce-0c6a-4462-af51-639ed2f16dec",
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
"revision": 0,
- "last_node_id": 121,
- "last_link_id": 230,
+ "last_node_id": 241,
+ "last_link_id": 491,
"nodes": [
{
- "id": 50,
- "type": "Trellis2PreProcessImage",
+ "id": 227,
+ "type": "Trellis2ReconstructMeshWithQuad",
"pos": [
- -514.210025695023,
- 250.92630997998222
+ 1176.3670349681408,
+ -707.7757475998008
],
"size": [
- 297.84375,
- 107.328125
+ 331.5878996659427,
+ 142.11360851901668
],
"flags": {},
- "order": 3,
- "mode": 0,
- "inputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "link": 110
- }
- ],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": [
- 223
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "15854282a73cf231b81d52ada22e652f414a078b",
- "Node name for S&R": "Trellis2PreProcessImage",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- 25,
- false
- ]
- },
- {
- "id": 6,
- "type": "Trellis2LoadImageWithTransparency",
- "pos": [
- -1354.11041692169,
- -103.28960088005408
- ],
- "size": [
- 610.640625,
- 788.515625
- ],
- "flags": {},
- "order": 0,
- "mode": 0,
- "inputs": [],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": null
- },
- {
- "name": "mask",
- "type": "MASK",
- "links": null
- },
- {
- "name": "image_with_alpha",
- "type": "IMAGE",
- "links": [
- 110
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
- "Node name for S&R": "Trellis2LoadImageWithTransparency",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- "Image_1024_00010_.png",
- "image"
- ]
- },
- {
- "id": 39,
- "type": "Trellis2LoadModel",
- "pos": [
- -502.8470854177922,
- -74.32512914672895
- ],
- "size": [
- 336.0625,
- 207.328125
- ],
- "flags": {
- "collapsed": false
- },
- "order": 1,
- "mode": 0,
- "inputs": [],
- "outputs": [
- {
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
- "links": [
- 222
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
- "Node name for S&R": "Trellis2LoadModel",
- "widget_ue_connectable": {}
- },
- "widgets_values": [
- "TRELLIS.2-4B",
- "flash_attn",
- "cuda",
- true,
- false
- ]
- },
- {
- "id": 103,
- "type": "Trellis2RemeshWithQuad",
- "pos": [
- 467.2386692680435,
- 243.87402592407818
- ],
- "size": [
- 455.3125,
- 262
- ],
- "flags": {
- "collapsed": false
- },
- "order": 6,
+ "order": 13,
"mode": 0,
"inputs": [
{
"name": "mesh",
"type": "MESHWITHVOXEL",
- "link": 225
+ "link": 470
}
],
"outputs": [
@@ -158,158 +30,42 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 226
+ 472
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "258cd607667d64b01b2abdaded4e59016ffd5cb6",
- "Node name for S&R": "Trellis2RemeshWithQuad",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
"widget_ue_connectable": {}
},
"widgets_values": [
1,
- 0,
- false,
- 0.03,
- "1024",
+ 1024,
true,
true
]
},
{
- "id": 114,
- "type": "Trellis2LoadImageWithTransparency",
+ "id": 231,
+ "type": "Trellis2FillHolesWithCuMesh",
"pos": [
- -1359.0614927706545,
- 735.4082082769835
+ 1189.269979801219,
+ -832.1907924780478
],
"size": [
- 609.125,
- 759.125
- ],
- "flags": {
- "collapsed": false
- },
- "order": 2,
- "mode": 0,
- "inputs": [],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": []
- },
- {
- "name": "mask",
- "type": "MASK",
- "links": []
- },
- {
- "name": "image_with_alpha",
- "type": "IMAGE",
- "links": [
- 221
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2LoadImageWithTransparency"
- },
- "widgets_values": [
- "Image_1024_00021_.png",
- "image"
- ]
- },
- {
- "id": 115,
- "type": "Trellis2PreProcessImage",
- "pos": [
- -520.4300253912838,
- 427.5271012938298
- ],
- "size": [
- 311.921875,
- 107.328125
- ],
- "flags": {
- "collapsed": false
- },
- "order": 4,
- "mode": 0,
- "inputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "link": 221
- }
- ],
- "outputs": [
- {
- "name": "image",
- "type": "IMAGE",
- "links": [
- 224
- ]
- }
- ],
- "properties": {
- "aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "15854282a73cf231b81d52ada22e652f414a078b",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2PreProcessImage"
- },
- "widgets_values": [
- 50,
- false
- ]
- },
- {
- "id": 116,
- "type": "Trellis2MeshWithVoxelMultiViewGenerator",
- "pos": [
- -76.41514399946482,
- 241.60506031220336
- ],
- "size": [
- 468.21875,
- 1000.53125
+ 312.4361328125,
+ 58
],
"flags": {},
- "order": 5,
+ "order": 12,
"mode": 0,
"inputs": [
{
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
- "link": 222
- },
- {
- "name": "front_image",
- "type": "IMAGE",
- "link": 223
- },
- {
- "name": "back_image",
- "shape": 7,
- "type": "IMAGE",
- "link": 224
- },
- {
- "name": "left_image",
- "shape": 7,
- "type": "IMAGE",
- "link": null
- },
- {
- "name": "right_image",
- "shape": 7,
- "type": "IMAGE",
- "link": null
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 487
}
],
"outputs": [
@@ -317,70 +73,46 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 225
+ 470
]
- },
- {
- "name": "bvh",
- "type": "BVH",
- "links": null
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator"
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
},
"widgets_values": [
- 12345,
- "randomize",
- "1024_cascade",
- 25,
- 6.5,
- 0.2,
- 4,
- 25,
- 6.5,
- 0.2,
- 4,
- 12,
- 3,
- 0.2,
- 3,
- 999999,
- 32,
- false,
- 0.1,
- 1,
- 0.1,
- 1,
- 0,
- 0.9,
- true,
- "z",
- 2
+ 1
]
},
{
- "id": 117,
+ "id": 229,
"type": "Trellis2SimplifyMesh",
"pos": [
- 472.141197337784,
- 564.9329628175803
+ 1181.0098963429245,
+ -503.10569576014126
],
"size": [
- 317,
- 115.046875
+ 270,
+ 82
],
"flags": {},
- "order": 7,
+ "order": 14,
"mode": 0,
"inputs": [
{
"name": "mesh",
"type": "MESHWITHVOXEL",
- "link": 226
+ "link": 472
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 475
}
],
"outputs": [
@@ -388,14 +120,13 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 227
+ 471
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
"Node name for S&R": "Trellis2SimplifyMesh"
},
"widgets_values": [
@@ -404,24 +135,93 @@
]
},
{
- "id": 118,
- "type": "Trellis2FillHolesWithMeshlib",
+ "id": 203,
+ "type": "PrimitiveInt",
"pos": [
- 473.1986723221411,
- 743.6571759057757
+ -840.0731525127071,
+ -237.35696845568904
],
"size": [
- 348.71875,
- 71.328125
+ 270,
+ 82
],
"flags": {},
- "order": 8,
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 475
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -848.2606713054529,
+ -370.80161968053943
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 479
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 228,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1184.3251213630779,
+ -362.82432619683135
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 15,
"mode": 0,
"inputs": [
{
"name": "mesh",
"type": "MESHWITHVOXEL",
- "link": 227
+ "link": 471
}
],
"outputs": [
@@ -429,7 +229,7 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 228
+ 481
]
},
{
@@ -440,31 +240,38 @@
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
"Node name for S&R": "Trellis2FillHolesWithMeshlib"
},
"widgets_values": []
},
{
- "id": 119,
+ "id": 233,
"type": "Trellis2ExportMesh",
"pos": [
- 471.08304460897625,
- 1037.6532301953084
+ 1191.3255568422467,
+ -77.49101003118169
],
"size": [
- 348.71875,
- 148.109375
+ 270,
+ 102
],
"flags": {},
- "order": 10,
+ "order": 17,
"mode": 0,
"inputs": [
{
"name": "trimesh",
"type": "TRIMESH",
- "link": 229
+ "link": 482
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 479
}
],
"outputs": [
@@ -472,75 +279,254 @@
"name": "glb_path",
"type": "STRING",
"links": [
- 230
+ 480
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 391.3481025606576,
+ -777.5583507484255
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 483
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2ExportMesh"
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
},
"widgets_values": [
- "Trellis2MV",
- "glb",
- true
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
]
},
{
- "id": 120,
- "type": "Trellis2MeshWithVoxelToTrimesh",
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
"pos": [
- 472.14061641396916,
- 874.7920616685096
+ 379.7979564010223,
+ -519.6169765183606
],
"size": [
- 332.859375,
- 92.5625
+ 281.8229166666667,
+ 106
],
- "flags": {},
- "order": 9,
+ "flags": {
+ "collapsed": false
+ },
+ "order": 7,
"mode": 0,
"inputs": [
{
- "name": "mesh",
- "type": "MESHWITHVOXEL",
- "link": 228
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
}
],
"outputs": [
{
- "name": "trimesh",
- "type": "TRIMESH",
+ "name": "image",
+ "type": "IMAGE",
"links": [
- 229
+ 484
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a19110a28a5c5c434386af0421005dd8edae82db",
- "widget_ue_connectable": {},
- "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
},
"widgets_values": [
- "90 degrees"
+ 10,
+ false,
+ 1024
]
},
{
- "id": 121,
- "type": "Preview3D",
+ "id": 239,
+ "type": "Trellis2PreProcessImage",
"pos": [
- 960.7538631636339,
- 252.95888156302374
+ 386.64212656289783,
+ -361.6545116326391
],
"size": [
- 892.296875,
- 936.96875
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 485
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 486
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 240,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 384.7958133512735,
+ -204.2542485606769
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 488
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 489
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 241,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 384.795813351274,
+ -50.03583437610661
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 490
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 491
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 360.99306605686377,
+ 121.97109784511967
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
],
"flags": {},
- "order": 11,
+ "order": 18,
"mode": 0,
"inputs": [
{
@@ -561,126 +547,525 @@
"widget": {
"name": "model_file"
},
- "link": 230
+ "link": 480
}
],
"outputs": [],
"properties": {
"cnr_id": "comfy-core",
- "ver": "0.13.0",
- "widget_ue_connectable": {},
+ "ver": "0.18.1",
"Node name for S&R": "Preview3D"
},
"widgets_values": [
"",
""
]
+ },
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -847.5380090037122,
+ -82.9715880857893
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "title": "Front Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 236,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -258.0644919609741,
+ -81.59760778432084
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 485
+ ]
+ }
+ ],
+ "title": "Back Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 237,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -834.3440812871166,
+ 489.10851591788764
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 488
+ ]
+ }
+ ],
+ "title": "Left Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 238,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -256.71749538654217,
+ 489.176741691081
+ ],
+ "size": [
+ 496.4332376519267,
+ 498.75383455750784
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 490
+ ]
+ }
+ ],
+ "title": "Right Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 235,
+ "type": "Trellis2MeshWithVoxelMultiViewGenerator",
+ "pos": [
+ 730.5993372134623,
+ -806.3203136622369
+ ],
+ "size": [
+ 416.855078125,
+ 786
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 483
+ },
+ {
+ "name": "front_image",
+ "type": "IMAGE",
+ "link": 484
+ },
+ {
+ "name": "back_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 486
+ },
+ {
+ "name": "left_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 489
+ },
+ {
+ "name": "right_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 491
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 487
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ "1024_cascade",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ 12,
+ 3,
+ 0.01,
+ 3,
+ 999999,
+ 32,
+ false,
+ 0.1,
+ 1,
+ 0.1,
+ 1,
+ 0,
+ 0.9,
+ true,
+ "z",
+ 1,
+ "heun"
+ ]
+ },
+ {
+ "id": 234,
+ "type": "Trellis2MeshWithVoxelToTrimesh",
+ "pos": [
+ 1171.5679242041226,
+ -219.6291722499163
+ ],
+ "size": [
+ 349.41171875,
+ 58
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 481
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 482
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ },
+ "widgets_values": [
+ "90 degrees"
+ ]
}
],
"links": [
[
- 110,
- 6,
+ 241,
+ 69,
2,
- 50,
- 0,
- "IMAGE"
- ],
- [
- 221,
- 114,
- 2,
- 115,
- 0,
- "IMAGE"
- ],
- [
- 222,
- 39,
- 0,
- 116,
- 0,
- "TRELLIS2PIPELINE"
- ],
- [
- 223,
- 50,
- 0,
- 116,
- 1,
- "IMAGE"
- ],
- [
- 224,
- 115,
- 0,
- 116,
- 2,
- "IMAGE"
- ],
- [
- 225,
- 116,
- 0,
- 103,
- 0,
- "MESHWITHVOXEL"
- ],
- [
- 226,
- 103,
- 0,
- 117,
- 0,
- "MESHWITHVOXEL"
- ],
- [
- 227,
- 117,
- 0,
- 118,
- 0,
- "MESHWITHVOXEL"
- ],
- [
- 228,
- 118,
- 0,
- 120,
- 0,
- "MESHWITHVOXEL"
- ],
- [
- 229,
- 120,
- 0,
119,
0,
+ "IMAGE"
+ ],
+ [
+ 470,
+ 231,
+ 0,
+ 227,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 471,
+ 229,
+ 0,
+ 228,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 472,
+ 227,
+ 0,
+ 229,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 475,
+ 203,
+ 0,
+ 229,
+ 1,
+ "INT"
+ ],
+ [
+ 479,
+ 204,
+ 0,
+ 233,
+ 1,
+ "STRING"
+ ],
+ [
+ 480,
+ 233,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 481,
+ 228,
+ 0,
+ 234,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 482,
+ 234,
+ 0,
+ 233,
+ 0,
"TRIMESH"
],
[
- 230,
+ 483,
+ 39,
+ 0,
+ 235,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 484,
119,
0,
- 121,
+ 235,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 485,
+ 236,
2,
- "STRING"
+ 239,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 486,
+ 239,
+ 0,
+ 235,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 487,
+ 235,
+ 0,
+ 231,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 488,
+ 237,
+ 2,
+ 240,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 489,
+ 240,
+ 0,
+ 235,
+ 3,
+ "IMAGE"
+ ],
+ [
+ 490,
+ 238,
+ 2,
+ 241,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 491,
+ 241,
+ 0,
+ 235,
+ 4,
+ "IMAGE"
]
],
- "groups": [],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -874.7851579398082,
+ -458.57410768053984,
+ 1193.2585052367917,
+ 1492.7062709864938
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 360.7008822276498,
+ -937.193468065821,
+ 1227.3780156118423,
+ 1005.7286067160883
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
"config": {},
"extra": {
- "workflowRendererVersion": "Vue",
+ "workflowRendererVersion": "LG",
"ue_links": [],
"ds": {
- "scale": 0.4736244074476824,
+ "scale": 0.513158118230708,
"offset": [
- 1666.544822537185,
- 301.4636979965617
+ 1358.4460295288495,
+ 1222.5907408893986
]
},
"links_added_by_ue": [],
- "frontendVersion": "1.38.13",
+ "frontendVersion": "1.42.8",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
diff --git a/example_workflows/Projection_6Views_Hy20.json b/example_workflows/Projection_6Views_Hy20.json
new file mode 100644
index 0000000..b6e70e1
--- /dev/null
+++ b/example_workflows/Projection_6Views_Hy20.json
@@ -0,0 +1,8578 @@
+{
+ "id": "d8f6c381-6a87-4599-b2c1-0b140608c3ce",
+ "revision": 0,
+ "last_node_id": 1620,
+ "last_link_id": 2048,
+ "nodes": [
+ {
+ "id": 419,
+ "type": "SetNode",
+ "pos": [
+ -745.6584936204468,
+ -1151.5145189856826
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 39,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "TRELLIS2PIPELINE",
+ "type": "TRELLIS2PIPELINE",
+ "link": 638
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_Trellis2Pipeline",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "Trellis2Pipeline"
+ },
+ "widgets_values": [
+ "Trellis2Pipeline"
+ ]
+ },
+ {
+ "id": 418,
+ "type": "SetNode",
+ "pos": [
+ -746.8312372724462,
+ -1195.6482856970033
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 38,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "TRIMESH",
+ "type": "TRIMESH",
+ "link": 637
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_WhiteMesh",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "WhiteMesh"
+ },
+ "widgets_values": [
+ "WhiteMesh"
+ ]
+ },
+ {
+ "id": 428,
+ "type": "GetNode",
+ "pos": [
+ -1314.190670687656,
+ -1187.3587574508188
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 658
+ ]
+ }
+ ],
+ "title": "Get_NormalImage",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "NormalImage"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 417,
+ "type": "GetNode",
+ "pos": [
+ -1310.2904954152943,
+ -1139.8984090506763
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 657
+ ]
+ }
+ ],
+ "title": "Get_Prefix",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 363,
+ "type": "GetNode",
+ "pos": [
+ -545.6263878998996,
+ 76.88764947131784
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 566
+ ]
+ }
+ ],
+ "title": "Get_Prefix",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 267,
+ "type": "StringConcatenate",
+ "pos": [
+ -377.11763534068075,
+ 75.22708582112858
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 27,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 566
+ }
+ ],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 480
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_Textured_MV",
+ ""
+ ]
+ },
+ {
+ "id": 429,
+ "type": "GetNode",
+ "pos": [
+ -1277.085676114816,
+ -573.0082510434125
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "TRELLIS2PIPELINE",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 1193
+ ]
+ }
+ ],
+ "title": "Get_Trellis2Pipeline",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Trellis2Pipeline"
+ ]
+ },
+ {
+ "id": 266,
+ "type": "Preview3D",
+ "pos": [
+ -183.09944061302795,
+ 64.39555879074847
+ ],
+ "size": [
+ 1171.9726831120965,
+ 1226.3422371338038
+ ],
+ "flags": {},
+ "order": 51,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 478
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "Preview3D",
+ "Last Time Model File": "C:/Git/ComfyUI/output/3D/Tank_4View_Hy20/Tank_4View_Hy20_Textured_MV_00002_.glb",
+ "Resource Folder": "Git/ComfyUI/output/3D/Tank_4View_Hy20",
+ "Scene Config": {
+ "showGrid": true,
+ "backgroundColor": "#282828",
+ "backgroundImage": "",
+ "backgroundRenderMode": "tiled"
+ },
+ "Camera Config": {
+ "cameraType": "perspective",
+ "fov": 35,
+ "state": {
+ "position": {
+ "x": 6.116219723662718,
+ "y": 2.598047886467756,
+ "z": 8.399990125907564
+ },
+ "target": {
+ "x": 0,
+ "y": 1.61200424353802,
+ "z": 0
+ },
+ "zoom": 1,
+ "cameraType": "perspective"
+ }
+ },
+ "Light Config": {
+ "intensity": 5
+ },
+ "Model Config": {
+ "upDirection": "original",
+ "materialMode": "original",
+ "showSkeleton": false
+ }
+ },
+ "widgets_values": [
+ "C:/Git/ComfyUI/output/3D/Tank_4View_Hy20/Tank_4View_Hy20_Textured_MV_00002_.glb",
+ ""
+ ]
+ },
+ {
+ "id": 259,
+ "type": "StringConcatenate",
+ "pos": [
+ -1767.7760695000165,
+ -1014.1318223903585
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 37,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "string_b",
+ "type": "STRING",
+ "widget": {
+ "name": "string_b"
+ },
+ "link": 464
+ }
+ ],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 466
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "3D/",
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 358,
+ "type": "SetNode",
+ "pos": [
+ -1588.7674516597126,
+ -1042.3594981549759
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 53,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "link": 561
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_Prefix",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "Prefix"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 260,
+ "type": "StringConcatenate",
+ "pos": [
+ -1592.981372429068,
+ -1002.3323455162318
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 50,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 466
+ },
+ {
+ "name": "string_b",
+ "type": "STRING",
+ "widget": {
+ "name": "string_b"
+ },
+ "link": 465
+ }
+ ],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 561
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "/",
+ "",
+ "/"
+ ]
+ },
+ {
+ "id": 361,
+ "type": "SetNode",
+ "pos": [
+ -1752.955634745735,
+ -780.3113555830523
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 33,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 564
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_TargetFaceNumber",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "TargetFaceNumber"
+ },
+ "widgets_values": [
+ "TargetFaceNumber"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 1477,
+ "type": "SetNode",
+ "pos": [
+ -1725.8334697028015,
+ -900.73932618393
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 29,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 1694
+ }
+ ],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "title": "Set_Seed",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "Seed"
+ },
+ "widgets_values": [
+ "Seed"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 436,
+ "type": "GetNode",
+ "pos": [
+ -1311.7255852250182,
+ -1044.7680632967733
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 671
+ ]
+ }
+ ],
+ "title": "Get_TargetFaceNumber",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "TargetFaceNumber"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 1478,
+ "type": "GetNode",
+ "pos": [
+ -1307.0333223276834,
+ -1098.482779160294
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 5,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 1695
+ ]
+ }
+ ],
+ "title": "Get_Seed",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Seed"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 432,
+ "type": "GetNode",
+ "pos": [
+ -1310.6835334127045,
+ -929.1947034503972
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 6,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 661
+ ]
+ }
+ ],
+ "title": "Get_Prefix",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 433,
+ "type": "GetNode",
+ "pos": [
+ -1317.8991439004014,
+ -883.1004178376324
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 7,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "TRIMESH",
+ "type": "TRIMESH",
+ "links": [
+ 663
+ ]
+ }
+ ],
+ "title": "Get_WhiteMesh",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "WhiteMesh"
+ ]
+ },
+ {
+ "id": 431,
+ "type": "GetNode",
+ "pos": [
+ -1291.2007962110185,
+ 75.26010684547481
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 8,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": [
+ 660
+ ]
+ }
+ ],
+ "title": "Get_TexturedMesh",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "TexturedMesh"
+ ]
+ },
+ {
+ "id": 465,
+ "type": "GetNode",
+ "pos": [
+ -1297.939277518252,
+ 192.06573820666307
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 9,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 745
+ ]
+ }
+ ],
+ "title": "Get_BackView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "BackView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 464,
+ "type": "GetNode",
+ "pos": [
+ -1305.6961669906457,
+ 124.51320291334166
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 10,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 744
+ ]
+ }
+ ],
+ "title": "Get_FrontView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "FrontView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 466,
+ "type": "GetNode",
+ "pos": [
+ -1308.041506018605,
+ 246.7257610683613
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 11,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1456
+ ]
+ }
+ ],
+ "title": "Get_LeftView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "LeftView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 467,
+ "type": "GetNode",
+ "pos": [
+ -1306.0647206290878,
+ 319.1279643724372
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 12,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1457
+ ]
+ }
+ ],
+ "title": "Get_RightView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "RightView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1476,
+ "type": "PrimitiveInt",
+ "pos": [
+ -2047.6461070066043,
+ -929.5690492212133
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 1694
+ ]
+ }
+ ],
+ "title": "Seed",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed"
+ ]
+ },
+ {
+ "id": 1597,
+ "type": "GetNode",
+ "pos": [
+ -1324.8554794577053,
+ -1442.423960365636
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 14,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1953
+ ]
+ }
+ ],
+ "title": "Get_SourceImage",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "SourceImage"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1598,
+ "type": "GetNode",
+ "pos": [
+ -1310.500842466565,
+ -1318.9820466669423
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 15,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 1955
+ ]
+ }
+ ],
+ "title": "Get_Seed",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "kijai/ComfyUI-KJNodes"
+ },
+ "widgets_values": [
+ "Seed"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 1599,
+ "type": "GetNode",
+ "pos": [
+ -1312.4383327980986,
+ -1379.7206754514782
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 16,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1954
+ ]
+ }
+ ],
+ "title": "Get_Prefix",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 1600,
+ "type": "SetNode",
+ "pos": [
+ -690.0219522319078,
+ -1367.462466851463
+ ],
+ "size": [
+ 210,
+ 50
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 41,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 1956
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "title": "Set_NormalImage",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "NormalImage"
+ },
+ "widgets_values": [
+ "NormalImage"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1601,
+ "type": "SetNode",
+ "pos": [
+ -706.4131438977086,
+ -1428.9185602519058
+ ],
+ "size": [
+ 232.4,
+ 50
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 40,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 1957
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "title": "Set_SourceImage_Processed",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "SourceImage_Processed"
+ },
+ "widgets_values": [
+ "SourceImage_Processed"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 435,
+ "type": "GetNode",
+ "pos": [
+ -1329.7904832679303,
+ -822.2830888092468
+ ],
+ "size": [
+ 233.862890625,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 17,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 669
+ ]
+ }
+ ],
+ "title": "Get_SourceImage_Processed",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "SourceImage_Processed"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 384,
+ "type": "GetNode",
+ "pos": [
+ -1294.3135844229482,
+ -513.6867133233288
+ ],
+ "size": [
+ 242.2095703125,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 18,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1972
+ ]
+ }
+ ],
+ "title": "Get_SourceImage_Processed",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "SourceImage_Processed"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 388,
+ "type": "GetNode",
+ "pos": [
+ -1281.730037434195,
+ -428.48056829609936
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 19,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "TRIMESH",
+ "type": "TRIMESH",
+ "links": [
+ 1192
+ ]
+ }
+ ],
+ "title": "Get_WhiteMesh",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "WhiteMesh"
+ ]
+ },
+ {
+ "id": 1481,
+ "type": "GetNode",
+ "pos": [
+ -1274.4385233484818,
+ -330.08116352707134
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 20,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 1702
+ ]
+ }
+ ],
+ "title": "Get_Seed",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Seed"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 1027,
+ "type": "SetNode",
+ "pos": [
+ -1583.8192421073943,
+ -711.6703867862403
+ ],
+ "size": [
+ 240.7466796875,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 36,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 2013
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "title": "Set_SourceImage",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "SourceImage"
+ },
+ "widgets_values": [
+ "SourceImage"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 209,
+ "type": "PrimitiveInt",
+ "pos": [
+ -2051.623505404044,
+ -793.4184503551801
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 21,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 564
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 1025,
+ "type": "Trellis2MeshTexturing",
+ "pos": [
+ -849.9080887367477,
+ -575.6484113326054
+ ],
+ "size": [
+ 419.15234375,
+ 506
+ ],
+ "flags": {},
+ "order": 32,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 1193
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1972
+ },
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 1192
+ },
+ {
+ "name": "seed",
+ "type": "INT",
+ "widget": {
+ "name": "seed"
+ },
+ "link": 1702
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 1188,
+ 1189
+ ]
+ },
+ {
+ "name": "base_color_texture",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "metallic_roughness_texture",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "41a6ddb2d48c408ead711846dbbfe587d4ff7102",
+ "Node name for S&R": "Trellis2MeshTexturing"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 12,
+ 3,
+ 0.01,
+ 3,
+ 1024,
+ 1024,
+ "OPAQUE",
+ false,
+ 0,
+ 0.9,
+ 1,
+ false,
+ false,
+ 60,
+ "heun",
+ "telea"
+ ]
+ },
+ {
+ "id": 1618,
+ "type": "SetNode",
+ "pos": [
+ -703.7275023648066,
+ -696.3227770155813
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 47,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 2041
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "title": "Set_BottomView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "kijai/ComfyUI-KJNodes",
+ "previousName": "BottomView"
+ },
+ "widgets_values": [
+ "BottomView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 422,
+ "type": "SetNode",
+ "pos": [
+ -711.2137626890823,
+ -933.2452185469294
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 42,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 680
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_FrontView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "FrontView"
+ },
+ "widgets_values": [
+ "FrontView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 423,
+ "type": "SetNode",
+ "pos": [
+ -702.1733471504639,
+ -885.4573667051945
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 43,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 681
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_LeftView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "LeftView"
+ },
+ "widgets_values": [
+ "LeftView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 424,
+ "type": "SetNode",
+ "pos": [
+ -707.5068883874511,
+ -840.4019794879641
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 44,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 682
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_BackView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "BackView"
+ },
+ "widgets_values": [
+ "BackView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 425,
+ "type": "SetNode",
+ "pos": [
+ -707.0424164713548,
+ -793.9808483962397
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 45,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 683
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_RightView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "RightView"
+ },
+ "widgets_values": [
+ "RightView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1617,
+ "type": "SetNode",
+ "pos": [
+ -698.1475594478394,
+ -749.7298950583769
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 46,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "link": 2040
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "title": "Set_TopView",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "kijai/ComfyUI-KJNodes",
+ "previousName": "TopView"
+ },
+ "widgets_values": [
+ "TopView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1619,
+ "type": "GetNode",
+ "pos": [
+ -1301.478855605668,
+ 388.23914525504057
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 22,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2042
+ ]
+ }
+ ],
+ "title": "Get_TopView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "kijai/ComfyUI-KJNodes"
+ },
+ "widgets_values": [
+ "TopView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 1620,
+ "type": "GetNode",
+ "pos": [
+ -1295.1019135179424,
+ 462.37137981688676
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 23,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2043
+ ]
+ }
+ ],
+ "title": "Get_BottomView",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "kijai/ComfyUI-KJNodes"
+ },
+ "widgets_values": [
+ "BottomView"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 360,
+ "type": "GetNode",
+ "pos": [
+ -363.8827042196567,
+ -585.5953164158891
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 24,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 563
+ ]
+ }
+ ],
+ "title": "Get_Prefix",
+ "properties": {
+ "Node name for S&R": "GetNode",
+ "aux_id": "GetNode"
+ },
+ "widgets_values": [
+ "Prefix"
+ ]
+ },
+ {
+ "id": 246,
+ "type": "StringConcatenate",
+ "pos": [
+ -196.5772097982741,
+ -581.9708677305606
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 35,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 563
+ }
+ ],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 447
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_Textured",
+ ""
+ ]
+ },
+ {
+ "id": 242,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ -193.40411998539903,
+ -508.7875293978828
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 49,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 1188
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 447
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 809
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "Tavern_Textured",
+ "glb"
+ ]
+ },
+ {
+ "id": 482,
+ "type": "Trellis2Continue",
+ "pos": [
+ -189.39491765604424,
+ -442.4017886046022
+ ],
+ "size": [
+ 161.33359375,
+ 46
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 52,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "input_1",
+ "type": "*",
+ "link": 1189
+ },
+ {
+ "name": "input_2",
+ "type": "*",
+ "link": 809
+ }
+ ],
+ "outputs": [
+ {
+ "name": "output_1",
+ "type": "*",
+ "links": [
+ 810
+ ]
+ },
+ {
+ "name": "output_2",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "b7ae30e26a4c8ab3e73d661668b80a8024465a40",
+ "Node name for S&R": "Trellis2Continue"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 389,
+ "type": "SetNode",
+ "pos": [
+ 25.12412753026314,
+ -441.9412931629978
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 54,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "link": 810
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_TexturedMesh",
+ "properties": {
+ "Node name for S&R": "SetNode",
+ "aux_id": "SetNode",
+ "previousName": "TexturedMesh"
+ },
+ "widgets_values": [
+ "TexturedMesh"
+ ]
+ },
+ {
+ "id": 265,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ -531.7753807563762,
+ 135.57688531430873
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 48,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 476
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 480
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 478
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "Test_v2",
+ "glb"
+ ]
+ },
+ {
+ "id": 264,
+ "type": "Trellis2MultiViewTexturing",
+ "pos": [
+ -974.3758998631852,
+ 75.46500664790204
+ ],
+ "size": [
+ 364.6066617624932,
+ 633.3109375310778
+ ],
+ "flags": {},
+ "order": 34,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 660
+ },
+ {
+ "name": "front_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 744
+ },
+ {
+ "name": "back_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 745
+ },
+ {
+ "name": "left_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 1456
+ },
+ {
+ "name": "right_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 1457
+ },
+ {
+ "name": "top_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 2042
+ },
+ {
+ "name": "bottom_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 2043
+ },
+ {
+ "name": "custom_images",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "camera_config",
+ "shape": 7,
+ "type": "HY3DCAMERA",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 476
+ ]
+ },
+ {
+ "name": "base_color",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "metallic_roughness",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2MultiViewTexturing"
+ },
+ "widgets_values": [
+ 1024,
+ true,
+ 1,
+ 1.2,
+ 1.15,
+ true,
+ 20,
+ false,
+ 0.01,
+ 1,
+ 1,
+ 0.5,
+ 0.5,
+ 0.1,
+ 0.1,
+ "",
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 416,
+ "type": "a7c4d8ae-5500-4f9b-9601-eafdb97e7a68",
+ "pos": [
+ -1058.861926280772,
+ -1190.3382989836646
+ ],
+ "size": [
+ 210,
+ 98
+ ],
+ "flags": {},
+ "order": 28,
+ "mode": 0,
+ "inputs": [
+ {
+ "label": "Normal_Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 658
+ },
+ {
+ "label": "Prefix",
+ "name": "",
+ "type": "*",
+ "link": 657
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "link": 671
+ },
+ {
+ "label": "seed",
+ "name": "_1",
+ "type": "*",
+ "link": 1695
+ }
+ ],
+ "outputs": [
+ {
+ "label": "white_mesh",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 637
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 638
+ ]
+ }
+ ],
+ "properties": {
+ "proxyWidgets": [],
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 239,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -2051.601088773679,
+ -652.112967692624
+ ],
+ "size": [
+ 627.638044195148,
+ 672.4931923313359
+ ],
+ "flags": {},
+ "order": 25,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 2013
+ ]
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": []
+ }
+ ],
+ "title": "Color Image",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency"
+ },
+ "widgets_values": [
+ "stylized_ogre_tavern_3d_game_asset_gray (2).jpeg",
+ "image"
+ ]
+ },
+ {
+ "id": 219,
+ "type": "PrimitiveString",
+ "pos": [
+ -2058.146657853287,
+ -1035.357925598092
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 26,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 464,
+ 465
+ ]
+ }
+ ],
+ "title": "Name -> output in 3D folder",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "OgreTavern_6View_Hy20"
+ ]
+ },
+ {
+ "id": 1596,
+ "type": "e62a2ed8-84aa-4855-9c23-e444bcf27791",
+ "pos": [
+ -1029.0444894493137,
+ -1435.6564661781129
+ ],
+ "size": [
+ 274.330078125,
+ 66
+ ],
+ "flags": {},
+ "order": 30,
+ "mode": 0,
+ "inputs": [
+ {
+ "label": "source_image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1953
+ },
+ {
+ "label": "prefix",
+ "name": "",
+ "type": "*",
+ "link": 1954
+ },
+ {
+ "label": "seed",
+ "name": "_1",
+ "type": "*",
+ "link": 1955
+ }
+ ],
+ "outputs": [
+ {
+ "label": "source_image",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1957
+ ]
+ },
+ {
+ "label": "normal_image_transparent",
+ "name": "IMAGE_1",
+ "type": "IMAGE",
+ "links": [
+ 1956
+ ]
+ }
+ ],
+ "properties": {
+ "proxyWidgets": [],
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 420,
+ "type": "bf1edf2f-ccb7-45ac-882e-55bdd5f0be86",
+ "pos": [
+ -1079.3175301918616,
+ -938.6000286674437
+ ],
+ "size": [
+ 210,
+ 148
+ ],
+ "flags": {},
+ "order": 31,
+ "mode": 0,
+ "inputs": [
+ {
+ "label": "prefix",
+ "name": "",
+ "type": "*",
+ "link": 661
+ },
+ {
+ "label": "white_mesh",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 663
+ },
+ {
+ "label": "color_image",
+ "name": "_1",
+ "type": "*",
+ "link": 669
+ }
+ ],
+ "outputs": [
+ {
+ "label": "FrontView",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 680
+ ]
+ },
+ {
+ "label": "LeftView",
+ "name": "IMAGE_1",
+ "type": "IMAGE",
+ "links": [
+ 681
+ ]
+ },
+ {
+ "label": "BackView",
+ "name": "IMAGE_2",
+ "type": "IMAGE",
+ "links": [
+ 682
+ ]
+ },
+ {
+ "label": "RightView",
+ "name": "IMAGE_3",
+ "type": "IMAGE",
+ "links": [
+ 683
+ ]
+ },
+ {
+ "label": "TopView",
+ "name": "IMAGE_4",
+ "type": "IMAGE",
+ "links": [
+ 2040
+ ]
+ },
+ {
+ "label": "BottomView",
+ "name": "IMAGE_5",
+ "type": "IMAGE",
+ "links": [
+ 2041
+ ]
+ }
+ ],
+ "properties": {
+ "proxyWidgets": [],
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4"
+ },
+ "widgets_values": []
+ }
+ ],
+ "links": [
+ [
+ 447,
+ 246,
+ 0,
+ 242,
+ 1,
+ "STRING"
+ ],
+ [
+ 464,
+ 219,
+ 0,
+ 259,
+ 0,
+ "STRING"
+ ],
+ [
+ 465,
+ 219,
+ 0,
+ 260,
+ 1,
+ "STRING"
+ ],
+ [
+ 466,
+ 259,
+ 0,
+ 260,
+ 0,
+ "STRING"
+ ],
+ [
+ 476,
+ 264,
+ 0,
+ 265,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 478,
+ 265,
+ 0,
+ 266,
+ 2,
+ "STRING"
+ ],
+ [
+ 480,
+ 267,
+ 0,
+ 265,
+ 1,
+ "STRING"
+ ],
+ [
+ 561,
+ 260,
+ 0,
+ 358,
+ 0,
+ "STRING"
+ ],
+ [
+ 563,
+ 360,
+ 0,
+ 246,
+ 0,
+ "STRING"
+ ],
+ [
+ 564,
+ 209,
+ 0,
+ 361,
+ 0,
+ "INT"
+ ],
+ [
+ 566,
+ 363,
+ 0,
+ 267,
+ 0,
+ "STRING"
+ ],
+ [
+ 637,
+ 416,
+ 0,
+ 418,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 638,
+ 416,
+ 1,
+ 419,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 657,
+ 417,
+ 0,
+ 416,
+ 1,
+ "STRING"
+ ],
+ [
+ 658,
+ 428,
+ 0,
+ 416,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 660,
+ 431,
+ 0,
+ 264,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 661,
+ 432,
+ 0,
+ 420,
+ 0,
+ "STRING"
+ ],
+ [
+ 663,
+ 433,
+ 0,
+ 420,
+ 1,
+ "TRIMESH"
+ ],
+ [
+ 669,
+ 435,
+ 0,
+ 420,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 671,
+ 436,
+ 0,
+ 416,
+ 2,
+ "INT"
+ ],
+ [
+ 680,
+ 420,
+ 0,
+ 422,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 681,
+ 420,
+ 1,
+ 423,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 682,
+ 420,
+ 2,
+ 424,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 683,
+ 420,
+ 3,
+ 425,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 744,
+ 464,
+ 0,
+ 264,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 745,
+ 465,
+ 0,
+ 264,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 809,
+ 242,
+ 0,
+ 482,
+ 1,
+ "STRING"
+ ],
+ [
+ 810,
+ 482,
+ 0,
+ 389,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 1188,
+ 1025,
+ 0,
+ 242,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 1189,
+ 1025,
+ 0,
+ 482,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 1192,
+ 388,
+ 0,
+ 1025,
+ 2,
+ "TRIMESH"
+ ],
+ [
+ 1193,
+ 429,
+ 0,
+ 1025,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 1456,
+ 466,
+ 0,
+ 264,
+ 3,
+ "IMAGE"
+ ],
+ [
+ 1457,
+ 467,
+ 0,
+ 264,
+ 4,
+ "IMAGE"
+ ],
+ [
+ 1694,
+ 1476,
+ 0,
+ 1477,
+ 0,
+ "INT"
+ ],
+ [
+ 1695,
+ 1478,
+ 0,
+ 416,
+ 3,
+ "INT"
+ ],
+ [
+ 1702,
+ 1481,
+ 0,
+ 1025,
+ 3,
+ "INT"
+ ],
+ [
+ 1953,
+ 1597,
+ 0,
+ 1596,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 1954,
+ 1599,
+ 0,
+ 1596,
+ 1,
+ "STRING"
+ ],
+ [
+ 1955,
+ 1598,
+ 0,
+ 1596,
+ 2,
+ "INT"
+ ],
+ [
+ 1956,
+ 1596,
+ 1,
+ 1600,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 1957,
+ 1596,
+ 0,
+ 1601,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 1972,
+ 384,
+ 0,
+ 1025,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 2013,
+ 239,
+ 0,
+ 1027,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 2040,
+ 420,
+ 4,
+ 1617,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 2041,
+ 420,
+ 5,
+ 1618,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 2042,
+ 1619,
+ 0,
+ 264,
+ 5,
+ "IMAGE"
+ ],
+ [
+ 2043,
+ 1620,
+ 0,
+ 264,
+ 6,
+ "IMAGE"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Texturing",
+ "bounding": [
+ -1331.067657728365,
+ -673.7818277278924,
+ 1553.6673346627758,
+ 624.0973323799652
+ ],
+ "color": "#A88",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 4,
+ "title": "Projection",
+ "bounding": [
+ -1330.3992143893013,
+ -23.42961664340983,
+ 2426.370430804953,
+ 1351.1024465541664
+ ],
+ "color": "#b58b2a",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 5,
+ "title": "Configuration",
+ "bounding": [
+ -2068.3222779518146,
+ -1128.221840574841,
+ 669.1786918496364,
+ 1186.8963044768402
+ ],
+ "color": "#b06634",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 8,
+ "title": "Mesh Generation",
+ "bounding": [
+ -1334.1541087377145,
+ -1274.931709235975,
+ 783.6469210441065,
+ 249.0021636156074
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 9,
+ "title": "MultiView Generation with Hy2.0",
+ "bounding": [
+ -1335.908517983496,
+ -1011.8851719087663,
+ 838.352169354825,
+ 319.25838307895276
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 32,
+ "title": "Normal Generation",
+ "bounding": [
+ -1334.8554794577053,
+ -1521.5982778096845,
+ 881.6773546690334,
+ 226.160224658576
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "definitions": {
+ "subgraphs": [
+ {
+ "id": "a7c4d8ae-5500-4f9b-9601-eafdb97e7a68",
+ "version": 1,
+ "state": {
+ "lastGroupId": 32,
+ "lastNodeId": 1620,
+ "lastLinkId": 2048,
+ "lastRerouteId": 0
+ },
+ "revision": 0,
+ "config": {},
+ "name": "Mesh Generation",
+ "inputNode": {
+ "id": -10,
+ "bounding": [
+ -1429.953594419078,
+ -690.030564252835,
+ 134.212890625,
+ 120
+ ]
+ },
+ "outputNode": {
+ "id": -20,
+ "bounding": [
+ 423.7021994103568,
+ -690.030564252835,
+ 120,
+ 80
+ ]
+ },
+ "inputs": [
+ {
+ "id": "8f27facf-c8f6-47f3-804e-db19f864d21f",
+ "name": "image",
+ "type": "IMAGE",
+ "linkIds": [
+ 631
+ ],
+ "label": "Normal_Image",
+ "pos": [
+ -1315.740703794078,
+ -670.030564252835
+ ]
+ },
+ {
+ "id": "ea5d96ce-c832-43b3-bf6b-63a09c722abe",
+ "name": "",
+ "type": "*",
+ "linkIds": [
+ 655
+ ],
+ "label": "Prefix",
+ "pos": [
+ -1315.740703794078,
+ -650.030564252835
+ ]
+ },
+ {
+ "id": "ae30eada-3ab4-491d-b92a-14a0627aaa34",
+ "name": "target_face_num",
+ "type": "INT",
+ "linkIds": [
+ 1964
+ ],
+ "pos": [
+ -1315.740703794078,
+ -630.030564252835
+ ]
+ },
+ {
+ "id": "f87e02a8-d1a3-48cd-b57c-5a1fa2c5d563",
+ "name": "_1",
+ "type": "*",
+ "linkIds": [
+ 1692
+ ],
+ "label": "seed",
+ "pos": [
+ -1315.740703794078,
+ -610.030564252835
+ ]
+ }
+ ],
+ "outputs": [
+ {
+ "id": "0f9ae109-484c-495b-9e82-7f454a26b6d5",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "linkIds": [
+ 750
+ ],
+ "label": "white_mesh",
+ "pos": [
+ 443.7021994103568,
+ -670.030564252835
+ ]
+ },
+ {
+ "id": "d2aad993-ce3c-4f5e-a2ba-e1ce7bef92d3",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "linkIds": [
+ 635
+ ],
+ "pos": [
+ 443.7021994103568,
+ -650.030564252835
+ ]
+ }
+ ],
+ "widgets": [],
+ "nodes": [
+ {
+ "id": 397,
+ "type": "Trellis2ImageCondGenerator",
+ "pos": [
+ -903.9659631794001,
+ -978.2409763036111
+ ],
+ "size": [
+ 308.0748046875,
+ 98
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 602
+ },
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 603
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "cond_512",
+ "name": "cond_512",
+ "type": "IMAGE_COND",
+ "links": [
+ 605,
+ 607
+ ]
+ },
+ {
+ "localized_name": "cond_1024",
+ "name": "cond_1024",
+ "type": "IMAGE_COND",
+ "links": [
+ 610
+ ]
+ },
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 604
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ImageCondGenerator"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 401,
+ "type": "Trellis2DecodeLatents",
+ "pos": [
+ -444.17325828513566,
+ -415.65865823859036
+ ],
+ "size": [
+ 270,
+ 122
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 614
+ },
+ {
+ "localized_name": "shape_slat",
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 615
+ },
+ {
+ "localized_name": "texture_slat",
+ "name": "texture_slat",
+ "shape": 7,
+ "type": "TEXTURE_SLAT",
+ "link": null
+ },
+ {
+ "localized_name": "resolution",
+ "name": "resolution",
+ "type": "INT",
+ "widget": {
+ "name": "resolution"
+ },
+ "link": 616
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 618
+ ]
+ },
+ {
+ "localized_name": "bvh",
+ "name": "bvh",
+ "type": "BVH",
+ "links": []
+ },
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2DecodeLatents"
+ },
+ "widgets_values": [
+ 0,
+ true
+ ]
+ },
+ {
+ "id": 410,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ -1249.953594419078,
+ -751.3094691431603
+ ],
+ "size": [
+ 281.7837890625,
+ 106
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 631
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 603
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "7ce26deae425d114f3deb78e802d6196e5987a4a",
+ "Node name for S&R": "Trellis2PreProcessImage"
+ },
+ "widgets_values": [
+ 5,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 1606,
+ "type": "Reroute",
+ "pos": [
+ -361.2137915568719,
+ -976.5651884005764
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 17,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 1964
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 2046
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 404,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ -1.3688675741572178,
+ -1197.2135353903182
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 619
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 1966
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 396,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ -1248.8098695647218,
+ -1061.4257138783487
+ ],
+ "size": [
+ 280.05208333333337,
+ 202
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 602,
+ 635
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
+ ]
+ },
+ {
+ "id": 398,
+ "type": "Trellis2SparseGenerator",
+ "pos": [
+ -909.3031482247874,
+ -824.5435537779637
+ ],
+ "size": [
+ 395.65625,
+ 314
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 604
+ },
+ {
+ "localized_name": "image_cond",
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 605
+ },
+ {
+ "localized_name": "seed",
+ "name": "seed",
+ "type": "INT",
+ "widget": {
+ "name": "seed"
+ },
+ "link": 1693
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "coords",
+ "name": "coords",
+ "type": "COORDS",
+ "links": [
+ 608
+ ]
+ },
+ {
+ "localized_name": "sparse_structure_resolution",
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "links": [
+ 613
+ ]
+ },
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 606
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2SparseGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ "euler",
+ 32,
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 399,
+ "type": "Trellis2ShapeGenerator",
+ "pos": [
+ -892.2604727135389,
+ -441.57534731890115
+ ],
+ "size": [
+ 335.24609375,
+ 266
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 606
+ },
+ {
+ "localized_name": "image_cond",
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 607
+ },
+ {
+ "localized_name": "coords",
+ "name": "coords",
+ "type": "COORDS",
+ "link": 608
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "shape_slat",
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 611
+ ]
+ },
+ {
+ "localized_name": "resolution",
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 612
+ ]
+ },
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 609
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeGenerator"
+ },
+ "widgets_values": [
+ 512,
+ 25,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 400,
+ "type": "Trellis2ShapeCascadeGenerator",
+ "pos": [
+ -465.2443600409803,
+ -820.2405791353129
+ ],
+ "size": [
+ 335.9791015625,
+ 338
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 609
+ },
+ {
+ "localized_name": "image_cond",
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 610
+ },
+ {
+ "localized_name": "shape_slat",
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 611
+ },
+ {
+ "localized_name": "from_resolution",
+ "name": "from_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "from_resolution"
+ },
+ "link": 612
+ },
+ {
+ "localized_name": "sparse_structure_resolution",
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "sparse_structure_resolution"
+ },
+ "link": 613
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "shape_slat",
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 615
+ ]
+ },
+ {
+ "localized_name": "resolution",
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 616
+ ]
+ },
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 614
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeCascadeGenerator"
+ },
+ "widgets_values": [
+ 0,
+ 1024,
+ 0,
+ 999999,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 414,
+ "type": "StringConcatenate",
+ "pos": [
+ -879.1728621391709,
+ -81.2258234995071
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 656
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 629
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_WhiteMesh",
+ ""
+ ]
+ },
+ {
+ "id": 406,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 50.28786765744911,
+ -792.0290399873467
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 4,
+ "inputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 2047
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 1971
+ ]
+ },
+ {
+ "localized_name": "holes_filled",
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1605,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 44.610272963955815,
+ -972.4658282598242
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 1966
+ },
+ {
+ "localized_name": "target_face_num",
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 2046
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 2047
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 402,
+ "type": "Trellis2MeshWithVoxelToTrimesh",
+ "pos": [
+ -19.274159374003318,
+ -597.7545005308027
+ ],
+ "size": [
+ 349.41171875,
+ 58
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 1971
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "trimesh",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 2044,
+ 2045
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ },
+ "widgets_values": [
+ "90 degrees"
+ ]
+ },
+ {
+ "id": 468,
+ "type": "Trellis2Continue",
+ "pos": [
+ 242.36320497352165,
+ -327.9179781939218
+ ],
+ "size": [
+ 161.33359375,
+ 46
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "input_1",
+ "name": "input_1",
+ "type": "*",
+ "link": 2045
+ },
+ {
+ "localized_name": "input_2",
+ "name": "input_2",
+ "type": "*",
+ "link": 749
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "output_1",
+ "name": "output_1",
+ "type": "*",
+ "links": [
+ 750
+ ]
+ },
+ {
+ "localized_name": "output_2",
+ "name": "output_2",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "6c3771938ad833fd840e489b02c839cd3b018e9f",
+ "Node name for S&R": "Trellis2Continue"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 409,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ -85.89777477147005,
+ -241.56574480172577
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "trimesh",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 2044
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 629
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "glb_path",
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 749
+ ]
+ },
+ {
+ "localized_name": "relative_path",
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "LaserPistol",
+ "glb"
+ ]
+ },
+ {
+ "id": 1475,
+ "type": "Reroute",
+ "pos": [
+ -1203.4611355198667,
+ -616.859229905566
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 1692
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 1693
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 427,
+ "type": "Reroute",
+ "pos": [
+ -1209.3136872322395,
+ -527.7593742715582
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 655
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 656
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 403,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 11.534077258920945,
+ -1321.6285802685652
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 618
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "mesh",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 619
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ }
+ ],
+ "groups": [],
+ "links": [
+ {
+ "id": 602,
+ "origin_id": 396,
+ "origin_slot": 0,
+ "target_id": 397,
+ "target_slot": 0,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 603,
+ "origin_id": 410,
+ "origin_slot": 0,
+ "target_id": 397,
+ "target_slot": 1,
+ "type": "IMAGE"
+ },
+ {
+ "id": 604,
+ "origin_id": 397,
+ "origin_slot": 2,
+ "target_id": 398,
+ "target_slot": 0,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 605,
+ "origin_id": 397,
+ "origin_slot": 0,
+ "target_id": 398,
+ "target_slot": 1,
+ "type": "IMAGE_COND"
+ },
+ {
+ "id": 606,
+ "origin_id": 398,
+ "origin_slot": 2,
+ "target_id": 399,
+ "target_slot": 0,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 607,
+ "origin_id": 397,
+ "origin_slot": 0,
+ "target_id": 399,
+ "target_slot": 1,
+ "type": "IMAGE_COND"
+ },
+ {
+ "id": 608,
+ "origin_id": 398,
+ "origin_slot": 0,
+ "target_id": 399,
+ "target_slot": 2,
+ "type": "COORDS"
+ },
+ {
+ "id": 609,
+ "origin_id": 399,
+ "origin_slot": 2,
+ "target_id": 400,
+ "target_slot": 0,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 610,
+ "origin_id": 397,
+ "origin_slot": 1,
+ "target_id": 400,
+ "target_slot": 1,
+ "type": "IMAGE_COND"
+ },
+ {
+ "id": 611,
+ "origin_id": 399,
+ "origin_slot": 0,
+ "target_id": 400,
+ "target_slot": 2,
+ "type": "SHAPE_SLAT"
+ },
+ {
+ "id": 612,
+ "origin_id": 399,
+ "origin_slot": 1,
+ "target_id": 400,
+ "target_slot": 3,
+ "type": "INT"
+ },
+ {
+ "id": 613,
+ "origin_id": 398,
+ "origin_slot": 1,
+ "target_id": 400,
+ "target_slot": 4,
+ "type": "INT"
+ },
+ {
+ "id": 614,
+ "origin_id": 400,
+ "origin_slot": 2,
+ "target_id": 401,
+ "target_slot": 0,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 615,
+ "origin_id": 400,
+ "origin_slot": 0,
+ "target_id": 401,
+ "target_slot": 1,
+ "type": "SHAPE_SLAT"
+ },
+ {
+ "id": 616,
+ "origin_id": 400,
+ "origin_slot": 1,
+ "target_id": 401,
+ "target_slot": 3,
+ "type": "INT"
+ },
+ {
+ "id": 618,
+ "origin_id": 401,
+ "origin_slot": 0,
+ "target_id": 403,
+ "target_slot": 0,
+ "type": "MESHWITHVOXEL"
+ },
+ {
+ "id": 619,
+ "origin_id": 403,
+ "origin_slot": 0,
+ "target_id": 404,
+ "target_slot": 0,
+ "type": "MESHWITHVOXEL"
+ },
+ {
+ "id": 629,
+ "origin_id": 414,
+ "origin_slot": 0,
+ "target_id": 409,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 631,
+ "origin_id": -10,
+ "origin_slot": 0,
+ "target_id": 410,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 635,
+ "origin_id": 396,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 1,
+ "type": "TRELLIS2PIPELINE"
+ },
+ {
+ "id": 655,
+ "origin_id": -10,
+ "origin_slot": 1,
+ "target_id": 427,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 656,
+ "origin_id": 427,
+ "origin_slot": 0,
+ "target_id": 414,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 749,
+ "origin_id": 409,
+ "origin_slot": 0,
+ "target_id": 468,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 750,
+ "origin_id": 468,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1692,
+ "origin_id": -10,
+ "origin_slot": 3,
+ "target_id": 1475,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1693,
+ "origin_id": 1475,
+ "origin_slot": 0,
+ "target_id": 398,
+ "target_slot": 2,
+ "type": "INT"
+ },
+ {
+ "id": 1964,
+ "origin_id": -10,
+ "origin_slot": 2,
+ "target_id": 1606,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1966,
+ "origin_id": 404,
+ "origin_slot": 0,
+ "target_id": 1605,
+ "target_slot": 0,
+ "type": "MESHWITHVOXEL"
+ },
+ {
+ "id": 1971,
+ "origin_id": 406,
+ "origin_slot": 0,
+ "target_id": 402,
+ "target_slot": 0,
+ "type": "MESHWITHVOXEL"
+ },
+ {
+ "id": 2044,
+ "origin_id": 402,
+ "origin_slot": 0,
+ "target_id": 409,
+ "target_slot": 0,
+ "type": "TRIMESH"
+ },
+ {
+ "id": 2045,
+ "origin_id": 402,
+ "origin_slot": 0,
+ "target_id": 468,
+ "target_slot": 0,
+ "type": "TRIMESH"
+ },
+ {
+ "id": 2046,
+ "origin_id": 1606,
+ "origin_slot": 0,
+ "target_id": 1605,
+ "target_slot": 1,
+ "type": "INT"
+ },
+ {
+ "id": 2047,
+ "origin_id": 1605,
+ "origin_slot": 0,
+ "target_id": 406,
+ "target_slot": 0,
+ "type": "MESHWITHVOXEL"
+ }
+ ],
+ "extra": {
+ "workflowRendererVersion": "LG"
+ }
+ },
+ {
+ "id": "bf1edf2f-ccb7-45ac-882e-55bdd5f0be86",
+ "version": 1,
+ "state": {
+ "lastGroupId": 32,
+ "lastNodeId": 1620,
+ "lastLinkId": 2048,
+ "lastRerouteId": 0
+ },
+ "revision": 0,
+ "config": {},
+ "name": "MultiView Generation",
+ "inputNode": {
+ "id": -10,
+ "bounding": [
+ -1746.4835931485231,
+ -470.3084268847315,
+ 120,
+ 100
+ ]
+ },
+ "outputNode": {
+ "id": -20,
+ "bounding": [
+ 820.8030880744058,
+ -593.0368552327766,
+ 120,
+ 160
+ ]
+ },
+ "inputs": [
+ {
+ "id": "e34c41cc-50b6-4907-9f86-5027585da336",
+ "name": "",
+ "type": "*",
+ "linkIds": [
+ 649
+ ],
+ "label": "prefix",
+ "pos": [
+ -1646.4835931485231,
+ -450.3084268847315
+ ]
+ },
+ {
+ "id": "9e8d195b-2c4c-4818-ab9a-1c87355b8e8f",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "linkIds": [
+ 662
+ ],
+ "label": "white_mesh",
+ "pos": [
+ -1646.4835931485231,
+ -430.3084268847315
+ ]
+ },
+ {
+ "id": "ccdf7db0-19ec-4036-b246-e0b04038529e",
+ "name": "_1",
+ "type": "*",
+ "linkIds": [
+ 1505
+ ],
+ "label": "color_image",
+ "pos": [
+ -1646.4835931485231,
+ -410.3084268847315
+ ]
+ }
+ ],
+ "outputs": [
+ {
+ "id": "9a3500ea-af9b-4260-a413-5960a0ed36d9",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "linkIds": [
+ 2034
+ ],
+ "label": "FrontView",
+ "pos": [
+ 840.8030880744058,
+ -573.0368552327766
+ ]
+ },
+ {
+ "id": "844113bb-886c-47a9-bf6a-3dc09a5e800e",
+ "name": "IMAGE_1",
+ "type": "IMAGE",
+ "linkIds": [
+ 2035
+ ],
+ "label": "LeftView",
+ "pos": [
+ 840.8030880744058,
+ -553.0368552327766
+ ]
+ },
+ {
+ "id": "fb5df4c8-d61e-48f5-b282-2e2da64f7a2e",
+ "name": "IMAGE_2",
+ "type": "IMAGE",
+ "linkIds": [
+ 2036
+ ],
+ "label": "BackView",
+ "pos": [
+ 840.8030880744058,
+ -533.0368552327766
+ ]
+ },
+ {
+ "id": "26886f31-ca69-429e-8829-abce39d19d0a",
+ "name": "IMAGE_3",
+ "type": "IMAGE",
+ "linkIds": [
+ 2037
+ ],
+ "label": "RightView",
+ "pos": [
+ 840.8030880744058,
+ -513.0368552327766
+ ]
+ },
+ {
+ "id": "f9ce56a7-2b8e-4eea-b2b6-0dbf9ff9f427",
+ "name": "IMAGE_4",
+ "type": "IMAGE",
+ "linkIds": [
+ 2038
+ ],
+ "label": "TopView",
+ "pos": [
+ 840.8030880744058,
+ -493.0368552327766
+ ]
+ },
+ {
+ "id": "0aa5299d-43ac-40ee-ad48-b4cf12470877",
+ "name": "IMAGE_5",
+ "type": "IMAGE",
+ "linkIds": [
+ 2039
+ ],
+ "label": "BottomView",
+ "pos": [
+ 840.8030880744058,
+ -473.0368552327766
+ ]
+ }
+ ],
+ "widgets": [],
+ "nodes": [
+ {
+ "id": 1328,
+ "type": "Hy3DDiffusersSchedulerConfig",
+ "pos": [
+ -862.8999634654654,
+ -581.3271020103193
+ ],
+ "size": [
+ 390.5999755859375,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "HY3DDIFFUSERSPIPE",
+ "link": 1502
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "diffusers_scheduler",
+ "name": "diffusers_scheduler",
+ "type": "NOISESCHEDULER",
+ "slot_index": 0,
+ "links": [
+ 1503
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper",
+ "ver": "6090a9051f109d2774d309cf6a8a22d4a11a1b15",
+ "Node name for S&R": "Hy3DDiffusersSchedulerConfig",
+ "widget_ue_connectable": {
+ "scheduler": true,
+ "sigmas": true
+ },
+ "cnr_id": "comfyui-hunyan3dwrapper"
+ },
+ "widgets_values": [
+ "Euler",
+ "default"
+ ]
+ },
+ {
+ "id": 1327,
+ "type": "DownloadAndLoadHy3DPaintModel",
+ "pos": [
+ -1274.0598329076045,
+ -714.5061282030463
+ ],
+ "size": [
+ 327.5999755859375,
+ 58
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 0,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "compile_args",
+ "name": "compile_args",
+ "shape": 7,
+ "type": "HY3DCOMPILEARGS",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "multiview_pipe",
+ "name": "multiview_pipe",
+ "type": "HY3DDIFFUSERSPIPE",
+ "slot_index": 0,
+ "links": [
+ 1502,
+ 1504
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper",
+ "ver": "6090a9051f109d2774d309cf6a8a22d4a11a1b15",
+ "Node name for S&R": "DownloadAndLoadHy3DPaintModel",
+ "widget_ue_connectable": {
+ "model": true
+ },
+ "cnr_id": "comfyui-hunyan3dwrapper"
+ },
+ "widgets_values": [
+ "hunyuan3d-paint-v2-0"
+ ]
+ },
+ {
+ "id": 296,
+ "type": "Hy3DRenderMultiView",
+ "pos": [
+ -1239.8610699824203,
+ -428.39700113665066
+ ],
+ "size": [
+ 303.787109375,
+ 234
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "trimesh",
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 662
+ },
+ {
+ "localized_name": "camera_config",
+ "name": "camera_config",
+ "shape": 7,
+ "type": "HY3DCAMERA",
+ "link": 546
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "normal_maps",
+ "name": "normal_maps",
+ "type": "IMAGE",
+ "links": [
+ 1507
+ ]
+ },
+ {
+ "localized_name": "position_maps",
+ "name": "position_maps",
+ "type": "IMAGE",
+ "links": [
+ 1508
+ ]
+ },
+ {
+ "localized_name": "renderer",
+ "name": "renderer",
+ "type": "MESHRENDER",
+ "links": null
+ },
+ {
+ "localized_name": "masks",
+ "name": "masks",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "localized_name": "textured_maps",
+ "name": "textured_maps",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper",
+ "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d",
+ "Node name for S&R": "Hy3DRenderMultiView"
+ },
+ "widgets_values": [
+ 1024,
+ 1024,
+ false,
+ "",
+ "world"
+ ]
+ },
+ {
+ "id": 1035,
+ "type": "StringConcatenate",
+ "pos": [
+ -990.4672803385581,
+ 51.17410543902827
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 1204
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1873
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_Ortho",
+ ""
+ ]
+ },
+ {
+ "id": 298,
+ "type": "Hy3DCameraConfig",
+ "pos": [
+ -1542.2009936232482,
+ -594.7805122274469
+ ],
+ "size": [
+ 270,
+ 154
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "camera_config",
+ "name": "camera_config",
+ "type": "HY3DCAMERA",
+ "links": [
+ 546,
+ 1506
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper",
+ "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d",
+ "Node name for S&R": "Hy3DCameraConfig"
+ },
+ "widgets_values": [
+ "0, 90, 180, 270, 0, 0",
+ "0, 0, 0, 0, 90, -90",
+ "1,1,1,1,1,1",
+ 1.1,
+ 1.2
+ ]
+ },
+ {
+ "id": 299,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -435.22855283205126,
+ -757.9872683426315
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1510
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1499,
+ 1514,
+ 2021
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "1",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 318,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -435.6953598313359,
+ -574.8401846327595
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1511
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1500,
+ 1518,
+ 2022
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "2",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 1543,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ -543.7388509308294,
+ 95.01661567145857
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 19,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1872
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1873
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 355,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -430.8103719299532,
+ -406.66851210910005
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1512
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1501,
+ 1519,
+ 2023
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "3",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 1326,
+ "type": "Hy3DSampleMultiView",
+ "pos": [
+ -883.0149101311823,
+ -469.245600284044
+ ],
+ "size": [
+ 270,
+ 274
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pipeline",
+ "name": "pipeline",
+ "type": "HY3DDIFFUSERSPIPE",
+ "link": 1504
+ },
+ {
+ "localized_name": "ref_image",
+ "name": "ref_image",
+ "type": "IMAGE",
+ "link": 1505
+ },
+ {
+ "localized_name": "normal_maps",
+ "name": "normal_maps",
+ "type": "IMAGE",
+ "link": 1507
+ },
+ {
+ "localized_name": "position_maps",
+ "name": "position_maps",
+ "type": "IMAGE",
+ "link": 1508
+ },
+ {
+ "localized_name": "camera_config",
+ "name": "camera_config",
+ "shape": 7,
+ "type": "HY3DCAMERA",
+ "link": 1506
+ },
+ {
+ "localized_name": "scheduler",
+ "name": "scheduler",
+ "shape": 7,
+ "type": "NOISESCHEDULER",
+ "link": 1503
+ },
+ {
+ "localized_name": "samples",
+ "name": "samples",
+ "shape": 7,
+ "type": "LATENT",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 1509,
+ 1510,
+ 1511,
+ 1512,
+ 1872,
+ 2024,
+ 2025
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper",
+ "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d",
+ "Node name for S&R": "Hy3DSampleMultiView"
+ },
+ "widgets_values": [
+ 1024,
+ 12,
+ 12345,
+ "fixed",
+ 1
+ ]
+ },
+ {
+ "id": 426,
+ "type": "Reroute",
+ "pos": [
+ -1496.3532269059936,
+ -389.05347596216643
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 649
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 650,
+ 651,
+ 652,
+ 653,
+ 1204,
+ 1491,
+ 1492,
+ 1493,
+ 1515,
+ 1516,
+ 1517,
+ 2028,
+ 2031
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 297,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -436.1636888919487,
+ -938.3479012181282
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1509
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2020
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "0",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 1610,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -424.2401697968086,
+ -67.4260309811222
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 25,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2025
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2027
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "5",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 1609,
+ "type": "VHS_SelectImages",
+ "pos": [
+ -435.6599296933568,
+ -241.75265413133576
+ ],
+ "size": [
+ 212.5712890625,
+ 106
+ ],
+ "flags": {},
+ "order": 24,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2024
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2026
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960",
+ "Node name for S&R": "VHS_SelectImages"
+ },
+ "widgets_values": {
+ "indexes": "4",
+ "err_if_missing": true,
+ "err_if_empty": true
+ }
+ },
+ {
+ "id": 364,
+ "type": "StringConcatenate",
+ "pos": [
+ 35.77617179788515,
+ -846.0155885271691
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 650
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1875
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_FrontView",
+ ""
+ ]
+ },
+ {
+ "id": 367,
+ "type": "StringConcatenate",
+ "pos": [
+ 34.080342327046154,
+ -666.1024115080415
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 651
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1877
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_LeftView",
+ ""
+ ]
+ },
+ {
+ "id": 371,
+ "type": "StringConcatenate",
+ "pos": [
+ 35.71309824482861,
+ -503.67579172607446
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 652
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1879
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_BackView",
+ ""
+ ]
+ },
+ {
+ "id": 372,
+ "type": "StringConcatenate",
+ "pos": [
+ 40.46595031847387,
+ -359.967454329042
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 653
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1881
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_RightView",
+ ""
+ ]
+ },
+ {
+ "id": 1613,
+ "type": "StringConcatenate",
+ "pos": [
+ 42.9445878581309,
+ -219.80206489015328
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 28,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 2028
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 2030
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_TopView",
+ ""
+ ]
+ },
+ {
+ "id": 1615,
+ "type": "StringConcatenate",
+ "pos": [
+ 35.09400401574455,
+ -19.439852278110624
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 30,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 2031
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 2033
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.16.4",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_BottomView",
+ ""
+ ]
+ },
+ {
+ "id": 494,
+ "type": "RMBG",
+ "pos": [
+ -19.26089667273442,
+ -908.2866484721291
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2020
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1874,
+ 2034
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1500,
+ "type": "RMBG",
+ "pos": [
+ -15.770746233324541,
+ -722.8726246922832
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2021
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1876,
+ 2035
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1501,
+ "type": "RMBG",
+ "pos": [
+ -11.102746066987152,
+ -556.9884037168202
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 17,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2022
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1878,
+ 2036
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1502,
+ "type": "RMBG",
+ "pos": [
+ -5.297787896089057,
+ -425.97143469459684
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 18,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2023
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1880,
+ 2037
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1611,
+ "type": "RMBG",
+ "pos": [
+ 0.23025981432448983,
+ -277.57154720273326
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 26,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2026
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2029,
+ 2038
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1612,
+ "type": "RMBG",
+ "pos": [
+ -9.806598738150342,
+ -89.90771460473213
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 27,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 2027
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 2032,
+ 2039
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1545,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 432.2244570200339,
+ -721.078003096569
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 21,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1876
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1877
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1546,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 444.4266599842291,
+ -570.5252945485121
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 22,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1878
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1879
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1547,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 430.51678393065424,
+ -407.6644013260513
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 23,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1880
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1881
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1614,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 439.89666728876546,
+ -267.60335440135475
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 29,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 2029
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 2030
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1616,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 444.9208198224709,
+ -109.84120169377852
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 31,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 2032
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 2033
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1544,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 430.0252087743253,
+ -899.771306242578
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 20,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1874
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1875
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "86f063655a00bcb3a325273ec941899cc06fd115",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ }
+ ],
+ "groups": [],
+ "links": [
+ {
+ "id": 546,
+ "origin_id": 298,
+ "origin_slot": 0,
+ "target_id": 296,
+ "target_slot": 1,
+ "type": "HY3DCAMERA"
+ },
+ {
+ "id": 649,
+ "origin_id": -10,
+ "origin_slot": 0,
+ "target_id": 426,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 650,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 364,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 651,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 367,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 652,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 371,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 653,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 372,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 662,
+ "origin_id": -10,
+ "origin_slot": 1,
+ "target_id": 296,
+ "target_slot": 0,
+ "type": "TRIMESH"
+ },
+ {
+ "id": 1204,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1035,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 1491,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1279,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1492,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1302,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1493,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1325,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1499,
+ "origin_id": 299,
+ "origin_slot": 0,
+ "target_id": 1279,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1500,
+ "origin_id": 318,
+ "origin_slot": 0,
+ "target_id": 1302,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1501,
+ "origin_id": 355,
+ "origin_slot": 0,
+ "target_id": 1325,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1502,
+ "origin_id": 1327,
+ "origin_slot": 0,
+ "target_id": 1328,
+ "target_slot": 0,
+ "type": "HY3DDIFFUSERSPIPE"
+ },
+ {
+ "id": 1503,
+ "origin_id": 1328,
+ "origin_slot": 0,
+ "target_id": 1326,
+ "target_slot": 5,
+ "type": "NOISESCHEDULER"
+ },
+ {
+ "id": 1504,
+ "origin_id": 1327,
+ "origin_slot": 0,
+ "target_id": 1326,
+ "target_slot": 0,
+ "type": "HY3DDIFFUSERSPIPE"
+ },
+ {
+ "id": 1505,
+ "origin_id": -10,
+ "origin_slot": 2,
+ "target_id": 1326,
+ "target_slot": 1,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1506,
+ "origin_id": 298,
+ "origin_slot": 0,
+ "target_id": 1326,
+ "target_slot": 4,
+ "type": "HY3DCAMERA"
+ },
+ {
+ "id": 1507,
+ "origin_id": 296,
+ "origin_slot": 0,
+ "target_id": 1326,
+ "target_slot": 2,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1508,
+ "origin_id": 296,
+ "origin_slot": 1,
+ "target_id": 1326,
+ "target_slot": 3,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1509,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 297,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1510,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 299,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1511,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 318,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1512,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 355,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1514,
+ "origin_id": 299,
+ "origin_slot": 0,
+ "target_id": 1351,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1515,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1351,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1516,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1374,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1517,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1397,
+ "target_slot": 3,
+ "type": "*"
+ },
+ {
+ "id": 1518,
+ "origin_id": 318,
+ "origin_slot": 0,
+ "target_id": 1374,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1519,
+ "origin_id": 355,
+ "origin_slot": 0,
+ "target_id": 1397,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1872,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 1543,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1873,
+ "origin_id": 1035,
+ "origin_slot": 0,
+ "target_id": 1543,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1874,
+ "origin_id": 494,
+ "origin_slot": 0,
+ "target_id": 1544,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1875,
+ "origin_id": 364,
+ "origin_slot": 0,
+ "target_id": 1544,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1876,
+ "origin_id": 1500,
+ "origin_slot": 0,
+ "target_id": 1545,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1877,
+ "origin_id": 367,
+ "origin_slot": 0,
+ "target_id": 1545,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1878,
+ "origin_id": 1501,
+ "origin_slot": 0,
+ "target_id": 1546,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1879,
+ "origin_id": 371,
+ "origin_slot": 0,
+ "target_id": 1546,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1880,
+ "origin_id": 1502,
+ "origin_slot": 0,
+ "target_id": 1547,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1881,
+ "origin_id": 372,
+ "origin_slot": 0,
+ "target_id": 1547,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 2020,
+ "origin_id": 297,
+ "origin_slot": 0,
+ "target_id": 494,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2021,
+ "origin_id": 299,
+ "origin_slot": 0,
+ "target_id": 1500,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2022,
+ "origin_id": 318,
+ "origin_slot": 0,
+ "target_id": 1501,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2023,
+ "origin_id": 355,
+ "origin_slot": 0,
+ "target_id": 1502,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2024,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 1609,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2025,
+ "origin_id": 1326,
+ "origin_slot": 0,
+ "target_id": 1610,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2026,
+ "origin_id": 1609,
+ "origin_slot": 0,
+ "target_id": 1611,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2027,
+ "origin_id": 1610,
+ "origin_slot": 0,
+ "target_id": 1612,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2028,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1613,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 2029,
+ "origin_id": 1611,
+ "origin_slot": 0,
+ "target_id": 1614,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2030,
+ "origin_id": 1613,
+ "origin_slot": 0,
+ "target_id": 1614,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 2031,
+ "origin_id": 426,
+ "origin_slot": 0,
+ "target_id": 1615,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 2032,
+ "origin_id": 1612,
+ "origin_slot": 0,
+ "target_id": 1616,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2033,
+ "origin_id": 1615,
+ "origin_slot": 0,
+ "target_id": 1616,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 2034,
+ "origin_id": 494,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2035,
+ "origin_id": 1500,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 1,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2036,
+ "origin_id": 1501,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 2,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2037,
+ "origin_id": 1502,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 3,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2038,
+ "origin_id": 1611,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 4,
+ "type": "IMAGE"
+ },
+ {
+ "id": 2039,
+ "origin_id": 1612,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 5,
+ "type": "IMAGE"
+ }
+ ],
+ "extra": {
+ "workflowRendererVersion": "LG"
+ }
+ },
+ {
+ "id": "e62a2ed8-84aa-4855-9c23-e444bcf27791",
+ "version": 1,
+ "state": {
+ "lastGroupId": 32,
+ "lastNodeId": 1620,
+ "lastLinkId": 2048,
+ "lastRerouteId": 0
+ },
+ "revision": 0,
+ "config": {},
+ "name": "Normal Generation",
+ "inputNode": {
+ "id": -10,
+ "bounding": [
+ -3817.1162130023004,
+ -1898.1700046856838,
+ 120,
+ 100
+ ]
+ },
+ "outputNode": {
+ "id": -20,
+ "bounding": [
+ 1856.3953075122079,
+ -2035.1527997843789,
+ 190.146484375,
+ 80
+ ]
+ },
+ "inputs": [
+ {
+ "id": "13fd56fd-b271-45d9-a15e-422894e3c54a",
+ "name": "image",
+ "type": "IMAGE",
+ "linkIds": [
+ 1900
+ ],
+ "label": "source_image",
+ "pos": [
+ -3717.1162130023004,
+ -1878.1700046856838
+ ]
+ },
+ {
+ "id": "4d41f897-1b58-4f8c-8ea3-53dca49bbade",
+ "name": "",
+ "type": "*",
+ "linkIds": [
+ 1876
+ ],
+ "label": "prefix",
+ "pos": [
+ -3717.1162130023004,
+ -1858.1700046856838
+ ]
+ },
+ {
+ "id": "cf851b38-1b59-43f9-8094-f0f28b7804ec",
+ "name": "_1",
+ "type": "*",
+ "linkIds": [
+ 1947
+ ],
+ "label": "seed",
+ "pos": [
+ -3717.1162130023004,
+ -1838.1700046856838
+ ]
+ }
+ ],
+ "outputs": [
+ {
+ "id": "89cda40d-3878-45a6-bc24-ef146cc3731d",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "linkIds": [
+ 1898
+ ],
+ "label": "source_image",
+ "pos": [
+ 1876.3953075122079,
+ -2015.1527997843789
+ ]
+ },
+ {
+ "id": "8cc5e680-c415-4b4e-803e-5d5115fe5847",
+ "name": "IMAGE_1",
+ "type": "IMAGE",
+ "linkIds": [
+ 1917
+ ],
+ "label": "normal_image_transparent",
+ "pos": [
+ 1876.3953075122079,
+ -1995.1527997843789
+ ]
+ }
+ ],
+ "widgets": [],
+ "nodes": [
+ {
+ "id": 1569,
+ "type": "ModelSamplingAuraFlow",
+ "pos": [
+ -115.00971403135668,
+ -2337.6245076749606
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "model",
+ "name": "model",
+ "type": "MODEL",
+ "link": 1841
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "MODEL",
+ "name": "MODEL",
+ "type": "MODEL",
+ "links": [
+ 1854
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "ModelSamplingAuraFlow",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 3.1
+ ]
+ },
+ {
+ "id": 1570,
+ "type": "KSampler",
+ "pos": [
+ 202.66715685192432,
+ -2335.7053574496863
+ ],
+ "size": [
+ 280,
+ 280
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "model",
+ "name": "model",
+ "type": "MODEL",
+ "link": 1842
+ },
+ {
+ "localized_name": "positive",
+ "name": "positive",
+ "type": "CONDITIONING",
+ "link": 1879
+ },
+ {
+ "localized_name": "negative",
+ "name": "negative",
+ "type": "CONDITIONING",
+ "link": 1880
+ },
+ {
+ "localized_name": "latent_image",
+ "name": "latent_image",
+ "type": "LATENT",
+ "link": 1845
+ },
+ {
+ "localized_name": "seed",
+ "name": "seed",
+ "type": "INT",
+ "widget": {
+ "name": "seed"
+ },
+ "link": 1948
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "LATENT",
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 1848
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "KSampler",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 4,
+ 1,
+ "euler",
+ "simple",
+ 1
+ ]
+ },
+ {
+ "id": 1571,
+ "type": "CFGNorm",
+ "pos": [
+ -118.33919486966624,
+ -2215.139128869213
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "model",
+ "name": "model",
+ "type": "MODEL",
+ "link": 1854
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "patched_model",
+ "name": "patched_model",
+ "type": "MODEL",
+ "links": [
+ 1842
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "CFGNorm",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 1572,
+ "type": "EmptyLatentImage",
+ "pos": [
+ -111.33960518501158,
+ -2102.8599395388537
+ ],
+ "size": [
+ 270,
+ 106
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "LATENT",
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 1845
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.14.1",
+ "Node name for S&R": "EmptyLatentImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1024,
+ 1024,
+ 1
+ ]
+ },
+ {
+ "id": 1573,
+ "type": "TextEncodeQwenImageEditPlus",
+ "pos": [
+ -637.9784327451256,
+ -2027.7176557548157
+ ],
+ "size": [
+ 420,
+ 168
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "clip",
+ "name": "clip",
+ "type": "CLIP",
+ "link": 1859
+ },
+ {
+ "localized_name": "vae",
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": 1860
+ },
+ {
+ "localized_name": "image1",
+ "name": "image1",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 1875
+ },
+ {
+ "localized_name": "image2",
+ "name": "image2",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "localized_name": "image3",
+ "name": "image3",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "CONDITIONING",
+ "name": "CONDITIONING",
+ "type": "CONDITIONING",
+ "links": [
+ 1850
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "TextEncodeQwenImageEditPlus",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ ""
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 1574,
+ "type": "CLIPLoader",
+ "pos": [
+ -1124.3007943593586,
+ -1970.878331573208
+ ],
+ "size": [
+ 396.1197916666667,
+ 106
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "CLIP",
+ "name": "CLIP",
+ "type": "CLIP",
+ "links": [
+ 1857,
+ 1859
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "CLIPLoader",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "models": [
+ {
+ "name": "qwen_2.5_vl_7b_fp8_scaled.safetensors",
+ "url": "https://huggingface.co/Comfy-Org/HunyuanVideo_1.5_repackaged/resolve/main/split_files/text_encoders/qwen_2.5_vl_7b_fp8_scaled.safetensors",
+ "directory": "text_encoders"
+ }
+ ],
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "qwen_2.5_vl_7b_fp8_scaled.safetensors",
+ "qwen_image",
+ "default"
+ ]
+ },
+ {
+ "id": 1575,
+ "type": "VAELoader",
+ "pos": [
+ -1125.7132689453608,
+ -1784.1974997982275
+ ],
+ "size": [
+ 396.1197916666667,
+ 58
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "VAE",
+ "name": "VAE",
+ "type": "VAE",
+ "slot_index": 0,
+ "links": [
+ 1849,
+ 1858,
+ 1860,
+ 1865
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "VAELoader",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "models": [
+ {
+ "name": "qwen_image_vae.safetensors",
+ "url": "https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/resolve/main/split_files/vae/qwen_image_vae.safetensors",
+ "directory": "vae"
+ }
+ ],
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "qwen_image_vae.safetensors"
+ ]
+ },
+ {
+ "id": 1576,
+ "type": "ImageResizeKJv2",
+ "pos": [
+ -1489.802570145589,
+ -2136.853995293102
+ ],
+ "size": [
+ 278.5893539007568,
+ 288.0639521984865
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1903
+ },
+ {
+ "localized_name": "mask",
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1874,
+ 1875,
+ 1878
+ ]
+ },
+ {
+ "localized_name": "width",
+ "name": "width",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "localized_name": "height",
+ "name": "height",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "localized_name": "mask",
+ "name": "mask",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "1.1.7",
+ "Node name for S&R": "ImageResizeKJv2"
+ },
+ "widgets_values": [
+ 1024,
+ 1024,
+ "lanczos",
+ "stretch",
+ "0, 0, 0",
+ "center",
+ 2,
+ "cpu"
+ ]
+ },
+ {
+ "id": 1577,
+ "type": "VAEEncode",
+ "pos": [
+ -426.07544431121744,
+ -1752.8109264596974
+ ],
+ "size": [
+ 190,
+ 46
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "pixels",
+ "name": "pixels",
+ "type": "IMAGE",
+ "link": 1878
+ },
+ {
+ "localized_name": "vae",
+ "name": "vae",
+ "type": "VAE",
+ "link": 1865
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "LATENT",
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 1851,
+ 1856
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.8.2",
+ "Node name for S&R": "VAEEncode",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1578,
+ "type": "ReferenceLatent",
+ "pos": [
+ -76.2171515196161,
+ -1834.0309760866128
+ ],
+ "size": [
+ 210,
+ 46
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "conditioning",
+ "name": "conditioning",
+ "type": "CONDITIONING",
+ "link": 1855
+ },
+ {
+ "localized_name": "latent",
+ "name": "latent",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 1856
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "CONDITIONING",
+ "name": "CONDITIONING",
+ "type": "CONDITIONING",
+ "links": [
+ 1879
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.8.2",
+ "Node name for S&R": "ReferenceLatent",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1579,
+ "type": "ReferenceLatent",
+ "pos": [
+ -74.0707456357083,
+ -1733.6745960744763
+ ],
+ "size": [
+ 204.134765625,
+ 46
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "conditioning",
+ "name": "conditioning",
+ "type": "CONDITIONING",
+ "link": 1850
+ },
+ {
+ "localized_name": "latent",
+ "name": "latent",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 1851
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "CONDITIONING",
+ "name": "CONDITIONING",
+ "type": "CONDITIONING",
+ "links": [
+ 1880
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.8.2",
+ "Node name for S&R": "ReferenceLatent",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1580,
+ "type": "TextEncodeQwenImageEditPlus",
+ "pos": [
+ -635.9353251063575,
+ -2326.2691568206615
+ ],
+ "size": [
+ 431.9022487170264,
+ 239.91270427521232
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "clip",
+ "name": "clip",
+ "type": "CLIP",
+ "link": 1857
+ },
+ {
+ "localized_name": "vae",
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": 1858
+ },
+ {
+ "localized_name": "image1",
+ "name": "image1",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 1874
+ },
+ {
+ "localized_name": "image2",
+ "name": "image2",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "localized_name": "image3",
+ "name": "image3",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "CONDITIONING",
+ "name": "CONDITIONING",
+ "type": "CONDITIONING",
+ "links": [
+ 1855
+ ]
+ }
+ ],
+ "title": "TextEncodeQwenImageEditPlus (Positive)",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "TextEncodeQwenImageEditPlus",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Generate a highly detailed normal map of the image"
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 1581,
+ "type": "VAEDecodeTiled",
+ "pos": [
+ 526.7607544549602,
+ -2532.6820137931068
+ ],
+ "size": [
+ 225,
+ 150
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "samples",
+ "name": "samples",
+ "type": "LATENT",
+ "link": 1848
+ },
+ {
+ "localized_name": "vae",
+ "name": "vae",
+ "type": "VAE",
+ "link": 1849
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1881
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.9.1",
+ "Node name for S&R": "VAEDecodeTiled",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 128,
+ 64,
+ 64,
+ 8
+ ]
+ },
+ {
+ "id": 1582,
+ "type": "Reroute",
+ "pos": [
+ -2190.0488823440746,
+ -1559.9544752931013
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 1876
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 1877,
+ 1883
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 1583,
+ "type": "UnloadAllModels",
+ "pos": [
+ 1447.9470179112147,
+ -2042.9135859133257
+ ],
+ "size": [
+ 151.941015625,
+ 26
+ ],
+ "flags": {},
+ "order": 16,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "value",
+ "name": "value",
+ "type": "*",
+ "link": 1916
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "*",
+ "name": "*",
+ "type": "*",
+ "links": [
+ 1897
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-unload-model",
+ "ver": "ac5ffb4ed05546545ce7cf38e7b69b5152714eed",
+ "Node name for S&R": "UnloadAllModels"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1584,
+ "type": "RMBG",
+ "pos": [
+ -2391.9355550426676,
+ -2025.8660535759227
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {},
+ "order": 17,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1900
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1901,
+ 1908
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1585,
+ "type": "StringConcatenate",
+ "pos": [
+ -1905.0159643022744,
+ -1418.6616489185215
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 18,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 1877
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1909
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.17.0",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_SourceImage",
+ ""
+ ]
+ },
+ {
+ "id": 1586,
+ "type": "RMBG",
+ "pos": [
+ 801.1227972789899,
+ -2357.1958642043965
+ ],
+ "size": [
+ 320,
+ 320.546875
+ ],
+ "flags": {},
+ "order": 19,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "Image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1881
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "IMAGE",
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 1910,
+ 1913
+ ]
+ },
+ {
+ "localized_name": "MASK",
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ },
+ {
+ "localized_name": "MASK IMAGE",
+ "name": "MASK_IMAGE",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfyui-rmbg",
+ "ver": "2.3.2",
+ "Node name for S&R": "RMBG",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "RMBG-2.0",
+ 1,
+ 1024,
+ 0,
+ 0,
+ false,
+ false,
+ "Alpha",
+ "#000000"
+ ],
+ "color": "#222e40",
+ "bgcolor": "#364254"
+ },
+ {
+ "id": 1587,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ -1393.5546159526318,
+ -1449.641997359841
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 20,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1908
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1909
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": [
+ 1914
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1589,
+ "type": "Trellis2SaveImage",
+ "pos": [
+ 1221.4525526830735,
+ -1497.1011446614482
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 22,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "images",
+ "name": "images",
+ "type": "IMAGE",
+ "link": 1910
+ },
+ {
+ "localized_name": "filename_prefix",
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 1911
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "images_path",
+ "name": "images_path",
+ "type": "STRING",
+ "links": [
+ 1915
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268",
+ "Node name for S&R": "Trellis2SaveImage"
+ },
+ "widgets_values": [
+ "ComfyUI",
+ 1
+ ]
+ },
+ {
+ "id": 1590,
+ "type": "StringConcatenate",
+ "pos": [
+ 714.1324446088805,
+ -1512.6780555515054
+ ],
+ "size": [
+ 400,
+ 200
+ ],
+ "flags": {
+ "collapsed": false
+ },
+ "order": 23,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "string_a",
+ "name": "string_a",
+ "type": "STRING",
+ "widget": {
+ "name": "string_a"
+ },
+ "link": 1883
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "STRING",
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 1911
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.17.0",
+ "Node name for S&R": "StringConcatenate"
+ },
+ "widgets_values": [
+ "",
+ "_NormalImage",
+ ""
+ ]
+ },
+ {
+ "id": 1591,
+ "type": "Trellis2CudaReset",
+ "pos": [
+ 1644.461580167075,
+ -2046.46286508813
+ ],
+ "size": [
+ 177.7330078125,
+ 26
+ ],
+ "flags": {},
+ "order": 24,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "input_1",
+ "name": "input_1",
+ "type": "*",
+ "link": 1897
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "output_1",
+ "name": "output_1",
+ "type": "*",
+ "links": [
+ 1898
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "63fe7ecf14006a8f4f2c91e7e0b1d39feaba4a74",
+ "Node name for S&R": "Trellis2CudaReset"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 1592,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ -1864.8219949814027,
+ -2026.3949704551314
+ ],
+ "size": [
+ 281.7837890625,
+ 106
+ ],
+ "flags": {},
+ "order": 25,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1901
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "image",
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 1903,
+ 1912
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2PreProcessImage"
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 1593,
+ "type": "LoraLoaderModelOnly",
+ "pos": [
+ -1120.8732878670087,
+ -2178.199053568686
+ ],
+ "size": [
+ 396.1197916666667,
+ 82
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "model",
+ "name": "model",
+ "type": "MODEL",
+ "link": 1866
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "MODEL",
+ "name": "MODEL",
+ "type": "MODEL",
+ "links": [
+ 1841
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.5.1",
+ "Node name for S&R": "LoraLoaderModelOnly",
+ "enableTabs": false,
+ "tabWidth": 65,
+ "tabXOffset": 10,
+ "hasSecondTab": false,
+ "secondTabText": "Send Back",
+ "secondTabOffset": 80,
+ "secondTabWidth": 65,
+ "models": [
+ {
+ "name": "Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors",
+ "url": "https://huggingface.co/lightx2v/Qwen-Image-Edit-2511-Lightning/resolve/main/Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors",
+ "directory": "loras"
+ }
+ ],
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors",
+ 1
+ ]
+ },
+ {
+ "id": 1594,
+ "type": "UnetLoaderGGUF",
+ "pos": [
+ -1124.495826042935,
+ -2331.0673251834373
+ ],
+ "size": [
+ 364.97395833333337,
+ 58
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "localized_name": "MODEL",
+ "name": "MODEL",
+ "type": "MODEL",
+ "links": [
+ 1866
+ ]
+ }
+ ],
+ "properties": {
+ "cnr_id": "ComfyUI-GGUF",
+ "ver": "795e45156ece99afbc3efef911e63fcb46e6a20d",
+ "Node name for S&R": "UnetLoaderGGUF",
+ "aux_id": "city96/ComfyUI-GGUF",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "qwen-image-edit-2511-Q4_K_M.gguf"
+ ]
+ },
+ {
+ "id": 1595,
+ "type": "Reroute",
+ "pos": [
+ -1457.9227051676305,
+ -1678.5900903360498
+ ],
+ "size": [
+ 140,
+ 60
+ ],
+ "flags": {},
+ "order": 26,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 1947
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "*",
+ "links": [
+ 1948
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 1588,
+ "type": "Trellis2Continue4",
+ "pos": [
+ 1244.0426972330154,
+ -1996.538284106618
+ ],
+ "size": [
+ 174.3150390625,
+ 86
+ ],
+ "flags": {},
+ "order": 21,
+ "mode": 0,
+ "inputs": [
+ {
+ "localized_name": "input_1",
+ "name": "input_1",
+ "type": "*",
+ "link": 1912
+ },
+ {
+ "localized_name": "input_2",
+ "name": "input_2",
+ "type": "*",
+ "link": 1913
+ },
+ {
+ "localized_name": "input_3",
+ "name": "input_3",
+ "type": "*",
+ "link": 1914
+ },
+ {
+ "localized_name": "input_4",
+ "name": "input_4",
+ "type": "*",
+ "link": 1915
+ }
+ ],
+ "outputs": [
+ {
+ "localized_name": "output_1",
+ "name": "output_1",
+ "type": "*",
+ "links": [
+ 1916
+ ]
+ },
+ {
+ "localized_name": "output_2",
+ "name": "output_2",
+ "type": "*",
+ "links": [
+ 1917
+ ]
+ },
+ {
+ "localized_name": "output_3",
+ "name": "output_3",
+ "type": "*",
+ "links": null
+ },
+ {
+ "localized_name": "output_4",
+ "name": "output_4",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268",
+ "Node name for S&R": "Trellis2Continue4"
+ },
+ "widgets_values": []
+ }
+ ],
+ "groups": [
+ {
+ "id": 30,
+ "title": "Resize and Remove Background",
+ "bounding": [
+ -2444.357644543909,
+ -2123.9220262537956,
+ 907.8307755616227,
+ 445.7473139007566
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "links": [
+ {
+ "id": 1841,
+ "origin_id": 1593,
+ "origin_slot": 0,
+ "target_id": 1569,
+ "target_slot": 0,
+ "type": "MODEL"
+ },
+ {
+ "id": 1842,
+ "origin_id": 1571,
+ "origin_slot": 0,
+ "target_id": 1570,
+ "target_slot": 0,
+ "type": "MODEL"
+ },
+ {
+ "id": 1845,
+ "origin_id": 1572,
+ "origin_slot": 0,
+ "target_id": 1570,
+ "target_slot": 3,
+ "type": "LATENT"
+ },
+ {
+ "id": 1848,
+ "origin_id": 1570,
+ "origin_slot": 0,
+ "target_id": 1581,
+ "target_slot": 0,
+ "type": "LATENT"
+ },
+ {
+ "id": 1849,
+ "origin_id": 1575,
+ "origin_slot": 0,
+ "target_id": 1581,
+ "target_slot": 1,
+ "type": "VAE"
+ },
+ {
+ "id": 1850,
+ "origin_id": 1573,
+ "origin_slot": 0,
+ "target_id": 1579,
+ "target_slot": 0,
+ "type": "CONDITIONING"
+ },
+ {
+ "id": 1851,
+ "origin_id": 1577,
+ "origin_slot": 0,
+ "target_id": 1579,
+ "target_slot": 1,
+ "type": "LATENT"
+ },
+ {
+ "id": 1854,
+ "origin_id": 1569,
+ "origin_slot": 0,
+ "target_id": 1571,
+ "target_slot": 0,
+ "type": "MODEL"
+ },
+ {
+ "id": 1855,
+ "origin_id": 1580,
+ "origin_slot": 0,
+ "target_id": 1578,
+ "target_slot": 0,
+ "type": "CONDITIONING"
+ },
+ {
+ "id": 1856,
+ "origin_id": 1577,
+ "origin_slot": 0,
+ "target_id": 1578,
+ "target_slot": 1,
+ "type": "LATENT"
+ },
+ {
+ "id": 1857,
+ "origin_id": 1574,
+ "origin_slot": 0,
+ "target_id": 1580,
+ "target_slot": 0,
+ "type": "CLIP"
+ },
+ {
+ "id": 1858,
+ "origin_id": 1575,
+ "origin_slot": 0,
+ "target_id": 1580,
+ "target_slot": 1,
+ "type": "VAE"
+ },
+ {
+ "id": 1859,
+ "origin_id": 1574,
+ "origin_slot": 0,
+ "target_id": 1573,
+ "target_slot": 0,
+ "type": "CLIP"
+ },
+ {
+ "id": 1860,
+ "origin_id": 1575,
+ "origin_slot": 0,
+ "target_id": 1573,
+ "target_slot": 1,
+ "type": "VAE"
+ },
+ {
+ "id": 1865,
+ "origin_id": 1575,
+ "origin_slot": 0,
+ "target_id": 1577,
+ "target_slot": 1,
+ "type": "VAE"
+ },
+ {
+ "id": 1866,
+ "origin_id": 1594,
+ "origin_slot": 0,
+ "target_id": 1593,
+ "target_slot": 0,
+ "type": "MODEL"
+ },
+ {
+ "id": 1874,
+ "origin_id": 1576,
+ "origin_slot": 0,
+ "target_id": 1580,
+ "target_slot": 2,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1875,
+ "origin_id": 1576,
+ "origin_slot": 0,
+ "target_id": 1573,
+ "target_slot": 2,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1876,
+ "origin_id": -10,
+ "origin_slot": 1,
+ "target_id": 1582,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1877,
+ "origin_id": 1582,
+ "origin_slot": 0,
+ "target_id": 1585,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 1878,
+ "origin_id": 1576,
+ "origin_slot": 0,
+ "target_id": 1577,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1879,
+ "origin_id": 1578,
+ "origin_slot": 0,
+ "target_id": 1570,
+ "target_slot": 1,
+ "type": "CONDITIONING"
+ },
+ {
+ "id": 1880,
+ "origin_id": 1579,
+ "origin_slot": 0,
+ "target_id": 1570,
+ "target_slot": 2,
+ "type": "CONDITIONING"
+ },
+ {
+ "id": 1881,
+ "origin_id": 1581,
+ "origin_slot": 0,
+ "target_id": 1586,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1883,
+ "origin_id": 1582,
+ "origin_slot": 0,
+ "target_id": 1590,
+ "target_slot": 0,
+ "type": "STRING"
+ },
+ {
+ "id": 1897,
+ "origin_id": 1583,
+ "origin_slot": 0,
+ "target_id": 1591,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1898,
+ "origin_id": 1591,
+ "origin_slot": 0,
+ "target_id": -20,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1900,
+ "origin_id": -10,
+ "origin_slot": 0,
+ "target_id": 1584,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1901,
+ "origin_id": 1584,
+ "origin_slot": 0,
+ "target_id": 1592,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1903,
+ "origin_id": 1592,
+ "origin_slot": 0,
+ "target_id": 1576,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1908,
+ "origin_id": 1584,
+ "origin_slot": 0,
+ "target_id": 1587,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1909,
+ "origin_id": 1585,
+ "origin_slot": 0,
+ "target_id": 1587,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1910,
+ "origin_id": 1586,
+ "origin_slot": 0,
+ "target_id": 1589,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1911,
+ "origin_id": 1590,
+ "origin_slot": 0,
+ "target_id": 1589,
+ "target_slot": 1,
+ "type": "STRING"
+ },
+ {
+ "id": 1912,
+ "origin_id": 1592,
+ "origin_slot": 0,
+ "target_id": 1588,
+ "target_slot": 0,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1913,
+ "origin_id": 1586,
+ "origin_slot": 0,
+ "target_id": 1588,
+ "target_slot": 1,
+ "type": "IMAGE"
+ },
+ {
+ "id": 1914,
+ "origin_id": 1587,
+ "origin_slot": 0,
+ "target_id": 1588,
+ "target_slot": 2,
+ "type": "STRING"
+ },
+ {
+ "id": 1915,
+ "origin_id": 1589,
+ "origin_slot": 0,
+ "target_id": 1588,
+ "target_slot": 3,
+ "type": "STRING"
+ },
+ {
+ "id": 1916,
+ "origin_id": 1588,
+ "origin_slot": 0,
+ "target_id": 1583,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1917,
+ "origin_id": 1588,
+ "origin_slot": 1,
+ "target_id": -20,
+ "target_slot": 1,
+ "type": "*"
+ },
+ {
+ "id": 1947,
+ "origin_id": -10,
+ "origin_slot": 2,
+ "target_id": 1595,
+ "target_slot": 0,
+ "type": "*"
+ },
+ {
+ "id": 1948,
+ "origin_id": 1595,
+ "origin_slot": 0,
+ "target_id": 1570,
+ "target_slot": 4,
+ "type": "INT"
+ }
+ ],
+ "extra": {}
+ }
+ ]
+ },
+ "config": {},
+ "extra": {
+ "workflowRendererVersion": "LG",
+ "ue_links": [],
+ "links_added_by_ue": [],
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true,
+ "ds": {
+ "scale": 0.7603121192389539,
+ "offset": [
+ 2009.2822506587647,
+ 1733.2788535228347
+ ]
+ }
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/ReconViaGen_MeshOnly.json b/example_workflows/ReconViaGen_MeshOnly.json
new file mode 100644
index 0000000..afdae1c
--- /dev/null
+++ b/example_workflows/ReconViaGen_MeshOnly.json
@@ -0,0 +1,1019 @@
+{
+ "id": "6eb9772a-caf3-4cd2-a1f3-8e21b2fcb356",
+ "revision": 0,
+ "last_node_id": 19,
+ "last_link_id": 36,
+ "nodes": [
+ {
+ "id": 2,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 860.5615222195892,
+ 320.28189087243953
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 1
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 3,
+ 7
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 1,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 818.1451367332586,
+ -43.32986975063045
+ ],
+ "size": [
+ 301.71875,
+ 226
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 2,
+ 6
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn",
+ true
+ ]
+ },
+ {
+ "id": 6,
+ "type": "Trellis2ImageCondGenerator",
+ "pos": [
+ 1283.125269298477,
+ -44.305212193867426
+ ],
+ "size": [
+ 308.0748046875,
+ 98
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 6
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 7
+ }
+ ],
+ "outputs": [
+ {
+ "name": "cond_512",
+ "type": "IMAGE_COND",
+ "links": null
+ },
+ {
+ "name": "cond_1024",
+ "type": "IMAGE_COND",
+ "links": [
+ 8,
+ 17
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ImageCondGenerator"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 4,
+ "type": "Trellis2SparseGeneratorWithReconViaGen",
+ "pos": [
+ 1257.1961548750526,
+ 122.64558954014163
+ ],
+ "size": [
+ 402.6431640625,
+ 314
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 2
+ },
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 3
+ }
+ ],
+ "outputs": [
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "links": [
+ 32
+ ]
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "links": [
+ 34
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 33
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2SparseGeneratorWithReconViaGen"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 25,
+ 7.5,
+ 0.01,
+ 5,
+ "euler",
+ 32,
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 5,
+ "type": "Trellis2ShapeGenerator",
+ "pos": [
+ 1741.2710306870335,
+ 171.79218707401608
+ ],
+ "size": [
+ 335.24609375,
+ 266
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 33
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 8
+ },
+ {
+ "name": "coords",
+ "type": "COORDS",
+ "link": 32
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 18
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 19
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 16
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ShapeGenerator"
+ },
+ "widgets_values": [
+ 512,
+ 25,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 8,
+ "type": "Trellis2ShapeCascadeGenerator",
+ "pos": [
+ 2144.5184531017253,
+ 89.08064130521795
+ ],
+ "size": [
+ 335.9791015625,
+ 338
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 16
+ },
+ {
+ "name": "image_cond",
+ "type": "IMAGE_COND",
+ "link": 17
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 18
+ },
+ {
+ "name": "from_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "from_resolution"
+ },
+ "link": 19
+ },
+ {
+ "name": "sparse_structure_resolution",
+ "type": "INT",
+ "widget": {
+ "name": "sparse_structure_resolution"
+ },
+ "link": 34
+ }
+ ],
+ "outputs": [
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "links": [
+ 11
+ ]
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "links": [
+ 12
+ ]
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 10
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ShapeCascadeGenerator"
+ },
+ "widgets_values": [
+ 0,
+ 1024,
+ 0,
+ 999999,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ "heun",
+ 0.1,
+ 1
+ ]
+ },
+ {
+ "id": 9,
+ "type": "Trellis2DecodeLatents",
+ "pos": [
+ 2550.5271755896983,
+ 85.2758108930415
+ ],
+ "size": [
+ 270,
+ 122
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 10
+ },
+ {
+ "name": "shape_slat",
+ "type": "SHAPE_SLAT",
+ "link": 11
+ },
+ {
+ "name": "texture_slat",
+ "shape": 7,
+ "type": "TEXTURE_SLAT",
+ "link": null
+ },
+ {
+ "name": "resolution",
+ "type": "INT",
+ "widget": {
+ "name": "resolution"
+ },
+ "link": 12
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 13
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": []
+ },
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2DecodeLatents"
+ },
+ "widgets_values": [
+ 0,
+ true
+ ]
+ },
+ {
+ "id": 10,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 2861.8616589898866,
+ 88.51255217697037
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 13
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 14
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 11,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 2847.559963718577,
+ 257.899269352995
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 14
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 15
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 12,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 2886.0029502911757,
+ 486.99086892176337
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 15
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 35
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 13,
+ "type": "Trellis2MeshWithVoxelToTrimesh",
+ "pos": [
+ 2840.8121187391353,
+ 764.4960805898208
+ ],
+ "size": [
+ 349.41171875,
+ 58
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 36
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 21
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ },
+ "widgets_values": [
+ "90 degrees"
+ ]
+ },
+ {
+ "id": 19,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 2889.6686029432403,
+ 637.1515820571416
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 4,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 35
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 36
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ }
+ },
+ {
+ "id": 14,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ 2878.801482335511,
+ 893.0809816705423
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 21
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 22
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "Ogre_VGGT_32",
+ "glb"
+ ]
+ },
+ {
+ "id": 15,
+ "type": "Preview3D",
+ "pos": [
+ 1271.3057615922355,
+ 522.7384991788329
+ ],
+ "size": [
+ 1051.7218022911484,
+ 1141.158091689083
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 22
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D",
+ "Camera Config": {
+ "cameraType": "perspective",
+ "fov": 75,
+ "state": {
+ "position": {
+ "x": 1.0315283207512576,
+ "y": 2.985009308965137,
+ "z": 4.0662863439166825
+ },
+ "target": {
+ "x": 0.20776096785142306,
+ "y": 2.467924964322487,
+ "z": -0.017876789746178047
+ },
+ "zoom": 1,
+ "cameraType": "perspective"
+ }
+ },
+ "Last Time Model File": "C:/Git/ComfyUI/output/Ogre_VGGT_32_00001_.glb",
+ "Resource Folder": "Git/ComfyUI/output",
+ "Scene Config": {
+ "showGrid": true,
+ "backgroundColor": "#282828",
+ "backgroundImage": "",
+ "backgroundRenderMode": "tiled"
+ },
+ "Light Config": {
+ "intensity": 3
+ },
+ "Model Config": {
+ "upDirection": "original",
+ "materialMode": "normal",
+ "showSkeleton": false
+ }
+ },
+ "widgets_values": [
+ "C:/Git/ComfyUI/output/Ogre_VGGT_32_00001_.glb",
+ ""
+ ]
+ },
+ {
+ "id": 3,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ 207.10417104607322,
+ 536.1883465110252
+ ],
+ "size": [
+ 1025.7731703829552,
+ 1112.7651147234092
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 1
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_2048_00106_.png",
+ "image"
+ ]
+ }
+ ],
+ "links": [
+ [
+ 1,
+ 3,
+ 2,
+ 2,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 2,
+ 1,
+ 0,
+ 4,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 3,
+ 2,
+ 0,
+ 4,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 6,
+ 1,
+ 0,
+ 6,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 7,
+ 2,
+ 0,
+ 6,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 8,
+ 6,
+ 1,
+ 5,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 10,
+ 8,
+ 2,
+ 9,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 11,
+ 8,
+ 0,
+ 9,
+ 1,
+ "SHAPE_SLAT"
+ ],
+ [
+ 12,
+ 8,
+ 1,
+ 9,
+ 3,
+ "INT"
+ ],
+ [
+ 13,
+ 9,
+ 0,
+ 10,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 14,
+ 10,
+ 0,
+ 11,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 15,
+ 11,
+ 0,
+ 12,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 16,
+ 5,
+ 2,
+ 8,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 17,
+ 6,
+ 1,
+ 8,
+ 1,
+ "IMAGE_COND"
+ ],
+ [
+ 18,
+ 5,
+ 0,
+ 8,
+ 2,
+ "SHAPE_SLAT"
+ ],
+ [
+ 19,
+ 5,
+ 1,
+ 8,
+ 3,
+ "INT"
+ ],
+ [
+ 21,
+ 13,
+ 0,
+ 14,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 22,
+ 14,
+ 0,
+ 15,
+ 2,
+ "STRING"
+ ],
+ [
+ 32,
+ 4,
+ 0,
+ 5,
+ 2,
+ "COORDS"
+ ],
+ [
+ 33,
+ 4,
+ 2,
+ 5,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 34,
+ 4,
+ 1,
+ 8,
+ 4,
+ "INT"
+ ],
+ [
+ 35,
+ 12,
+ 0,
+ 19,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 36,
+ 19,
+ 0,
+ 13,
+ 0,
+ "MESHWITHVOXEL"
+ ]
+ ],
+ "groups": [],
+ "config": {},
+ "extra": {
+ "ds": {
+ "scale": 0.5644739300537774,
+ "offset": [
+ -319.70457827331774,
+ 353.8321991591369
+ ]
+ },
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/RefineMesh.json b/example_workflows/RefineMesh.json
new file mode 100644
index 0000000..b604831
--- /dev/null
+++ b/example_workflows/RefineMesh.json
@@ -0,0 +1,810 @@
+{
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
+ "revision": 0,
+ "last_node_id": 233,
+ "last_link_id": 480,
+ "nodes": [
+ {
+ "id": 226,
+ "type": "Trellis2LoadMesh",
+ "pos": [
+ -385.0735510204828,
+ 158.88137472879367
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 460
+ ]
+ }
+ ],
+ "title": "Original Mesh",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2LoadMesh"
+ },
+ "widgets_values": [
+ ""
+ ]
+ },
+ {
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 432.4981118440155,
+ -182.40504275448276
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 461
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 418.8462464575081,
+ -492.4195525023097
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 469
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
+ ]
+ },
+ {
+ "id": 227,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1190.2002753188488,
+ -496.8180011267565
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 470
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 472
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 231,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 1203.103220151927,
+ -621.2330460050034
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 474
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 470
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 229,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1194.8431366936325,
+ -292.14794928709534
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 472
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 475
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 471
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 228,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1198.1583617137858,
+ -151.8665797237854
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 471
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 476
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 232,
+ "type": "Trellis2UnWrapAndRasterizer",
+ "pos": [
+ 1195.383895062337,
+ -50.23863923208595
+ ],
+ "size": [
+ 419.15234375,
+ 314
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 476
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "link": 477
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 478
+ ]
+ },
+ {
+ "name": "base_color_texture",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "metallic_roughness_texture",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2UnWrapAndRasterizer"
+ },
+ "widgets_values": [
+ 60,
+ 0,
+ 1,
+ 1,
+ 2048,
+ "OPAQUE",
+ false,
+ false,
+ false,
+ "telea"
+ ]
+ },
+ {
+ "id": 233,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ 1672.6947901006665,
+ -49.057670650420995
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 478
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 479
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 480
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 369.6388214873717,
+ 332.92876516342716
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 480
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
+ },
+ "widgets_values": [
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -383.4145219550909,
+ 310.9862093076388
+ ],
+ "size": [
+ 717.521556382955,
+ 915.5313594325766
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -379.85391723715725,
+ 7.869022281792564
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 475
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -388.041436029903,
+ -125.57562894305721
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 479
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 225,
+ "type": "Trellis2MeshRefiner",
+ "pos": [
+ 753.899038005419,
+ -389.13958558695583
+ ],
+ "size": [
+ 340.390625,
+ 578
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 469
+ },
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 460
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 461
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 474
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": [
+ 477
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshRefiner"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 1024,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ 12,
+ 3,
+ 0.01,
+ 3,
+ 999999,
+ true,
+ 16,
+ 0.1,
+ 1,
+ 0,
+ 0.9,
+ true,
+ 1,
+ "heun"
+ ]
+ }
+ ],
+ "links": [
+ [
+ 241,
+ 69,
+ 2,
+ 119,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 460,
+ 226,
+ 0,
+ 225,
+ 1,
+ "TRIMESH"
+ ],
+ [
+ 461,
+ 119,
+ 0,
+ 225,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 469,
+ 39,
+ 0,
+ 225,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 470,
+ 231,
+ 0,
+ 227,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 471,
+ 229,
+ 0,
+ 228,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 472,
+ 227,
+ 0,
+ 229,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 474,
+ 225,
+ 0,
+ 231,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 475,
+ 203,
+ 0,
+ 229,
+ 1,
+ "INT"
+ ],
+ [
+ 476,
+ 228,
+ 0,
+ 232,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 477,
+ 225,
+ 1,
+ 232,
+ 1,
+ "BVH"
+ ],
+ [
+ 478,
+ 232,
+ 0,
+ 233,
+ 0,
+ "TRIMESH"
+ ],
+ [
+ 479,
+ 204,
+ 0,
+ 233,
+ 1,
+ "STRING"
+ ],
+ [
+ 480,
+ 233,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -414.5659226642583,
+ -213.34811694305725,
+ 764.6855244821979,
+ 1466.2928336195214
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 374.5341225783577,
+ -726.2357215927766,
+ 1820.8266085195564,
+ 1004.7525102622308
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "config": {},
+ "extra": {
+ "workflowRendererVersion": "LG",
+ "ue_links": [],
+ "ds": {
+ "scale": 0.6830134553650716,
+ "offset": [
+ 111.46261093344825,
+ 859.8232012529436
+ ]
+ },
+ "links_added_by_ue": [],
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/RefineMesh_MeshOnly.json b/example_workflows/RefineMesh_MeshOnly.json
new file mode 100644
index 0000000..e79f714
--- /dev/null
+++ b/example_workflows/RefineMesh_MeshOnly.json
@@ -0,0 +1,776 @@
+{
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
+ "revision": 0,
+ "last_node_id": 234,
+ "last_link_id": 482,
+ "nodes": [
+ {
+ "id": 226,
+ "type": "Trellis2LoadMesh",
+ "pos": [
+ -385.0735510204828,
+ 158.88137472879367
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 460
+ ]
+ }
+ ],
+ "title": "Original Mesh",
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2LoadMesh"
+ },
+ "widgets_values": [
+ ""
+ ]
+ },
+ {
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 432.4981118440155,
+ -182.40504275448276
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 461
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
+ {
+ "id": 39,
+ "type": "Trellis2LoadModel",
+ "pos": [
+ 418.8462464575081,
+ -492.4195525023097
+ ],
+ "size": [
+ 301.71875,
+ 202
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "links": [
+ 469
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03",
+ "Node name for S&R": "Trellis2LoadModel",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "microsoft/TRELLIS.2-4B",
+ "flash_attn",
+ "cuda",
+ true,
+ false,
+ "flex_gemm",
+ "flash_attn"
+ ]
+ },
+ {
+ "id": 227,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1190.2002753188488,
+ -496.8180011267565
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 470
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 472
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 1,
+ 1024,
+ true,
+ true
+ ]
+ },
+ {
+ "id": 231,
+ "type": "Trellis2FillHolesWithCuMesh",
+ "pos": [
+ 1203.103220151927,
+ -621.2330460050034
+ ],
+ "size": [
+ 312.4361328125,
+ 58
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 474
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 470
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 229,
+ "type": "Trellis2SimplifyMesh",
+ "pos": [
+ 1194.8431366936325,
+ -292.14794928709534
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 472
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 475
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 471
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
+ },
+ "widgets_values": [
+ 500000,
+ "Cumesh"
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 369.6388214873717,
+ 332.92876516342716
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "camera_info",
+ "shape": 7,
+ "type": "LOAD3D_CAMERA",
+ "link": null
+ },
+ {
+ "name": "bg_image",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": null
+ },
+ {
+ "name": "model_file",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
+ "widget": {
+ "name": "model_file"
+ },
+ "link": 480
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
+ },
+ "widgets_values": [
+ "",
+ ""
+ ]
+ },
+ {
+ "id": 69,
+ "type": "Trellis2LoadImageWithTransparency",
+ "pos": [
+ -383.4145219550909,
+ 310.9862093076388
+ ],
+ "size": [
+ 717.521556382955,
+ 915.5313594325766
+ ],
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": []
+ },
+ {
+ "name": "image_with_alpha",
+ "type": "IMAGE",
+ "links": [
+ 241
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
+ "Node name for S&R": "Trellis2LoadImageWithTransparency",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ "Image_1024_00101_.png",
+ "image"
+ ]
+ },
+ {
+ "id": 203,
+ "type": "PrimitiveInt",
+ "pos": [
+ -379.85391723715725,
+ 7.869022281792564
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 475
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -388.041436029903,
+ -125.57562894305721
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 479
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 225,
+ "type": "Trellis2MeshRefiner",
+ "pos": [
+ 753.899038005419,
+ -389.13958558695583
+ ],
+ "size": [
+ 340.390625,
+ 578
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 469
+ },
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 460
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 461
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 474
+ ]
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "links": []
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshRefiner"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ 1024,
+ 12,
+ 7.5,
+ 0.01,
+ 3,
+ 12,
+ 3,
+ 0.01,
+ 3,
+ 999999,
+ false,
+ 16,
+ 0.1,
+ 1,
+ 0,
+ 0.9,
+ true,
+ 1,
+ "heun"
+ ]
+ },
+ {
+ "id": 228,
+ "type": "Trellis2FillHolesWithMeshlib",
+ "pos": [
+ 1198.1583617137858,
+ -151.8665797237854
+ ],
+ "size": [
+ 249.284765625,
+ 46
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 471
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 481
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 234,
+ "type": "Trellis2MeshWithVoxelToTrimesh",
+ "pos": [
+ 1183.7879296915492,
+ 4.235191654770113
+ ],
+ "size": [
+ 349.41171875,
+ 58
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 481
+ }
+ ],
+ "outputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "links": [
+ 482
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh"
+ },
+ "widgets_values": [
+ "90 degrees"
+ ]
+ },
+ {
+ "id": 233,
+ "type": "Trellis2ExportMesh",
+ "pos": [
+ 1205.1587971929546,
+ 133.46673644186396
+ ],
+ "size": [
+ 270,
+ 102
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 482
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 479
+ }
+ ],
+ "outputs": [
+ {
+ "name": "glb_path",
+ "type": "STRING",
+ "links": [
+ 480
+ ]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
+ },
+ "widgets_values": [
+ "3D/Trellis2",
+ "glb"
+ ]
+ }
+ ],
+ "links": [
+ [
+ 241,
+ 69,
+ 2,
+ 119,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 460,
+ 226,
+ 0,
+ 225,
+ 1,
+ "TRIMESH"
+ ],
+ [
+ 461,
+ 119,
+ 0,
+ 225,
+ 2,
+ "IMAGE"
+ ],
+ [
+ 469,
+ 39,
+ 0,
+ 225,
+ 0,
+ "TRELLIS2PIPELINE"
+ ],
+ [
+ 470,
+ 231,
+ 0,
+ 227,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 471,
+ 229,
+ 0,
+ 228,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 472,
+ 227,
+ 0,
+ 229,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 474,
+ 225,
+ 0,
+ 231,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 475,
+ 203,
+ 0,
+ 229,
+ 1,
+ "INT"
+ ],
+ [
+ 479,
+ 204,
+ 0,
+ 233,
+ 1,
+ "STRING"
+ ],
+ [
+ 480,
+ 233,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 481,
+ 228,
+ 0,
+ 234,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 482,
+ 234,
+ 0,
+ 233,
+ 0,
+ "TRIMESH"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -414.5659226642583,
+ -213.34811694305725,
+ 764.6855244821979,
+ 1466.2928336195214
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 374.5341225783577,
+ -726.2357215927766,
+ 1227.3780156118423,
+ 1005.7286067160883
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "config": {},
+ "extra": {
+ "workflowRendererVersion": "LG",
+ "ue_links": [],
+ "ds": {
+ "scale": 0.6830134553650716,
+ "offset": [
+ 200.2846180257333,
+ 818.8284012529438
+ ]
+ },
+ "links_added_by_ue": [],
+ "frontendVersion": "1.42.8",
+ "VHS_latentpreview": false,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/example_workflows/Simple.json b/example_workflows/Simple.json
index 7877040..18ecd1a 100644
--- a/example_workflows/Simple.json
+++ b/example_workflows/Simple.json
@@ -1,19 +1,19 @@
{
- "id": "440427ce-0c6a-4462-af51-639ed2f16dec",
+ "id": "cd6e2e00-83cc-4795-abf1-09b46428c270",
"revision": 0,
- "last_node_id": 44,
- "last_link_id": 90,
+ "last_node_id": 205,
+ "last_link_id": 413,
"nodes": [
{
- "id": 6,
+ "id": 69,
"type": "Trellis2LoadImageWithTransparency",
"pos": [
- 160.23097908593746,
- 1093.4781077920234
+ -377.72410571388025,
+ 328.4870830026766
],
"size": [
- 454.3125,
- 496.03125
+ 717.521556382955,
+ 915.5313594325766
],
"flags": {},
"order": 0,
@@ -23,18 +23,18 @@
{
"name": "image",
"type": "IMAGE",
- "links": null
+ "links": []
},
{
"name": "mask",
"type": "MASK",
- "links": null
+ "links": []
},
{
"name": "image_with_alpha",
"type": "IMAGE",
"links": [
- 89
+ 241
]
}
],
@@ -45,20 +45,62 @@
"widget_ue_connectable": {}
},
"widgets_values": [
- "Image_04291_.png",
+ "Image_1024_00101_.png",
"image"
]
},
+ {
+ "id": 119,
+ "type": "Trellis2PreProcessImage",
+ "pos": [
+ 398.0246741688026,
+ 149.83708875416676
+ ],
+ "size": [
+ 281.8229166666667,
+ 106
+ ],
+ "flags": {},
+ "order": 4,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 241
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 391
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "15854282a73cf231b81d52ada22e652f414a078b",
+ "Node name for S&R": "Trellis2PreProcessImage",
+ "widget_ue_connectable": {}
+ },
+ "widgets_values": [
+ 10,
+ false,
+ 1024
+ ]
+ },
{
"id": 39,
"type": "Trellis2LoadModel",
"pos": [
- 143.14495071454974,
- 622.610212818676
+ 388.892170207283,
+ -228.80616406738957
],
"size": [
- 362.0625,
- 207.328125
+ 301.71875,
+ 202
],
"flags": {},
"order": 1,
@@ -69,7 +111,7 @@
"name": "pipeline",
"type": "TRELLIS2PIPELINE",
"links": [
- 79
+ 390
]
}
],
@@ -80,37 +122,34 @@
"widget_ue_connectable": {}
},
"widgets_values": [
- "TRELLIS.2-4B",
+ "microsoft/TRELLIS.2-4B",
"flash_attn",
"cuda",
true,
- false
+ false,
+ "flex_gemm",
+ "flash_attn"
]
},
{
- "id": 41,
- "type": "Trellis2MeshWithVoxelGenerator",
+ "id": 196,
+ "type": "Trellis2FillHolesWithCuMesh",
"pos": [
- 673.7142651175388,
- 836.004782154444
+ 1117.157533597802,
+ -362.86738879994687
],
"size": [
- 410.71875,
- 386
+ 312.4361328125,
+ 58
],
"flags": {},
- "order": 3,
+ "order": 6,
"mode": 0,
"inputs": [
{
- "name": "pipeline",
- "type": "TRELLIS2PIPELINE",
- "link": 79
- },
- {
- "name": "image",
- "type": "IMAGE",
- "link": 90
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 396
}
],
"outputs": [
@@ -118,92 +157,192 @@
"name": "mesh",
"type": "MESHWITHVOXEL",
"links": [
- 86
- ]
- },
- {
- "name": "bvh",
- "type": "BVH",
- "links": [
- 87
+ 392
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a28d6cf93b0cf24aa9e7d3dccfd14f406227fbd0",
- "Node name for S&R": "Trellis2MeshWithVoxelGenerator",
+ "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af",
+ "Node name for S&R": "Trellis2FillHolesWithCuMesh"
+ },
+ "widgets_values": [
+ 1
+ ]
+ },
+ {
+ "id": 197,
+ "type": "Trellis2ReconstructMeshWithQuad",
+ "pos": [
+ 1124.8153675662377,
+ -239.75157745064854
+ ],
+ "size": [
+ 331.5878996659427,
+ 142.11360851901668
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 392
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 393
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5",
+ "Node name for S&R": "Trellis2ReconstructMeshWithQuad",
"widget_ue_connectable": {}
},
"widgets_values": [
- 12345,
- "randomize",
- "1024_cascade",
- 12,
- 12,
- 12,
- 999999,
- 4,
- 32,
+ 1,
+ 1024,
true,
true
]
},
{
- "id": 19,
- "type": "Trellis2ExportMesh",
+ "id": 198,
+ "type": "Trellis2SimplifyMesh",
"pos": [
- 1480.0390859207857,
- 603.8997701786345
+ 1134.132576655357,
+ -43.60066335397867
],
"size": [
- 324,
- 146
+ 270,
+ 82
],
"flags": {},
- "order": 5,
+ "order": 8,
"mode": 0,
"inputs": [
{
- "name": "trimesh",
- "type": "TRIMESH",
- "link": 88
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 393
+ },
+ {
+ "name": "target_face_num",
+ "type": "INT",
+ "widget": {
+ "name": "target_face_num"
+ },
+ "link": 409
}
],
"outputs": [
{
- "name": "glb_path",
- "type": "STRING",
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
"links": [
- 22
+ 394
]
}
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254",
- "Node name for S&R": "Trellis2ExportMesh",
- "widget_ue_connectable": {}
+ "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229",
+ "Node name for S&R": "Trellis2SimplifyMesh"
},
"widgets_values": [
- "Trellis2Mesh",
- "glb",
- true
+ 500000,
+ "Cumesh"
]
},
{
- "id": 10,
- "type": "Preview3D",
+ "id": 203,
+ "type": "PrimitiveInt",
"pos": [
- 1595.8257224427462,
- 828.6814144924502
+ -373.0898096413628,
+ 163.12227474974895
],
"size": [
- 752.25,
- 729.90625
+ 270,
+ 82
],
"flags": {},
- "order": 6,
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 409
+ ]
+ }
+ ],
+ "title": "Target Face Number",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveInt"
+ },
+ "widgets_values": [
+ 300000,
+ "fixed"
+ ]
+ },
+ {
+ "id": 204,
+ "type": "PrimitiveString",
+ "pos": [
+ -368.2858630795249,
+ 34.401714106564214
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "STRING",
+ "type": "STRING",
+ "links": [
+ 410
+ ]
+ }
+ ],
+ "title": "Name",
+ "properties": {
+ "cnr_id": "comfy-core",
+ "ver": "0.18.1",
+ "Node name for S&R": "PrimitiveString"
+ },
+ "widgets_values": [
+ "Tank"
+ ]
+ },
+ {
+ "id": 202,
+ "type": "Preview3D",
+ "pos": [
+ 376.48890096546415,
+ 331.2515281089069
+ ],
+ "size": [
+ 868.8731050199159,
+ 920.1382602693357
+ ],
+ "flags": {},
+ "order": 12,
"mode": 0,
"inputs": [
{
@@ -220,77 +359,152 @@
},
{
"name": "model_file",
- "type": "STRING",
+ "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D",
"widget": {
"name": "model_file"
},
- "link": 22
+ "link": 398
}
],
"outputs": [],
"properties": {
"cnr_id": "comfy-core",
- "ver": "0.4.0",
- "Node name for S&R": "Preview3D",
- "widget_ue_connectable": {},
- "Last Time Model File": "Home_2K_00001_.glb",
- "Scene Config": {
- "showGrid": true,
- "backgroundColor": "#282828",
- "backgroundImage": "",
- "backgroundRenderMode": "tiled"
- },
- "Camera Config": {
- "cameraType": "perspective",
- "fov": 35,
- "state": {
- "position": {
- "x": 12.514315832403488,
- "y": 3.895614510886169,
- "z": 9.139010516199487
- },
- "target": {
- "x": 0,
- "y": 2.1870527424637647,
- "z": 0
- },
- "zoom": 1,
- "cameraType": "perspective"
- }
- },
- "Light Config": {
- "intensity": 3
- }
+ "ver": "0.18.1",
+ "Node name for S&R": "Preview3D"
},
"widgets_values": [
- "Home_2K_00001_.glb",
+ "",
""
]
},
{
- "id": 43,
- "type": "Trellis2PostProcessAndUnWrapAndRasterizer",
+ "id": 199,
+ "type": "Trellis2FillHolesWithMeshlib",
"pos": [
- 1115.7105853634694,
- 836.2591783750445
+ 1132.506716841996,
+ 89.90828212755481
],
"size": [
- 419.125,
- 651.328125
+ 249.284765625,
+ 46
],
"flags": {},
- "order": 4,
+ "order": 9,
"mode": 0,
"inputs": [
{
"name": "mesh",
"type": "MESHWITHVOXEL",
- "link": 86
+ "link": 394
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 411
+ ]
+ },
+ {
+ "name": "holes_filled",
+ "type": "INT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985",
+ "Node name for S&R": "Trellis2FillHolesWithMeshlib"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 195,
+ "type": "Trellis2MeshWithVoxelGenerator",
+ "pos": [
+ 746.136770520365,
+ -107.27192057879526
+ ],
+ "size": [
+ 342.2818359375,
+ 342
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "pipeline",
+ "type": "TRELLIS2PIPELINE",
+ "link": 390
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 391
+ }
+ ],
+ "outputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "links": [
+ 396
+ ]
},
{
"name": "bvh",
"type": "BVH",
- "link": 87
+ "links": [
+ 412
+ ]
+ }
+ ],
+ "properties": {
+ "aux_id": "visualbruno/ComfyUI-Trellis2",
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2MeshWithVoxelGenerator"
+ },
+ "widgets_values": [
+ 12345,
+ "fixed",
+ "1024_cascade",
+ 12,
+ 12,
+ 12,
+ 999999,
+ 1,
+ 32,
+ true,
+ true,
+ "euler"
+ ]
+ },
+ {
+ "id": 205,
+ "type": "Trellis2UnWrapAndRasterizer",
+ "pos": [
+ 1509.48883255403,
+ -231.18851954550428
+ ],
+ "size": [
+ 419.15234375,
+ 314
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "mesh",
+ "type": "MESHWITHVOXEL",
+ "link": 411
+ },
+ {
+ "name": "bvh",
+ "type": "BVH",
+ "link": 412
}
],
"outputs": [
@@ -298,7 +512,7 @@
"name": "trimesh",
"type": "TRIMESH",
"links": [
- 88
+ 413
]
},
{
@@ -314,144 +528,223 @@
],
"properties": {
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72",
- "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer",
- "widget_ue_connectable": {}
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2UnWrapAndRasterizer"
},
"widgets_values": [
60,
0,
1,
1,
- 4096,
- true,
- 1,
- 0,
- 2000000,
- "Cumesh",
- true,
+ 2048,
"OPAQUE",
- "1024",
- false,
- true,
false,
false,
- true
+ false,
+ "telea"
]
},
{
- "id": 44,
- "type": "Trellis2PreProcessImage",
+ "id": 201,
+ "type": "Trellis2ExportMesh",
"pos": [
- 246.66298457453752,
- 916.7884835448142
+ 1517.3851156356834,
+ 145.96275340737535
],
"size": [
- 281.78125,
- 79.328125
+ 270,
+ 102
],
"flags": {},
- "order": 2,
+ "order": 11,
"mode": 0,
"inputs": [
{
- "name": "image",
- "type": "IMAGE",
- "link": 89
+ "name": "trimesh",
+ "type": "TRIMESH",
+ "link": 413
+ },
+ {
+ "name": "filename_prefix",
+ "type": "STRING",
+ "widget": {
+ "name": "filename_prefix"
+ },
+ "link": 410
}
],
"outputs": [
{
- "name": "image",
- "type": "IMAGE",
+ "name": "glb_path",
+ "type": "STRING",
"links": [
- 90
+ 398
]
+ },
+ {
+ "name": "relative_path",
+ "type": "STRING",
+ "links": null
}
],
"properties": {
- "widget_ue_connectable": {},
"aux_id": "visualbruno/ComfyUI-Trellis2",
- "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72",
- "Node name for S&R": "Trellis2PreProcessImage"
+ "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237",
+ "Node name for S&R": "Trellis2ExportMesh"
},
"widgets_values": [
- 0
+ "3D/Trellis2",
+ "glb"
]
}
],
"links": [
[
- 22,
- 19,
- 0,
- 10,
+ 241,
+ 69,
2,
- "STRING"
+ 119,
+ 0,
+ "IMAGE"
],
[
- 79,
+ 390,
39,
0,
- 41,
+ 195,
0,
"TRELLIS2PIPELINE"
],
[
- 86,
- 41,
+ 391,
+ 119,
0,
- 43,
+ 195,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 392,
+ 196,
+ 0,
+ 197,
0,
"MESHWITHVOXEL"
],
[
- 87,
- 41,
+ 393,
+ 197,
+ 0,
+ 198,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 394,
+ 198,
+ 0,
+ 199,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 396,
+ 195,
+ 0,
+ 196,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 398,
+ 201,
+ 0,
+ 202,
+ 2,
+ "STRING"
+ ],
+ [
+ 409,
+ 203,
+ 0,
+ 198,
1,
- 43,
+ "INT"
+ ],
+ [
+ 410,
+ 204,
+ 0,
+ 201,
+ 1,
+ "STRING"
+ ],
+ [
+ 411,
+ 199,
+ 0,
+ 205,
+ 0,
+ "MESHWITHVOXEL"
+ ],
+ [
+ 412,
+ 195,
+ 1,
+ 205,
1,
"BVH"
],
[
- 88,
- 43,
+ 413,
+ 205,
0,
- 19,
+ 201,
0,
"TRIMESH"
- ],
- [
- 89,
- 6,
- 2,
- 44,
- 0,
- "IMAGE"
- ],
- [
- 90,
- 44,
- 0,
- 41,
- 1,
- "IMAGE"
]
],
- "groups": [],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Configuration",
+ "bounding": [
+ -387.72410571388025,
+ -53.37077389343578,
+ 737.521556382955,
+ 1307.389216328689
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 2,
+ "title": "Generation",
+ "bounding": [
+ 378.892170207283,
+ -436.46738879994683,
+ 1706.6665797550602,
+ 727.1063315541132
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
"config": {},
"extra": {
- "workflowRendererVersion": "Vue",
+ "workflowRendererVersion": "LG",
"ue_links": [],
"ds": {
- "scale": 0.6303940863128564,
+ "scale": 0.5644739300537773,
"offset": [
- -52.07567851462492,
- -327.7388708088908
+ 683.341395587175,
+ 735.5254552429528
]
},
"links_added_by_ue": [],
- "frontendVersion": "1.37.11",
+ "frontendVersion": "1.42.8",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
diff --git a/nodes.py b/nodes.py
index 7a462c0..3813f39 100644
--- a/nodes.py
+++ b/nodes.py
@@ -331,6 +331,7 @@ class Trellis2LoadModel:
"keep_models_loaded": ("BOOLEAN", {"default":True}),
"conv_backend": (["spconv","torchsparse","flex_gemm"],{"default":"flex_gemm"}),
"sparse_backend": (["xformers","flash_attn"],{"default":"flash_attn"}),
+ "use_reconviagen": ("BOOLEAN",{"default":False}),
},
}
@@ -340,7 +341,9 @@ class Trellis2LoadModel:
CATEGORY = "Trellis2Wrapper"
OUTPUT_NODE = True
- def process(self, modelname, backend, device, low_vram, keep_models_loaded, conv_backend, sparse_backend):
+ def process(self, modelname, backend, device, low_vram, keep_models_loaded, conv_backend, sparse_backend, use_reconviagen):
+ import requests
+
os.environ['OPENCV_IO_ENABLE_OPENEXR'] = '1'
os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True" # Can save GPU memory
#os.environ["FLEX_GEMM_AUTOTUNE_CACHE_PATH"] = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'autotune_cache.json')
@@ -365,15 +368,19 @@ class Trellis2LoadModel:
local_dir=model_path,
local_dir_use_symlinks=False,
)
-
+
+ reconviagen_pipeline_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','reconviagen_pipeline.json')
+ if not os.path.exists(reconviagen_pipeline_file):
+ source_reconviagen_pipeline_file = os.path.join(script_directory,'reconviagen_pipeline.json')
+ shutil.copyfile(source_reconviagen_pipeline_file,reconviagen_pipeline_file)
+
dinov3_model_path = os.path.join(folder_paths.models_dir,"facebook","dinov3-vitl16-pretrain-lvd1689m","model.safetensors")
if not os.path.exists(dinov3_model_path):
raise Exception("Facebook Dinov3 model not found in models/facebook/dinov3-vitl16-pretrain-lvd1689m folder")
trellis_image_large_path = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS-image-large","ckpts","ss_dec_conv3d_16l8_fp16.safetensors")
if not os.path.exists(trellis_image_large_path):
- print('Trellis-Image-Large ss_dec_conv3d_16l8_fp16 files not found. Trying to download the files from huggingface ...')
- import requests
+ print('Trellis-Image-Large ss_dec_conv3d_16l8_fp16 files not found. Trying to download the files from huggingface ...')
url = "https://huggingface.co/microsoft/TRELLIS-image-large/resolve/main/ckpts/ss_dec_conv3d_16l8_fp16.json?download=true"
filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS-image-large","ckpts","ss_dec_conv3d_16l8_fp16.json")
path = Path(filename)
@@ -398,12 +405,79 @@ class Trellis2LoadModel:
else:
raise Exception("Cannot download Trellis-Image-Large file ss_dec_conv3d_16l8_fp16.safetensors")
+ if use_reconviagen:
+ reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_vggt_cond.safetensors')
+ if not os.path.exists(reconviagen_file):
+ print('ReconViaGen file ss_vggt_cond.safetensors not found. Trying to download the files from huggingface ...')
+ url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_vggt_cond.safetensors?download=true"
+ filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_vggt_cond.safetensors")
+ path = Path(filename)
+ path.parent.mkdir(parents=True, exist_ok=True)
+
+ response = requests.get(url)
+ if response.status_code == 200:
+ with open(filename, "wb") as f:
+ f.write(response.content)
+ print("Download ss_vggt_cond.safetensors complete!")
+ else:
+ raise Exception("Cannot download ReconViaGen file ss_vggt_cond.safetensors")
+
+ reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_vggt_cond.json')
+ if not os.path.exists(reconviagen_file):
+ print('ReconViaGen file ss_vggt_cond.json not found. Trying to download the files from huggingface ...')
+ url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_vggt_cond.json?download=true"
+ filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_vggt_cond.json")
+ path = Path(filename)
+ path.parent.mkdir(parents=True, exist_ok=True)
+
+ response = requests.get(url)
+ if response.status_code == 200:
+ with open(filename, "wb") as f:
+ f.write(response.content)
+ print("Download ss_vggt_cond.json complete!")
+ else:
+ raise Exception("Cannot download ReconViaGen file ss_vggt_cond.json")
+
+ reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_flow_img_dit_L_16l8_fp16.safetensors')
+ if not os.path.exists(reconviagen_file):
+ print('ReconViaGen file ss_flow_img_dit_L_16l8_fp16.safetensors not found. Trying to download the files from huggingface ...')
+ url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_flow_img_dit_L_16l8_fp16.safetensors?download=true"
+ filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_flow_img_dit_L_16l8_fp16.safetensors")
+ path = Path(filename)
+ path.parent.mkdir(parents=True, exist_ok=True)
+
+ response = requests.get(url)
+ if response.status_code == 200:
+ with open(filename, "wb") as f:
+ f.write(response.content)
+ print("Download ss_flow_img_dit_L_16l8_fp16.safetensors complete!")
+ else:
+ raise Exception("Cannot download ReconViaGen file ss_flow_img_dit_L_16l8_fp16.safetensors")
+
+ reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_flow_img_dit_L_16l8_fp16.json')
+ if not os.path.exists(reconviagen_file):
+ print('ReconViaGen file ss_flow_img_dit_L_16l8_fp16.json not found. Trying to download the files from huggingface ...')
+ url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_flow_img_dit_L_16l8_fp16.json?download=true"
+ filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_flow_img_dit_L_16l8_fp16.json")
+ path = Path(filename)
+ path.parent.mkdir(parents=True, exist_ok=True)
+
+ response = requests.get(url)
+ if response.status_code == 200:
+ with open(filename, "wb") as f:
+ f.write(response.content)
+ print("Download ss_flow_img_dit_L_16l8_fp16.json complete!")
+ else:
+ raise Exception("Cannot download ReconViaGen file ss_flow_img_dit_L_16l8_fp16.json")
+
if modelname == "visualbruno/TRELLIS.2-4B-FP8":
use_fp8 = True
+ if use_reconviagen:
+ raise Exception("ReconViaGen cannot be used with TRELLIS.2-4B-FP8. Select microsoft/TRELLIS.2-4B")
else:
use_fp8 = False
- pipeline = Trellis2ImageTo3DPipeline.from_pretrained(model_path, keep_models_loaded = keep_models_loaded, use_fp8=use_fp8)
+ pipeline = Trellis2ImageTo3DPipeline.from_pretrained(model_path, keep_models_loaded = keep_models_loaded, use_fp8=use_fp8, use_reconviagen=use_reconviagen)
pipeline.low_vram = low_vram
if device=="cuda":
@@ -4613,6 +4687,380 @@ class Trellis2VoxelToMesh:
return (mesh_copy,)
+class Trellis2UnloadAllModels:
+ @classmethod
+ def INPUT_TYPES(s):
+ return {
+ "required": {
+ "input_1": (any,)
+ },
+ }
+
+ RETURN_TYPES = (any,)
+ RETURN_NAMES = ("output_1",)
+ FUNCTION = "process"
+ CATEGORY = "Trellis2Wrapper"
+ OUTPUT_NODE = True
+
+ def process(self, input_1):
+ print('Unloading all models ...')
+ if hasattr(mm, 'current_loaded_models'):
+ # Iterate backwards to safely remove items
+ for i in range(len(mm.current_loaded_models) - 1, -1, -1):
+ loaded_model = mm.current_loaded_models[i]
+
+ print(f"[AbsoluteUnload] Force-killing: {loaded_model.model.model.__class__.__name__}")
+
+ # Force VRAM unload
+ loaded_model.model_unload(1e32)
+
+ # Force System RAM unpinning (This is what the standard loop skipped)
+ if hasattr(loaded_model.model, 'partially_unload_ram'):
+ loaded_model.model.partially_unload_ram(1e32)
+
+ # Clear ComfyUI's intermediate cross-attention and tensor caches
+ if hasattr(mm, 'current_loaded_models'):
+ mm.current_loaded_models.clear()
+
+ import comfy.controlnet
+ if hasattr(comfy.controlnet, 'controlnet_loaded_models'):
+ comfy.controlnet.controlnet_loaded_models.clear()
+
+ mm.free_memory(memory_required = 1e30,
+ device = mm.get_torch_device(),
+ ram_required = 1e30)
+
+ print('Clearing cache ...')
+ mm.soft_empty_cache()
+
+ gc.collect()
+ gc.collect()
+
+ if torch.cuda.is_available():
+ torch.cuda.empty_cache()
+ torch.cuda.ipc_collect()
+
+ print('Memory cleared')
+
+ return (input_1,)
+
+class Trellis2SparseGeneratorWithReconViaGen:
+ @classmethod
+ def INPUT_TYPES(s):
+ return {
+ "required": {
+ "pipeline": ("TRELLIS2PIPELINE",),
+ "images": ("IMAGE",),
+ "seed": ("INT", {"default": 0, "min": 0, "max": 0x7fffffff}),
+ "sparse_structure_steps": ("INT",{"default":12, "min":1, "max":100},),
+ "sparse_structure_guidance_strength": ("FLOAT",{"default":6.50,"min":0.00,"max":99.99,"step":0.01}),
+ "sparse_structure_guidance_rescale": ("FLOAT",{"default":0.05,"min":0.00,"max":1.00,"step":0.01}),
+ "sparse_structure_rescale_t": ("FLOAT",{"default":4.00,"min":0.00,"max":9.99,"step":0.01}),
+ "sparse_structure_sampler": (["euler", "heun", "rk4", "rk5"], {"default": "euler"}),
+ "sparse_structure_resolution": ("INT", {"default":32,"min":32,"max":128,"step":4}),
+ "sparse_structure_guidance_interval_start": ("FLOAT",{"default":0.10,"min":0.00,"max":1.00,"step":0.01}),
+ "sparse_structure_guidance_interval_end": ("FLOAT",{"default":1.00,"min":0.00,"max":1.00,"step":0.01}),
+ },
+ }
+
+ RETURN_TYPES = ("COORDS", "INT", "TRELLIS2PIPELINE",)
+ RETURN_NAMES = ("coords", "sparse_structure_resolution", "pipeline",)
+ FUNCTION = "process"
+ CATEGORY = "Trellis2Wrapper"
+ OUTPUT_NODE = True
+
+ def process(self, pipeline, images, seed,
+ # sparse
+ sparse_structure_steps,
+ sparse_structure_guidance_strength,
+ sparse_structure_guidance_rescale,
+ sparse_structure_rescale_t,
+ sparse_structure_sampler,
+ sparse_structure_resolution,
+ sparse_structure_guidance_interval_start,
+ sparse_structure_guidance_interval_end,
+ ):
+
+ self.seed_all(seed)
+
+ self.load_vggt_model(pipeline)
+
+ sparse_structure_guidance_interval = [sparse_structure_guidance_interval_start,sparse_structure_guidance_interval_end]
+ sparse_structure_sampler_params = {"steps":sparse_structure_steps,"guidance_strength":sparse_structure_guidance_strength,"guidance_rescale":sparse_structure_guidance_rescale,"guidance_interval":sparse_structure_guidance_interval,"rescale_t":sparse_structure_rescale_t}
+
+ args = pipeline._pretrained_args
+ sparse_sampler_prefix = pipeline.GetSamplerName(sparse_structure_sampler)
+ pipeline.sparse_structure_sampler = getattr(samplers, f"Flow{sparse_sampler_prefix}GuidanceIntervalSampler")(**args['sparse_structure_sampler']['args'])
+ pipeline.load_sparse_structure_vggt_model()
+ pipeline.load_sparse_structure_vggt_cond()
+
+ if images.ndim == 3:
+ images = images.unsqueeze(0)
+
+ coords = self._run_ss_stage_direct(pipeline, images, sparse_structure_resolution, sparse_structure_sampler_params)
+
+ if not pipeline.keep_models_loaded:
+ pipeline.unload_sparse_structure_vggt_model()
+ pipeline.unload_sparse_structure_vggt_cond()
+ self.unload_vggt_model(pipeline)
+
+ return (coords, sparse_structure_resolution, pipeline,)
+
+ def load_vggt_model(self, pipeline):
+ if pipeline.VGGT_model is None:
+ from .vggt.vggt.models.vggt import VGGT
+ pipeline.VGGT_dtype = torch.bfloat16 if torch.cuda.get_device_capability()[0] >= 8 else torch.float16
+ model_path = os.path.join(folder_paths.models_dir,'recongenvia')
+ pipeline.VGGT_model = VGGT.from_pretrained(model_path)
+ pipeline.VGGT_model.to('cuda')
+ del pipeline.VGGT_model.depth_head
+ del pipeline.VGGT_model.track_head
+ pipeline.VGGT_model.eval()
+
+ self._init_image_cond_model(pipeline)
+
+
+ def unload_vggt_model(self, pipeline):
+ del pipeline.VGGT_model
+ pipeline.VGGT_model = None
+
+ del pipeline.models['image_cond_model_vggt']
+ pipeline.models['image_cond_model_vggt'] = None
+ pipeline.image_cond_model_transform = None
+
+ gc.collect()
+
+ if torch.cuda.is_available():
+ torch.cuda.synchronize()
+ torch.cuda.empty_cache()
+
+
+ def seed_all(self, seed: int = 0):
+ import random
+ """
+ Set random seeds of all components.
+ """
+ random.seed(seed)
+ np.random.seed(seed)
+ torch.manual_seed(seed)
+ torch.cuda.manual_seed_all(seed)
+
+ @torch.no_grad()
+ def _run_ss_stage_direct(
+ self,
+ pipeline,
+ images,
+ target_ss_res: int,
+ ss_sampler_params: dict,
+ ) -> torch.Tensor:
+ """
+ Run only ReconViaGen's sparse structure diffusion stage to obtain coords
+ directly, without proceeding to the SLAT/mesh stage.
+
+ Returns:
+ coords : (N, 4) int tensor [batch_idx, x, y, z] in [0, target_ss_res)
+ """
+
+ cuda_device = torch.device('cuda')
+
+ if pipeline.low_vram:
+ pipeline.VGGT_model.to(cuda_device)
+
+ with torch.no_grad():
+ with torch.cuda.amp.autocast(dtype=pipeline.VGGT_dtype):
+ aggregated_tokens_list, _ = self.vggt_feat(pipeline, images)
+ b, n, _, _ = aggregated_tokens_list[0].shape
+ image_cond = self.encode_image(pipeline, images).reshape(b, n, -1, 1024)
+ ss_cond = self.get_ss_cond(pipeline, image_cond[:, :, 5:], aggregated_tokens_list, 1)
+
+ ss_flow_model = pipeline.models['sparse_structure_flow_vggt_model']
+ sampler_params = {**pipeline.sparse_structure_sampler_params, **ss_sampler_params}
+ reso = ss_flow_model.resolution
+ ss_noise = torch.randn(1, ss_flow_model.in_channels, reso, reso, reso).to(cuda_device)
+
+ with torch.autocast('cuda', dtype=torch.float16):
+ ss_latent = pipeline.sparse_structure_sampler.sample(
+ ss_flow_model,
+ ss_noise,
+ **ss_cond,
+ **sampler_params,
+ verbose=True,
+ ).samples
+
+ decoder = pipeline.models['sparse_structure_decoder']
+ decoded = decoder(ss_latent) > 0
+ if target_ss_res != decoded.shape[2]:
+ ratio = decoded.shape[2] // target_ss_res
+ decoded = torch.nn.functional.max_pool3d(decoded.float(), ratio, ratio, 0) > 0.5
+ coords = torch.argwhere(decoded)[:, [0, 2, 3, 4]].int()
+
+ if pipeline.low_vram:
+ pipeline.VGGT_model.to('cpu')
+ decoder.to('cpu')
+ ss_cond = pipeline._cond_cpu(ss_cond)
+ torch.cuda.empty_cache()
+
+ return coords
+
+ @torch.no_grad()
+ def _run_ss_stage(
+ self,
+ pipeline,
+ images,
+ target_ss_res: int,
+ ss_sampler_params: dict,
+ slat_sampler_params: dict,
+ ) -> torch.Tensor:
+ """
+ Generate a rough mesh via vggt_pipeline, then voxelise it into
+ surface-only coords at target_ss_res^3 for the downstream shape/tex stages.
+
+ Returns:
+ coords : (N, 4) int tensor [batch_idx, x, y, z] in [0, target_ss_res)
+ """
+ vp = self.vggt_pipeline
+ # vp.device is dynamic (inferred from model params), so when models are on
+ # CPU it returns 'cpu'. Hardcode the target cuda device instead.
+ cuda_device = torch.device('cuda')
+
+ if self.low_vram:
+ self._vggt_models_to(cuda_device)
+
+ outputs, _, _ = vp.run(
+ image=images,
+ formats=["mesh"],
+ preprocess_image=False,
+ sparse_structure_sampler_params=ss_sampler_params,
+ slat_sampler_params=slat_sampler_params,
+ )
+ mesh_result = outputs["mesh"][0]
+ coords = self._mesh_to_surface_coords(mesh_result, target_ss_res, cuda_device)
+
+ if self.low_vram:
+ self._vggt_models_to('cpu')
+ torch.cuda.empty_cache()
+
+ return coords
+
+ @torch.no_grad()
+ def vggt_feat(self, pipeline, image):
+ """
+ Encode the image.
+
+ Args:
+ image (Union[torch.Tensor, list[Image.Image]]): The image to encode
+
+ Returns:
+ torch.Tensor: The encoded features.
+ """
+ if isinstance(image, torch.Tensor):
+ assert image.ndim == 4, "Image tensor should be batched (B, H, W, C) or (B, C, H, W)"
+ # ComfyUI IMAGE tensors are (B, H, W, C); convert to (B, C, H, W)
+ if image.shape[-1] in (3, 4):
+ image = image.permute(0, 3, 1, 2)
+ image = F.interpolate(image, 518, mode='bilinear', align_corners=False)
+ image = image.to(pipeline.device)
+ elif isinstance(image, list):
+ assert all(isinstance(i, Image.Image) for i in image), "Image list should be list of PIL images"
+ image = [i.resize((518, 518), Image.LANCZOS) for i in image]
+ image = [np.array(i.convert('RGB')).astype(np.float32) / 255 for i in image]
+ image = [torch.from_numpy(i).permute(2, 0, 1).float() for i in image]
+ image = torch.stack(image).to(pipeline.device)
+ else:
+ raise ValueError(f"Unsupported type of image: {type(image)}")
+
+ with torch.no_grad():
+ with torch.cuda.amp.autocast(dtype=pipeline.VGGT_dtype):
+ # Predict attributes including cameras, depth maps, and point maps.
+ aggregated_tokens_list, _ = pipeline.VGGT_model.aggregator(image[None])
+
+ return aggregated_tokens_list, image
+
+ def get_ss_cond(self, pipeline, image_cond: torch.Tensor, aggregated_tokens_list: list, num_samples: int) -> dict:
+ """
+ Get the conditioning information for the model.
+
+ Args:
+ image (Union[torch.Tensor, list[Image.Image]]): The image prompts.
+
+ Returns:
+ dict: The conditioning information
+ """
+ cond = pipeline.models['sparse_structure_vggt_cond'](aggregated_tokens_list, image_cond)
+ neg_cond = torch.zeros_like(cond)
+ return {
+ 'cond': cond,
+ 'neg_cond': neg_cond,
+ }
+
+ def get_slat_cond(self, pipeline, image_cond: torch.Tensor, aggregated_tokens_list: list, num_samples: int) -> dict:
+ """
+ Get the conditioning information for the model.
+
+ Args:
+ image (Union[torch.Tensor, list[Image.Image]]): The image prompts.
+
+ Returns:
+ dict: The conditioning information
+ """
+ b, n, _, _ = aggregated_tokens_list[0].shape
+ cond = pipeline.models['slat_vggt_cond'](aggregated_tokens_list, image_cond).reshape(b, n, -1, 1024)
+ cond = [c.squeeze(1) for c in cond.split(1, dim=1)]
+ neg_cond = [torch.zeros_like(c) for c in cond]
+ return {
+ 'cond': cond,
+ 'neg_cond': neg_cond,
+ }
+
+ @torch.no_grad()
+ def encode_image(self, pipeline, image, w_layernorm=True) -> torch.Tensor:
+ """
+ Encode the image.
+
+ Args:
+ image (Union[torch.Tensor, list[Image.Image]]): The image to encode
+
+ Returns:
+ torch.Tensor: The encoded features.
+ """
+ if isinstance(image, torch.Tensor):
+ assert image.ndim == 4, "Image tensor should be batched (B, H, W, C) or (B, C, H, W)"
+ # ComfyUI IMAGE tensors are (B, H, W, C); convert to (B, C, H, W)
+ if image.shape[-1] in (3, 4):
+ image = image.permute(0, 3, 1, 2)
+ image = F.interpolate(image, 518, mode='bilinear', align_corners=False)
+ image = image.to(pipeline.device)
+ elif isinstance(image, list):
+ assert all(isinstance(i, Image.Image) for i in image), "Image list should be list of PIL images"
+ image = [i.resize((518, 518), Image.LANCZOS) for i in image]
+ image = [np.array(i.convert('RGB')).astype(np.float32) / 255 for i in image]
+ image = [torch.from_numpy(i).permute(2, 0, 1).float() for i in image]
+ image = torch.stack(image).to(pipeline.device)
+ else:
+ raise ValueError(f"Unsupported type of image: {type(image)}")
+
+ image = pipeline.image_cond_model_transform(image).to(pipeline.device)
+ pipeline.models['image_cond_model_vggt'].to(pipeline.device)
+ features = pipeline.models['image_cond_model_vggt'](image, is_training=True)['x_prenorm']
+ if w_layernorm:
+ features = F.layer_norm(features, features.shape[-1:])
+ return features
+
+ def _init_image_cond_model(self, pipeline, name: str = "dinov2_vitl14_reg"):
+ """
+ Initialize the image conditioning model.
+ """
+ try:
+ dinov2_model = torch.hub.load(os.path.join(torch.hub.get_dir(), 'facebookresearch_dinov2_main'), name, source='local',pretrained=True)
+ except:
+ dinov2_model = torch.hub.load('facebookresearch/dinov2', name, pretrained=True)
+ dinov2_model.eval()
+ pipeline.models['image_cond_model_vggt'] = dinov2_model
+ transform = transforms.Compose([
+ transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),
+ ])
+ pipeline.image_cond_model_transform = transform
+
NODE_CLASS_MAPPINGS = {
"Trellis2LoadModel": Trellis2LoadModel,
"Trellis2MeshWithVoxelGenerator": Trellis2MeshWithVoxelGenerator,
@@ -4668,6 +5116,8 @@ NODE_CLASS_MAPPINGS = {
"Trellis2RenderMultiView": Trellis2RenderMultiView,
"Trellis2SaveImage": Trellis2SaveImage,
"Trellis2VoxelToMesh": Trellis2VoxelToMesh,
+ "Trellis2UnloadAllModels": Trellis2UnloadAllModels,
+ "Trellis2SparseGeneratorWithReconViaGen": Trellis2SparseGeneratorWithReconViaGen,
}
@@ -4726,4 +5176,6 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"Trellis2RenderMultiView": "Trellis2 - Render MultiView",
"Trellis2SaveImage": "Trellis2 - Save Image",
"Trellis2VoxelToMesh": "Trellis2 - Voxel to Mesh",
+ "Trellis2UnloadAllModels": "Trellis2 - Unload All ComfyUI Models",
+ "Trellis2SparseGeneratorWithReconViaGen": "Trellis2 - Sparse Generator with ReconViaGen",
}
diff --git a/pyproject.toml b/pyproject.toml
index d940e8d..570f346 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -1,7 +1,7 @@
[project]
name = "trellis2"
description = "ComfyUI Wrapper for Microsoft Trellis.2 - Native and Compact Structured Latents for 3D Generation"
-version = "1.0.19"
+version = "1.0.20"
license = {file = "LICENSE"}
# classifiers = [
# # For OS-independent nodes (works on all operating systems)
diff --git a/reconviagen_pipeline.json b/reconviagen_pipeline.json
new file mode 100644
index 0000000..d5a6aaa
--- /dev/null
+++ b/reconviagen_pipeline.json
@@ -0,0 +1,98 @@
+{
+ "name": "Trellis2ImageTo3DPipeline",
+ "args": {
+ "models": {
+ "sparse_structure_decoder": "microsoft/TRELLIS-image-large/ckpts/ss_dec_conv3d_16l8_fp16",
+ "sparse_structure_flow_model": "ckpts/ss_flow_img_dit_1_3B_64_bf16",
+ "shape_slat_decoder": "ckpts/shape_dec_next_dc_f16c32_fp16",
+ "shape_slat_flow_model_512": "ckpts/slat_flow_img2shape_dit_1_3B_512_bf16",
+ "shape_slat_flow_model_1024": "ckpts/slat_flow_img2shape_dit_1_3B_1024_bf16",
+ "tex_slat_decoder": "ckpts/tex_dec_next_dc_f16c32_fp16",
+ "tex_slat_flow_model_512": "ckpts/slat_flow_imgshape2tex_dit_1_3B_512_bf16",
+ "tex_slat_flow_model_1024": "ckpts/slat_flow_imgshape2tex_dit_1_3B_1024_bf16",
+ "sparse_structure_vggt_cond": "ckpts/ss_vggt_cond",
+ "slat_vggt_cond": "ckpts/slat_vggt_cond",
+ "sparse_structure_flow_vggt_model": "ckpts/ss_flow_img_dit_L_16l8_fp16"
+ },
+ "sparse_structure_sampler": {
+ "name": "FlowEulerGuidanceIntervalSampler",
+ "args": {
+ "sigma_min": 1e-5
+ },
+ "params": {
+ "steps": 12,
+ "guidance_strength": 7.5,
+ "guidance_rescale": 0.7,
+ "guidance_interval": [0.3, 1.0],
+ "rescale_t": 5.0
+ }
+ },
+ "shape_slat_sampler": {
+ "name": "FlowEulerGuidanceIntervalSampler",
+ "args": {
+ "sigma_min": 1e-5
+ },
+ "params": {
+ "steps": 12,
+ "guidance_strength": 7.5,
+ "guidance_rescale": 0.5,
+ "guidance_interval": [0.3, 1.0],
+ "rescale_t": 3.0
+ }
+ },
+ "shape_slat_normalization": {
+ "mean": [
+ 0.781296, 0.018091, -0.495192, -0.558457, 1.060530, 0.093252, 1.518149, -0.933218,
+ -0.732996, 2.604095, -0.118341, -2.143904, 0.495076, -2.179512, -2.130751, -0.996944,
+ 0.261421, -2.217463, 1.260067, -0.150213, 3.790713, 1.481266, -1.046058, -1.523667,
+ -0.059621, 2.220780, 1.621212, 0.877230, 0.567247, -3.175944, -3.186688, 1.578665
+ ],
+ "std": [
+ 5.972266, 4.706852, 5.445010, 5.209927, 5.320220, 4.547237, 5.020802, 5.444004,
+ 5.226681, 5.683095, 4.831436, 5.286469, 5.652043, 5.367606, 5.525084, 4.730578,
+ 4.805265, 5.124013, 5.530808, 5.619001, 5.103930, 5.417670, 5.269677, 5.547194,
+ 5.634698, 5.235274, 6.110351, 5.511298, 6.237273, 4.879207, 5.347008, 5.405691
+ ]
+ },
+ "tex_slat_sampler": {
+ "name": "FlowEulerGuidanceIntervalSampler",
+ "args": {
+ "sigma_min": 1e-5
+ },
+ "params": {
+ "steps": 12,
+ "guidance_strength": 1.0,
+ "guidance_rescale": 0.0,
+ "guidance_interval": [0.6, 0.9],
+ "rescale_t": 3.0
+ }
+ },
+ "tex_slat_normalization": {
+ "mean": [
+ 3.501659, 2.212398, 2.226094, 0.251093, -0.026248, -0.687364, 0.439898, -0.928075,
+ 0.029398, -0.339596, -0.869527, 1.038479, -0.972385, 0.126042, -1.129303, 0.455149,
+ -1.209521, 2.069067, 0.544735, 2.569128, -0.323407, 2.293000, -1.925608, -1.217717,
+ 1.213905, 0.971588, -0.023631, 0.106750, 2.021786, 0.250524, -0.662387, -0.768862
+ ],
+ "std": [
+ 2.665652, 2.743913, 2.765121, 2.595319, 3.037293, 2.291316, 2.144656, 2.911822,
+ 2.969419, 2.501689, 2.154811, 3.163343, 2.621215, 2.381943, 3.186697, 3.021588,
+ 2.295916, 3.234985, 3.233086, 2.260140, 2.874801, 2.810596, 3.292720, 2.674999,
+ 2.680878, 2.372054, 2.451546, 2.353556, 2.995195, 2.379849, 2.786195, 2.775190
+ ]
+ },
+ "image_cond_model": {
+ "name": "DinoV3FeatureExtractor",
+ "args": {
+ "model_name": "facebook/dinov3-vitl16-pretrain-lvd1689m"
+ }
+ },
+ "rembg_model": {
+ "name": "BiRefNet",
+ "args": {
+ "model_name": "briaai/RMBG-2.0"
+ }
+ },
+ "default_pipeline_type": "1024_cascade"
+ }
+}
\ No newline at end of file
diff --git a/trellis2/models/__init__.py b/trellis2/models/__init__.py
index d4fed03..889b6c7 100644
--- a/trellis2/models/__init__.py
+++ b/trellis2/models/__init__.py
@@ -14,7 +14,10 @@ __attributes = {
'SparseUnetVaeEncoder': 'sc_vaes.sparse_unet_vae',
'SparseUnetVaeDecoder': 'sc_vaes.sparse_unet_vae',
'FlexiDualGridVaeEncoder': 'sc_vaes.fdg_vae',
- 'FlexiDualGridVaeDecoder': 'sc_vaes.fdg_vae'
+ 'FlexiDualGridVaeDecoder': 'sc_vaes.fdg_vae',
+
+ # vggt
+ 'ModulatedMultiViewCond': 'sparse_structure_flow',
}
__submodules = []
diff --git a/trellis2/models/sparse_structure_flow.py b/trellis2/models/sparse_structure_flow.py
index 5c38c18..60baf1c 100644
--- a/trellis2/models/sparse_structure_flow.py
+++ b/trellis2/models/sparse_structure_flow.py
@@ -4,8 +4,8 @@ import torch
import torch.nn as nn
import torch.nn.functional as F
import numpy as np
-from ..modules.utils import convert_module_to, manual_cast, str_to_dtype
-from ..modules.transformer import AbsolutePositionEmbedder, ModulatedTransformerCrossBlock
+from ..modules.utils import convert_module_to, manual_cast, str_to_dtype, convert_module_to_f16
+from ..modules.transformer import AbsolutePositionEmbedder, ModulatedTransformerCrossBlock, ModulatedTransformerCrossBlock_woT
from ..modules.attention import RotaryPositionEmbedder
@@ -247,3 +247,76 @@ class SparseStructureFlowModel(nn.Module):
h = h.permute(0, 2, 1).view(h.shape[0], h.shape[2], *[self.resolution] * 3).contiguous()
return h
+
+class ModulatedMultiViewCond(nn.Module):
+ """
+ Transformer cross-attention block (MSA + MCA + FFN) with adaptive layer norm conditioning.
+ """
+ def __init__(
+ self,
+ channels: int,
+ ctx_channels: int,
+ num_heads: int,
+ mlp_ratio: float = 4.0,
+ attn_mode: Literal["full", "windowed"] = "full",
+ window_size: Optional[int] = None,
+ shift_window: Optional[Tuple[int, int, int]] = None,
+ use_checkpoint: bool = False,
+ use_rope: bool = False,
+ qk_rms_norm: bool = False,
+ qk_rms_norm_cross: bool = False,
+ qkv_bias: bool = True,
+ share_mod: bool = False,
+ num_init_tokens: int = 4096,
+ dtype: Optional[torch.dtype] = torch.float32,
+ use_fp16: bool = False,
+ ):
+ super().__init__()
+ self.cond_blocks = nn.ModuleList([
+ ModulatedTransformerCrossBlock_woT(
+ channels,
+ ctx_channels,
+ num_heads=num_heads,
+ mlp_ratio=mlp_ratio,
+ attn_mode=attn_mode,
+ use_checkpoint=use_checkpoint,
+ use_rope=use_rope,
+ share_mod=share_mod,
+ qk_rms_norm=qk_rms_norm,
+ qk_rms_norm_cross=qk_rms_norm_cross,
+ )
+ for _ in range(4)
+ ])
+ self.use_fp16 = use_fp16
+ if use_fp16:
+ self.dtype = torch.float16
+ else:
+ self.dtype = dtype
+ self.multiview_cond_tokens = nn.Parameter(torch.randn(1, num_init_tokens, channels).to(dtype))
+ nn.init.normal_(self.multiview_cond_tokens, std=1e-6)
+ self.intermediate_layer_idx = [4, 11, 17, 23]
+ if use_fp16:
+ self.convert_to_fp16()
+
+
+ def convert_to_fp16(self) -> None:
+ """
+ Convert the torso of the model to float16.
+ """
+ self.use_fp16 = True
+ self.dtype = torch.float16
+ self.cond_blocks.apply(convert_module_to_f16)
+ self.multiview_cond_tokens = nn.Parameter(self.multiview_cond_tokens.data.to(self.dtype))
+ def forward(self, aggregated_tokens_list: List, image_cond: torch.Tensor):
+
+ b = aggregated_tokens_list[0].shape[0]
+ patch_start_idx = 5
+ idx = 0
+ cond = self.multiview_cond_tokens.repeat(b, 1, 1)
+ for layer_idx in self.intermediate_layer_idx:
+ x = aggregated_tokens_list[layer_idx][:, :, patch_start_idx:]
+ # x = x.reshape(b, -1, 2048) + torch.cat([image_cond.reshape(b, -1, 1024), image_cond.reshape(b, -1, 1024)],dim=-1)
+ x = torch.cat([x.reshape(b, -1, 2048), image_cond.reshape(b, -1, 1024)],dim=-1).to(self.dtype)
+ cond = self.cond_blocks[idx](cond, x)
+ idx = idx + 1
+ return cond
\ No newline at end of file
diff --git a/trellis2/modules/transformer/modulated.py b/trellis2/modules/transformer/modulated.py
index 3c56eda..e2f6923 100644
--- a/trellis2/modules/transformer/modulated.py
+++ b/trellis2/modules/transformer/modulated.py
@@ -5,7 +5,6 @@ from ..attention import MultiHeadAttention
from ..norm import LayerNorm32
from .blocks import FeedForwardNet
-
class ModulatedTransformerBlock(nn.Module):
"""
Transformer block (MSA + FFN) with adaptive layer norm conditioning.
@@ -162,4 +161,73 @@ class ModulatedTransformerCrossBlock(nn.Module):
return torch.utils.checkpoint.checkpoint(self._forward, x, mod, context, phases, use_reentrant=False)
else:
return self._forward(x, mod, context, phases)
-
\ No newline at end of file
+
+class ModulatedTransformerCrossBlock_woT(nn.Module):
+ """
+ Transformer cross-attention block (MSA + MCA + FFN) with adaptive layer norm conditioning.
+ """
+ def __init__(
+ self,
+ channels: int,
+ ctx_channels: int,
+ num_heads: int,
+ mlp_ratio: float = 4.0,
+ attn_mode: Literal["full", "windowed"] = "full",
+ window_size: Optional[int] = None,
+ shift_window: Optional[Tuple[int, int, int]] = None,
+ use_checkpoint: bool = False,
+ use_rope: bool = False,
+ qk_rms_norm: bool = False,
+ qk_rms_norm_cross: bool = False,
+ qkv_bias: bool = True,
+ share_mod: bool = False,
+ ):
+ super().__init__()
+ self.use_checkpoint = use_checkpoint
+ self.share_mod = share_mod
+ self.norm1 = LayerNorm32(channels, elementwise_affine=False, eps=1e-6)
+ self.norm2 = LayerNorm32(channels, elementwise_affine=True, eps=1e-6)
+ self.norm3 = LayerNorm32(channels, elementwise_affine=False, eps=1e-6)
+ self.self_attn = MultiHeadAttention(
+ channels,
+ num_heads=num_heads,
+ type="self",
+ attn_mode=attn_mode,
+ window_size=window_size,
+ shift_window=shift_window,
+ qkv_bias=qkv_bias,
+ use_rope=use_rope,
+ qk_rms_norm=qk_rms_norm,
+ )
+ self.cross_attn = MultiHeadAttention(
+ channels,
+ ctx_channels=ctx_channels,
+ num_heads=num_heads,
+ type="cross",
+ attn_mode="full",
+ qkv_bias=qkv_bias,
+ qk_rms_norm=qk_rms_norm_cross,
+ )
+ self.mlp = FeedForwardNet(
+ channels,
+ mlp_ratio=mlp_ratio,
+ )
+
+ def _forward(self, x: torch.Tensor, context: torch.Tensor):
+
+ h = self.norm1(x)
+ h = self.self_attn(h)
+ x = x + h
+ h = self.norm2(x)
+ h = self.cross_attn(h, context)
+ x = x + h
+ h = self.norm3(x)
+ h = self.mlp(h)
+ x = x + h
+ return x
+
+ def forward(self, x: torch.Tensor, context: torch.Tensor):
+ if self.use_checkpoint:
+ return torch.utils.checkpoint.checkpoint(self._forward, x, context, use_reentrant=False)
+ else:
+ return self._forward(x, context)
\ No newline at end of file
diff --git a/trellis2/pipelines/trellis2_image_to_3d.py b/trellis2/pipelines/trellis2_image_to_3d.py
index 10d02d7..595b914 100644
--- a/trellis2/pipelines/trellis2_image_to_3d.py
+++ b/trellis2/pipelines/trellis2_image_to_3d.py
@@ -26,6 +26,8 @@ import random
from comfy.utils import ProgressBar
+script_directory = os.path.dirname(os.path.abspath(__file__))
+
def pil2tensor(image):
return torch.from_numpy(np.array(image).astype(np.float32) / 255.0)[None,]
@@ -98,6 +100,7 @@ class Trellis2ImageTo3DPipeline(Pipeline):
self.rembg_model = rembg_model
self._low_vram = low_vram
self.default_pipeline_type = default_pipeline_type
+ self.VGGT_model = None
self.pbr_attr_layout = {
'base_color': slice(0, 3),
'metallic': slice(3, 4),
@@ -148,14 +151,16 @@ class Trellis2ImageTo3DPipeline(Pipeline):
torch.cuda.empty_cache()
@classmethod
- def from_pretrained(cls, path: str, config_file: str = "pipeline.json", keep_models_loaded = True, use_fp8 = False) -> "Trellis2ImageTo3DPipeline":
+ def from_pretrained(cls, path: str, config_file: str = "pipeline.json", keep_models_loaded = True, use_fp8 = False, use_reconviagen = False) -> "Trellis2ImageTo3DPipeline":
"""
Load a pretrained model.
Args:
path (str): The path to the model. Can be either local path or a Hugging Face repository.
"""
- if use_fp8:
+ if use_reconviagen:
+ config_file = "reconviagen_pipeline.json"
+ elif use_fp8:
config_file = "pipeline_fp8.json"
pipeline = super().from_pretrained(path, config_file)
@@ -212,6 +217,45 @@ class Trellis2ImageTo3DPipeline(Pipeline):
self.models['sparse_structure_decoder'].to(self._device)
if hasattr(self.models['sparse_structure_decoder'], 'low_vram'):
self.models['sparse_structure_decoder'].low_vram = self.low_vram
+
+ def load_sparse_structure_vggt_model(self):
+ if self.models['sparse_structure_flow_vggt_model'] is None:
+ print('Loading Sparse Structure VGGT model ...')
+ self.models['sparse_structure_flow_vggt_model'] = models.from_pretrained(f"{self.path}/{self._pretrained_args['models']['sparse_structure_flow_vggt_model']}")
+ self.models['sparse_structure_flow_vggt_model'].eval()
+ self.models['sparse_structure_flow_vggt_model'].to(self._device)
+
+ if self.models['sparse_structure_decoder'] is None:
+ self.models['sparse_structure_decoder'] = models.from_pretrained(self._pretrained_args['models']['sparse_structure_decoder'])
+ self.models['sparse_structure_decoder'].eval()
+ self.models['sparse_structure_decoder'].to(self._device)
+ if hasattr(self.models['sparse_structure_decoder'], 'low_vram'):
+ self.models['sparse_structure_decoder'].low_vram = self.low_vram
+
+ def unload_sparse_structure_vggt_model(self):
+ if self.models['sparse_structure_flow_vggt_model']:
+ del self.models['sparse_structure_flow_vggt_model']
+ self.models['sparse_structure_flow_vggt_model'] = None
+
+ if self.models['sparse_structure_decoder']:
+ del self.models['sparse_structure_decoder']
+ self.models['sparse_structure_decoder'] = None
+
+ self._cleanup_cuda()
+
+ def load_sparse_structure_vggt_cond(self):
+ if self.models['sparse_structure_vggt_cond'] is None:
+ print('Loading Sparse Structure VGGT cond ...')
+ self.models['sparse_structure_vggt_cond'] = models.from_pretrained(f"{self.path}/{self._pretrained_args['models']['sparse_structure_vggt_cond']}")
+ self.models['sparse_structure_vggt_cond'].eval()
+ self.models['sparse_structure_vggt_cond'].to(self._device)
+
+ def unload_sparse_structure_vggt_cond(self):
+ if self.models['sparse_structure_vggt_cond']:
+ del self.models['sparse_structure_vggt_cond']
+ self.models['sparse_structure_vggt_cond'] = None
+
+ self._cleanup_cuda()
def unload_sparse_structure_model(self):
if self.models['sparse_structure_flow_model']:
diff --git a/vggt/vggt/heads/camera_head.py b/vggt/vggt/heads/camera_head.py
new file mode 100644
index 0000000..153b020
--- /dev/null
+++ b/vggt/vggt/heads/camera_head.py
@@ -0,0 +1,162 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import math
+import numpy as np
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+from ..layers import Mlp
+from ..layers.block import Block
+from ..heads.head_act import activate_pose
+
+
+class CameraHead(nn.Module):
+ """
+ CameraHead predicts camera parameters from token representations using iterative refinement.
+
+ It applies a series of transformer blocks (the "trunk") to dedicated camera tokens.
+ """
+
+ def __init__(
+ self,
+ dim_in: int = 2048,
+ trunk_depth: int = 4,
+ pose_encoding_type: str = "absT_quaR_FoV",
+ num_heads: int = 16,
+ mlp_ratio: int = 4,
+ init_values: float = 0.01,
+ trans_act: str = "linear",
+ quat_act: str = "linear",
+ fl_act: str = "relu", # Field of view activations: ensures FOV values are positive.
+ ):
+ super().__init__()
+
+ if pose_encoding_type == "absT_quaR_FoV":
+ self.target_dim = 9
+ else:
+ raise ValueError(f"Unsupported camera encoding type: {pose_encoding_type}")
+
+ self.trans_act = trans_act
+ self.quat_act = quat_act
+ self.fl_act = fl_act
+ self.trunk_depth = trunk_depth
+
+ # Build the trunk using a sequence of transformer blocks.
+ self.trunk = nn.Sequential(
+ *[
+ Block(
+ dim=dim_in,
+ num_heads=num_heads,
+ mlp_ratio=mlp_ratio,
+ init_values=init_values,
+ )
+ for _ in range(trunk_depth)
+ ]
+ )
+
+ # Normalizations for camera token and trunk output.
+ self.token_norm = nn.LayerNorm(dim_in)
+ self.trunk_norm = nn.LayerNorm(dim_in)
+
+ # Learnable empty camera pose token.
+ self.empty_pose_tokens = nn.Parameter(torch.zeros(1, 1, self.target_dim))
+ self.embed_pose = nn.Linear(self.target_dim, dim_in)
+
+ # Module for producing modulation parameters: shift, scale, and a gate.
+ self.poseLN_modulation = nn.Sequential(nn.SiLU(), nn.Linear(dim_in, 3 * dim_in, bias=True))
+
+ # Adaptive layer normalization without affine parameters.
+ self.adaln_norm = nn.LayerNorm(dim_in, elementwise_affine=False, eps=1e-6)
+ self.pose_branch = Mlp(
+ in_features=dim_in,
+ hidden_features=dim_in // 2,
+ out_features=self.target_dim,
+ drop=0,
+ )
+
+ def forward(self, aggregated_tokens_list: list, num_iterations: int = 4) -> list:
+ """
+ Forward pass to predict camera parameters.
+
+ Args:
+ aggregated_tokens_list (list): List of token tensors from the network;
+ the last tensor is used for prediction.
+ num_iterations (int, optional): Number of iterative refinement steps. Defaults to 4.
+
+ Returns:
+ list: A list of predicted camera encodings (post-activation) from each iteration.
+ """
+ # Use tokens from the last block for camera prediction.
+ tokens = aggregated_tokens_list[-1]
+
+ # Extract the camera tokens
+ pose_tokens = tokens[:, :, 0]
+ pose_tokens = self.token_norm(pose_tokens)
+
+ pred_pose_enc_list = self.trunk_fn(pose_tokens, num_iterations)
+ return pred_pose_enc_list
+
+ def trunk_fn(self, pose_tokens: torch.Tensor, num_iterations: int) -> list:
+ """
+ Iteratively refine camera pose predictions.
+
+ Args:
+ pose_tokens (torch.Tensor): Normalized camera tokens with shape [B, 1, C].
+ num_iterations (int): Number of refinement iterations.
+
+ Returns:
+ list: List of activated camera encodings from each iteration.
+ """
+ B, S, C = pose_tokens.shape # S is expected to be 1.
+ pred_pose_enc = None
+ pred_pose_enc_list = []
+
+ for _ in range(num_iterations):
+ # Use a learned empty pose for the first iteration.
+ if pred_pose_enc is None:
+ module_input = self.embed_pose(self.empty_pose_tokens.expand(B, S, -1))
+ else:
+ # Detach the previous prediction to avoid backprop through time.
+ pred_pose_enc = pred_pose_enc.detach()
+ module_input = self.embed_pose(pred_pose_enc)
+
+ # Generate modulation parameters and split them into shift, scale, and gate components.
+ shift_msa, scale_msa, gate_msa = self.poseLN_modulation(module_input).chunk(3, dim=-1)
+
+ # Adaptive layer normalization and modulation.
+ pose_tokens_modulated = gate_msa * modulate(self.adaln_norm(pose_tokens), shift_msa, scale_msa)
+ pose_tokens_modulated = pose_tokens_modulated + pose_tokens
+
+ pose_tokens_modulated = self.trunk(pose_tokens_modulated)
+ # Compute the delta update for the pose encoding.
+ pred_pose_enc_delta = self.pose_branch(self.trunk_norm(pose_tokens_modulated))
+
+ if pred_pose_enc is None:
+ pred_pose_enc = pred_pose_enc_delta
+ else:
+ pred_pose_enc = pred_pose_enc + pred_pose_enc_delta
+
+ # Apply final activation functions for translation, quaternion, and field-of-view.
+ activated_pose = activate_pose(
+ pred_pose_enc,
+ trans_act=self.trans_act,
+ quat_act=self.quat_act,
+ fl_act=self.fl_act,
+ )
+ pred_pose_enc_list.append(activated_pose)
+
+ return pred_pose_enc_list
+
+
+def modulate(x: torch.Tensor, shift: torch.Tensor, scale: torch.Tensor) -> torch.Tensor:
+ """
+ Modulate the input tensor using scaling and shifting parameters.
+ """
+ # modified from https://github.com/facebookresearch/DiT/blob/796c29e532f47bba17c5b9c5eb39b9354b8b7c64/models.py#L19
+ return x * (1 + scale) + shift
diff --git a/vggt/vggt/heads/dpt_head.py b/vggt/vggt/heads/dpt_head.py
new file mode 100644
index 0000000..c8c8af9
--- /dev/null
+++ b/vggt/vggt/heads/dpt_head.py
@@ -0,0 +1,497 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+
+# Inspired by https://github.com/DepthAnything/Depth-Anything-V2
+
+
+import os
+from typing import List, Dict, Tuple, Union
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from .head_act import activate_head
+from .utils import create_uv_grid, position_grid_to_embed
+
+
+class DPTHead(nn.Module):
+ """
+ DPT Head for dense prediction tasks.
+
+ This implementation follows the architecture described in "Vision Transformers for Dense Prediction"
+ (https://arxiv.org/abs/2103.13413). The DPT head processes features from a vision transformer
+ backbone and produces dense predictions by fusing multi-scale features.
+
+ Args:
+ dim_in (int): Input dimension (channels).
+ patch_size (int, optional): Patch size. Default is 14.
+ output_dim (int, optional): Number of output channels. Default is 4.
+ activation (str, optional): Activation type. Default is "inv_log".
+ conf_activation (str, optional): Confidence activation type. Default is "expp1".
+ features (int, optional): Feature channels for intermediate representations. Default is 256.
+ out_channels (List[int], optional): Output channels for each intermediate layer.
+ intermediate_layer_idx (List[int], optional): Indices of layers from aggregated tokens used for DPT.
+ pos_embed (bool, optional): Whether to use positional embedding. Default is True.
+ feature_only (bool, optional): If True, return features only without the last several layers and activation head. Default is False.
+ down_ratio (int, optional): Downscaling factor for the output resolution. Default is 1.
+ """
+
+ def __init__(
+ self,
+ dim_in: int,
+ patch_size: int = 14,
+ output_dim: int = 4,
+ activation: str = "inv_log",
+ conf_activation: str = "expp1",
+ features: int = 256,
+ out_channels: List[int] = [256, 512, 1024, 1024],
+ intermediate_layer_idx: List[int] = [4, 11, 17, 23],
+ pos_embed: bool = True,
+ feature_only: bool = False,
+ down_ratio: int = 1,
+ ) -> None:
+ super(DPTHead, self).__init__()
+ self.patch_size = patch_size
+ self.activation = activation
+ self.conf_activation = conf_activation
+ self.pos_embed = pos_embed
+ self.feature_only = feature_only
+ self.down_ratio = down_ratio
+ self.intermediate_layer_idx = intermediate_layer_idx
+
+ self.norm = nn.LayerNorm(dim_in)
+
+ # Projection layers for each output channel from tokens.
+ self.projects = nn.ModuleList(
+ [
+ nn.Conv2d(
+ in_channels=dim_in,
+ out_channels=oc,
+ kernel_size=1,
+ stride=1,
+ padding=0,
+ )
+ for oc in out_channels
+ ]
+ )
+
+ # Resize layers for upsampling feature maps.
+ self.resize_layers = nn.ModuleList(
+ [
+ nn.ConvTranspose2d(
+ in_channels=out_channels[0], out_channels=out_channels[0], kernel_size=4, stride=4, padding=0
+ ),
+ nn.ConvTranspose2d(
+ in_channels=out_channels[1], out_channels=out_channels[1], kernel_size=2, stride=2, padding=0
+ ),
+ nn.Identity(),
+ nn.Conv2d(
+ in_channels=out_channels[3], out_channels=out_channels[3], kernel_size=3, stride=2, padding=1
+ ),
+ ]
+ )
+
+ self.scratch = _make_scratch(
+ out_channels,
+ features,
+ expand=False,
+ )
+
+ # Attach additional modules to scratch.
+ self.scratch.stem_transpose = None
+ self.scratch.refinenet1 = _make_fusion_block(features)
+ self.scratch.refinenet2 = _make_fusion_block(features)
+ self.scratch.refinenet3 = _make_fusion_block(features)
+ self.scratch.refinenet4 = _make_fusion_block(features, has_residual=False)
+
+ head_features_1 = features
+ head_features_2 = 32
+
+ if feature_only:
+ self.scratch.output_conv1 = nn.Conv2d(head_features_1, head_features_1, kernel_size=3, stride=1, padding=1)
+ else:
+ self.scratch.output_conv1 = nn.Conv2d(
+ head_features_1, head_features_1 // 2, kernel_size=3, stride=1, padding=1
+ )
+ conv2_in_channels = head_features_1 // 2
+
+ self.scratch.output_conv2 = nn.Sequential(
+ nn.Conv2d(conv2_in_channels, head_features_2, kernel_size=3, stride=1, padding=1),
+ nn.ReLU(inplace=True),
+ nn.Conv2d(head_features_2, output_dim, kernel_size=1, stride=1, padding=0),
+ )
+
+ def forward(
+ self,
+ aggregated_tokens_list: List[torch.Tensor],
+ images: torch.Tensor,
+ patch_start_idx: int,
+ frames_chunk_size: int = 8,
+ ) -> Union[torch.Tensor, Tuple[torch.Tensor, torch.Tensor]]:
+ """
+ Forward pass through the DPT head, supports processing by chunking frames.
+ Args:
+ aggregated_tokens_list (List[Tensor]): List of token tensors from different transformer layers.
+ images (Tensor): Input images with shape [B, S, 3, H, W], in range [0, 1].
+ patch_start_idx (int): Starting index for patch tokens in the token sequence.
+ Used to separate patch tokens from other tokens (e.g., camera or register tokens).
+ frames_chunk_size (int, optional): Number of frames to process in each chunk.
+ If None or larger than S, all frames are processed at once. Default: 8.
+
+ Returns:
+ Tensor or Tuple[Tensor, Tensor]:
+ - If feature_only=True: Feature maps with shape [B, S, C, H, W]
+ - Otherwise: Tuple of (predictions, confidence) both with shape [B, S, 1, H, W]
+ """
+ B, S, _, H, W = images.shape
+
+ # If frames_chunk_size is not specified or greater than S, process all frames at once
+ if frames_chunk_size is None or frames_chunk_size >= S:
+ return self._forward_impl(aggregated_tokens_list, images, patch_start_idx)
+
+ # Otherwise, process frames in chunks to manage memory usage
+ assert frames_chunk_size > 0
+
+ # Process frames in batches
+ all_preds = []
+ all_conf = []
+
+ for frames_start_idx in range(0, S, frames_chunk_size):
+ frames_end_idx = min(frames_start_idx + frames_chunk_size, S)
+
+ # Process batch of frames
+ if self.feature_only:
+ chunk_output = self._forward_impl(
+ aggregated_tokens_list, images, patch_start_idx, frames_start_idx, frames_end_idx
+ )
+ all_preds.append(chunk_output)
+ else:
+ chunk_preds, chunk_conf = self._forward_impl(
+ aggregated_tokens_list, images, patch_start_idx, frames_start_idx, frames_end_idx
+ )
+ all_preds.append(chunk_preds)
+ all_conf.append(chunk_conf)
+
+ # Concatenate results along the sequence dimension
+ if self.feature_only:
+ return torch.cat(all_preds, dim=1)
+ else:
+ return torch.cat(all_preds, dim=1), torch.cat(all_conf, dim=1)
+
+ def _forward_impl(
+ self,
+ aggregated_tokens_list: List[torch.Tensor],
+ images: torch.Tensor,
+ patch_start_idx: int,
+ frames_start_idx: int = None,
+ frames_end_idx: int = None,
+ ) -> Union[torch.Tensor, Tuple[torch.Tensor, torch.Tensor]]:
+ """
+ Implementation of the forward pass through the DPT head.
+
+ This method processes a specific chunk of frames from the sequence.
+
+ Args:
+ aggregated_tokens_list (List[Tensor]): List of token tensors from different transformer layers.
+ images (Tensor): Input images with shape [B, S, 3, H, W].
+ patch_start_idx (int): Starting index for patch tokens.
+ frames_start_idx (int, optional): Starting index for frames to process.
+ frames_end_idx (int, optional): Ending index for frames to process.
+
+ Returns:
+ Tensor or Tuple[Tensor, Tensor]: Feature maps or (predictions, confidence).
+ """
+ if frames_start_idx is not None and frames_end_idx is not None:
+ images = images[:, frames_start_idx:frames_end_idx]
+
+ B, S, _, H, W = images.shape
+
+ patch_h, patch_w = H // self.patch_size, W // self.patch_size
+
+ out = []
+ dpt_idx = 0
+
+ for layer_idx in self.intermediate_layer_idx:
+ x = aggregated_tokens_list[layer_idx][:, :, patch_start_idx:]
+
+ # Select frames if processing a chunk
+ if frames_start_idx is not None and frames_end_idx is not None:
+ x = x[:, frames_start_idx:frames_end_idx]
+
+ x = x.view(B * S, -1, x.shape[-1])
+
+ x = self.norm(x)
+
+ x = x.permute(0, 2, 1).reshape((x.shape[0], x.shape[-1], patch_h, patch_w))
+
+ x = self.projects[dpt_idx](x)
+ if self.pos_embed:
+ x = self._apply_pos_embed(x, W, H)
+ x = self.resize_layers[dpt_idx](x)
+
+ out.append(x)
+ dpt_idx += 1
+
+ # Fuse features from multiple layers.
+ out = self.scratch_forward(out)
+ # Interpolate fused output to match target image resolution.
+ out = custom_interpolate(
+ out,
+ (int(patch_h * self.patch_size / self.down_ratio), int(patch_w * self.patch_size / self.down_ratio)),
+ mode="bilinear",
+ align_corners=True,
+ )
+
+ if self.pos_embed:
+ out = self._apply_pos_embed(out, W, H)
+
+ if self.feature_only:
+ return out.view(B, S, *out.shape[1:])
+
+ out = self.scratch.output_conv2(out)
+ preds, conf = activate_head(out, activation=self.activation, conf_activation=self.conf_activation)
+
+ preds = preds.view(B, S, *preds.shape[1:])
+ conf = conf.view(B, S, *conf.shape[1:])
+ return preds, conf
+
+ def _apply_pos_embed(self, x: torch.Tensor, W: int, H: int, ratio: float = 0.1) -> torch.Tensor:
+ """
+ Apply positional embedding to tensor x.
+ """
+ patch_w = x.shape[-1]
+ patch_h = x.shape[-2]
+ pos_embed = create_uv_grid(patch_w, patch_h, aspect_ratio=W / H, dtype=x.dtype, device=x.device)
+ pos_embed = position_grid_to_embed(pos_embed, x.shape[1])
+ pos_embed = pos_embed * ratio
+ pos_embed = pos_embed.permute(2, 0, 1)[None].expand(x.shape[0], -1, -1, -1)
+ return x + pos_embed
+
+ def scratch_forward(self, features: List[torch.Tensor]) -> torch.Tensor:
+ """
+ Forward pass through the fusion blocks.
+
+ Args:
+ features (List[Tensor]): List of feature maps from different layers.
+
+ Returns:
+ Tensor: Fused feature map.
+ """
+ layer_1, layer_2, layer_3, layer_4 = features
+
+ layer_1_rn = self.scratch.layer1_rn(layer_1)
+ layer_2_rn = self.scratch.layer2_rn(layer_2)
+ layer_3_rn = self.scratch.layer3_rn(layer_3)
+ layer_4_rn = self.scratch.layer4_rn(layer_4)
+
+ out = self.scratch.refinenet4(layer_4_rn, size=layer_3_rn.shape[2:])
+ del layer_4_rn, layer_4
+
+ out = self.scratch.refinenet3(out, layer_3_rn, size=layer_2_rn.shape[2:])
+ del layer_3_rn, layer_3
+
+ out = self.scratch.refinenet2(out, layer_2_rn, size=layer_1_rn.shape[2:])
+ del layer_2_rn, layer_2
+
+ out = self.scratch.refinenet1(out, layer_1_rn)
+ del layer_1_rn, layer_1
+
+ out = self.scratch.output_conv1(out)
+ return out
+
+
+################################################################################
+# Modules
+################################################################################
+
+
+def _make_fusion_block(features: int, size: int = None, has_residual: bool = True, groups: int = 1) -> nn.Module:
+ return FeatureFusionBlock(
+ features,
+ nn.ReLU(inplace=True),
+ deconv=False,
+ bn=False,
+ expand=False,
+ align_corners=True,
+ size=size,
+ has_residual=has_residual,
+ groups=groups,
+ )
+
+
+def _make_scratch(in_shape: List[int], out_shape: int, groups: int = 1, expand: bool = False) -> nn.Module:
+ scratch = nn.Module()
+ out_shape1 = out_shape
+ out_shape2 = out_shape
+ out_shape3 = out_shape
+ if len(in_shape) >= 4:
+ out_shape4 = out_shape
+
+ if expand:
+ out_shape1 = out_shape
+ out_shape2 = out_shape * 2
+ out_shape3 = out_shape * 4
+ if len(in_shape) >= 4:
+ out_shape4 = out_shape * 8
+
+ scratch.layer1_rn = nn.Conv2d(
+ in_shape[0], out_shape1, kernel_size=3, stride=1, padding=1, bias=False, groups=groups
+ )
+ scratch.layer2_rn = nn.Conv2d(
+ in_shape[1], out_shape2, kernel_size=3, stride=1, padding=1, bias=False, groups=groups
+ )
+ scratch.layer3_rn = nn.Conv2d(
+ in_shape[2], out_shape3, kernel_size=3, stride=1, padding=1, bias=False, groups=groups
+ )
+ if len(in_shape) >= 4:
+ scratch.layer4_rn = nn.Conv2d(
+ in_shape[3], out_shape4, kernel_size=3, stride=1, padding=1, bias=False, groups=groups
+ )
+ return scratch
+
+
+class ResidualConvUnit(nn.Module):
+ """Residual convolution module."""
+
+ def __init__(self, features, activation, bn, groups=1):
+ """Init.
+
+ Args:
+ features (int): number of features
+ """
+ super().__init__()
+
+ self.bn = bn
+ self.groups = groups
+ self.conv1 = nn.Conv2d(features, features, kernel_size=3, stride=1, padding=1, bias=True, groups=self.groups)
+ self.conv2 = nn.Conv2d(features, features, kernel_size=3, stride=1, padding=1, bias=True, groups=self.groups)
+
+ self.norm1 = None
+ self.norm2 = None
+
+ self.activation = activation
+ self.skip_add = nn.quantized.FloatFunctional()
+
+ def forward(self, x):
+ """Forward pass.
+
+ Args:
+ x (tensor): input
+
+ Returns:
+ tensor: output
+ """
+
+ out = self.activation(x)
+ out = self.conv1(out)
+ if self.norm1 is not None:
+ out = self.norm1(out)
+
+ out = self.activation(out)
+ out = self.conv2(out)
+ if self.norm2 is not None:
+ out = self.norm2(out)
+
+ return self.skip_add.add(out, x)
+
+
+class FeatureFusionBlock(nn.Module):
+ """Feature fusion block."""
+
+ def __init__(
+ self,
+ features,
+ activation,
+ deconv=False,
+ bn=False,
+ expand=False,
+ align_corners=True,
+ size=None,
+ has_residual=True,
+ groups=1,
+ ):
+ """Init.
+
+ Args:
+ features (int): number of features
+ """
+ super(FeatureFusionBlock, self).__init__()
+
+ self.deconv = deconv
+ self.align_corners = align_corners
+ self.groups = groups
+ self.expand = expand
+ out_features = features
+ if self.expand == True:
+ out_features = features // 2
+
+ self.out_conv = nn.Conv2d(
+ features, out_features, kernel_size=1, stride=1, padding=0, bias=True, groups=self.groups
+ )
+
+ if has_residual:
+ self.resConfUnit1 = ResidualConvUnit(features, activation, bn, groups=self.groups)
+
+ self.has_residual = has_residual
+ self.resConfUnit2 = ResidualConvUnit(features, activation, bn, groups=self.groups)
+
+ self.skip_add = nn.quantized.FloatFunctional()
+ self.size = size
+
+ def forward(self, *xs, size=None):
+ """Forward pass.
+
+ Returns:
+ tensor: output
+ """
+ output = xs[0]
+
+ if self.has_residual:
+ res = self.resConfUnit1(xs[1])
+ output = self.skip_add.add(output, res)
+
+ output = self.resConfUnit2(output)
+
+ if (size is None) and (self.size is None):
+ modifier = {"scale_factor": 2}
+ elif size is None:
+ modifier = {"size": self.size}
+ else:
+ modifier = {"size": size}
+
+ output = custom_interpolate(output, **modifier, mode="bilinear", align_corners=self.align_corners)
+ output = self.out_conv(output)
+
+ return output
+
+
+def custom_interpolate(
+ x: torch.Tensor,
+ size: Tuple[int, int] = None,
+ scale_factor: float = None,
+ mode: str = "bilinear",
+ align_corners: bool = True,
+) -> torch.Tensor:
+ """
+ Custom interpolate to avoid INT_MAX issues in nn.functional.interpolate.
+ """
+ if size is None:
+ size = (int(x.shape[-2] * scale_factor), int(x.shape[-1] * scale_factor))
+
+ INT_MAX = 1610612736
+
+ input_elements = size[0] * size[1] * x.shape[0] * x.shape[1]
+
+ if input_elements > INT_MAX:
+ chunks = torch.chunk(x, chunks=(input_elements // INT_MAX) + 1, dim=0)
+ interpolated_chunks = [
+ nn.functional.interpolate(chunk, size=size, mode=mode, align_corners=align_corners) for chunk in chunks
+ ]
+ x = torch.cat(interpolated_chunks, dim=0)
+ return x.contiguous()
+ else:
+ return nn.functional.interpolate(x, size=size, mode=mode, align_corners=align_corners)
diff --git a/vggt/vggt/heads/head_act.py b/vggt/vggt/heads/head_act.py
new file mode 100644
index 0000000..2dedfcf
--- /dev/null
+++ b/vggt/vggt/heads/head_act.py
@@ -0,0 +1,125 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+
+import torch
+import torch.nn.functional as F
+
+
+def activate_pose(pred_pose_enc, trans_act="linear", quat_act="linear", fl_act="linear"):
+ """
+ Activate pose parameters with specified activation functions.
+
+ Args:
+ pred_pose_enc: Tensor containing encoded pose parameters [translation, quaternion, focal length]
+ trans_act: Activation type for translation component
+ quat_act: Activation type for quaternion component
+ fl_act: Activation type for focal length component
+
+ Returns:
+ Activated pose parameters tensor
+ """
+ T = pred_pose_enc[..., :3]
+ quat = pred_pose_enc[..., 3:7]
+ fl = pred_pose_enc[..., 7:] # or fov
+
+ T = base_pose_act(T, trans_act)
+ quat = base_pose_act(quat, quat_act)
+ fl = base_pose_act(fl, fl_act) # or fov
+
+ pred_pose_enc = torch.cat([T, quat, fl], dim=-1)
+
+ return pred_pose_enc
+
+
+def base_pose_act(pose_enc, act_type="linear"):
+ """
+ Apply basic activation function to pose parameters.
+
+ Args:
+ pose_enc: Tensor containing encoded pose parameters
+ act_type: Activation type ("linear", "inv_log", "exp", "relu")
+
+ Returns:
+ Activated pose parameters
+ """
+ if act_type == "linear":
+ return pose_enc
+ elif act_type == "inv_log":
+ return inverse_log_transform(pose_enc)
+ elif act_type == "exp":
+ return torch.exp(pose_enc)
+ elif act_type == "relu":
+ return F.relu(pose_enc)
+ else:
+ raise ValueError(f"Unknown act_type: {act_type}")
+
+
+def activate_head(out, activation="norm_exp", conf_activation="expp1"):
+ """
+ Process network output to extract 3D points and confidence values.
+
+ Args:
+ out: Network output tensor (B, C, H, W)
+ activation: Activation type for 3D points
+ conf_activation: Activation type for confidence values
+
+ Returns:
+ Tuple of (3D points tensor, confidence tensor)
+ """
+ # Move channels from last dim to the 4th dimension => (B, H, W, C)
+ fmap = out.permute(0, 2, 3, 1) # B,H,W,C expected
+
+ # Split into xyz (first C-1 channels) and confidence (last channel)
+ xyz = fmap[:, :, :, :-1]
+ conf = fmap[:, :, :, -1]
+
+ if activation == "norm_exp":
+ d = xyz.norm(dim=-1, keepdim=True).clamp(min=1e-8)
+ xyz_normed = xyz / d
+ pts3d = xyz_normed * torch.expm1(d)
+ elif activation == "norm":
+ pts3d = xyz / xyz.norm(dim=-1, keepdim=True)
+ elif activation == "exp":
+ pts3d = torch.exp(xyz)
+ elif activation == "relu":
+ pts3d = F.relu(xyz)
+ elif activation == "inv_log":
+ pts3d = inverse_log_transform(xyz)
+ elif activation == "xy_inv_log":
+ xy, z = xyz.split([2, 1], dim=-1)
+ z = inverse_log_transform(z)
+ pts3d = torch.cat([xy * z, z], dim=-1)
+ elif activation == "sigmoid":
+ pts3d = torch.sigmoid(xyz)
+ elif activation == "linear":
+ pts3d = xyz
+ else:
+ raise ValueError(f"Unknown activation: {activation}")
+
+ if conf_activation == "expp1":
+ conf_out = 1 + conf.exp()
+ elif conf_activation == "expp0":
+ conf_out = conf.exp()
+ elif conf_activation == "sigmoid":
+ conf_out = torch.sigmoid(conf)
+ else:
+ raise ValueError(f"Unknown conf_activation: {conf_activation}")
+
+ return pts3d, conf_out
+
+
+def inverse_log_transform(y):
+ """
+ Apply inverse log transform: sign(y) * (exp(|y|) - 1)
+
+ Args:
+ y: Input tensor
+
+ Returns:
+ Transformed tensor
+ """
+ return torch.sign(y) * (torch.expm1(torch.abs(y)))
diff --git a/vggt/vggt/heads/track_head.py b/vggt/vggt/heads/track_head.py
new file mode 100644
index 0000000..9ec7199
--- /dev/null
+++ b/vggt/vggt/heads/track_head.py
@@ -0,0 +1,108 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch.nn as nn
+from .dpt_head import DPTHead
+from .track_modules.base_track_predictor import BaseTrackerPredictor
+
+
+class TrackHead(nn.Module):
+ """
+ Track head that uses DPT head to process tokens and BaseTrackerPredictor for tracking.
+ The tracking is performed iteratively, refining predictions over multiple iterations.
+ """
+
+ def __init__(
+ self,
+ dim_in,
+ patch_size=14,
+ features=128,
+ iters=4,
+ predict_conf=True,
+ stride=2,
+ corr_levels=7,
+ corr_radius=4,
+ hidden_size=384,
+ ):
+ """
+ Initialize the TrackHead module.
+
+ Args:
+ dim_in (int): Input dimension of tokens from the backbone.
+ patch_size (int): Size of image patches used in the vision transformer.
+ features (int): Number of feature channels in the feature extractor output.
+ iters (int): Number of refinement iterations for tracking predictions.
+ predict_conf (bool): Whether to predict confidence scores for tracked points.
+ stride (int): Stride value for the tracker predictor.
+ corr_levels (int): Number of correlation pyramid levels
+ corr_radius (int): Radius for correlation computation, controlling the search area.
+ hidden_size (int): Size of hidden layers in the tracker network.
+ """
+ super().__init__()
+
+ self.patch_size = patch_size
+
+ # Feature extractor based on DPT architecture
+ # Processes tokens into feature maps for tracking
+ self.feature_extractor = DPTHead(
+ dim_in=dim_in,
+ patch_size=patch_size,
+ features=features,
+ feature_only=True, # Only output features, no activation
+ down_ratio=2, # Reduces spatial dimensions by factor of 2
+ pos_embed=False,
+ )
+
+ # Tracker module that predicts point trajectories
+ # Takes feature maps and predicts coordinates and visibility
+ self.tracker = BaseTrackerPredictor(
+ latent_dim=features, # Match the output_dim of feature extractor
+ predict_conf=predict_conf,
+ stride=stride,
+ corr_levels=corr_levels,
+ corr_radius=corr_radius,
+ hidden_size=hidden_size,
+ )
+
+ self.iters = iters
+
+ def forward(self, aggregated_tokens_list, images, patch_start_idx, query_points=None, iters=None):
+ """
+ Forward pass of the TrackHead.
+
+ Args:
+ aggregated_tokens_list (list): List of aggregated tokens from the backbone.
+ images (torch.Tensor): Input images of shape (B, S, C, H, W) where:
+ B = batch size, S = sequence length.
+ patch_start_idx (int): Starting index for patch tokens.
+ query_points (torch.Tensor, optional): Initial query points to track.
+ If None, points are initialized by the tracker.
+ iters (int, optional): Number of refinement iterations. If None, uses self.iters.
+
+ Returns:
+ tuple:
+ - coord_preds (torch.Tensor): Predicted coordinates for tracked points.
+ - vis_scores (torch.Tensor): Visibility scores for tracked points.
+ - conf_scores (torch.Tensor): Confidence scores for tracked points (if predict_conf=True).
+ """
+ B, S, _, H, W = images.shape
+
+ # Extract features from tokens
+ # feature_maps has shape (B, S, C, H//2, W//2) due to down_ratio=2
+ feature_maps = self.feature_extractor(aggregated_tokens_list, images, patch_start_idx)
+
+ # Use default iterations if not specified
+ if iters is None:
+ iters = self.iters
+
+ # Perform tracking using the extracted features
+ coord_preds, vis_scores, conf_scores = self.tracker(
+ query_points=query_points,
+ fmaps=feature_maps,
+ iters=iters,
+ )
+
+ return coord_preds, vis_scores, conf_scores
diff --git a/vggt/vggt/heads/track_modules/__init__.py b/vggt/vggt/heads/track_modules/__init__.py
new file mode 100644
index 0000000..0952fcc
--- /dev/null
+++ b/vggt/vggt/heads/track_modules/__init__.py
@@ -0,0 +1,5 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
diff --git a/vggt/vggt/heads/track_modules/base_track_predictor.py b/vggt/vggt/heads/track_modules/base_track_predictor.py
new file mode 100644
index 0000000..3ce8ec4
--- /dev/null
+++ b/vggt/vggt/heads/track_modules/base_track_predictor.py
@@ -0,0 +1,209 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch
+import torch.nn as nn
+from einops import rearrange, repeat
+
+
+from .blocks import EfficientUpdateFormer, CorrBlock
+from .utils import sample_features4d, get_2d_embedding, get_2d_sincos_pos_embed
+from .modules import Mlp
+
+
+class BaseTrackerPredictor(nn.Module):
+ def __init__(
+ self,
+ stride=1,
+ corr_levels=5,
+ corr_radius=4,
+ latent_dim=128,
+ hidden_size=384,
+ use_spaceatt=True,
+ depth=6,
+ max_scale=518,
+ predict_conf=True,
+ ):
+ super(BaseTrackerPredictor, self).__init__()
+ """
+ The base template to create a track predictor
+
+ Modified from https://github.com/facebookresearch/co-tracker/
+ and https://github.com/facebookresearch/vggsfm
+ """
+
+ self.stride = stride
+ self.latent_dim = latent_dim
+ self.corr_levels = corr_levels
+ self.corr_radius = corr_radius
+ self.hidden_size = hidden_size
+ self.max_scale = max_scale
+ self.predict_conf = predict_conf
+
+ self.flows_emb_dim = latent_dim // 2
+
+ self.corr_mlp = Mlp(
+ in_features=self.corr_levels * (self.corr_radius * 2 + 1) ** 2,
+ hidden_features=self.hidden_size,
+ out_features=self.latent_dim,
+ )
+
+ self.transformer_dim = self.latent_dim + self.latent_dim + self.latent_dim + 4
+
+ self.query_ref_token = nn.Parameter(torch.randn(1, 2, self.transformer_dim))
+
+ space_depth = depth if use_spaceatt else 0
+ time_depth = depth
+
+ self.updateformer = EfficientUpdateFormer(
+ space_depth=space_depth,
+ time_depth=time_depth,
+ input_dim=self.transformer_dim,
+ hidden_size=self.hidden_size,
+ output_dim=self.latent_dim + 2,
+ mlp_ratio=4.0,
+ add_space_attn=use_spaceatt,
+ )
+
+ self.fmap_norm = nn.LayerNorm(self.latent_dim)
+ self.ffeat_norm = nn.GroupNorm(1, self.latent_dim)
+
+ # A linear layer to update track feats at each iteration
+ self.ffeat_updater = nn.Sequential(nn.Linear(self.latent_dim, self.latent_dim), nn.GELU())
+
+ self.vis_predictor = nn.Sequential(nn.Linear(self.latent_dim, 1))
+
+ if predict_conf:
+ self.conf_predictor = nn.Sequential(nn.Linear(self.latent_dim, 1))
+
+ def forward(self, query_points, fmaps=None, iters=6, return_feat=False, down_ratio=1, apply_sigmoid=True):
+ """
+ query_points: B x N x 2, the number of batches, tracks, and xy
+ fmaps: B x S x C x HH x WW, the number of batches, frames, and feature dimension.
+ note HH and WW is the size of feature maps instead of original images
+ """
+ B, N, D = query_points.shape
+ B, S, C, HH, WW = fmaps.shape
+
+ assert D == 2, "Input points must be 2D coordinates"
+
+ # apply a layernorm to fmaps here
+ fmaps = self.fmap_norm(fmaps.permute(0, 1, 3, 4, 2))
+ fmaps = fmaps.permute(0, 1, 4, 2, 3)
+
+ # Scale the input query_points because we may downsample the images
+ # by down_ratio or self.stride
+ # e.g., if a 3x1024x1024 image is processed to a 128x256x256 feature map
+ # its query_points should be query_points/4
+ if down_ratio > 1:
+ query_points = query_points / float(down_ratio)
+
+ query_points = query_points / float(self.stride)
+
+ # Init with coords as the query points
+ # It means the search will start from the position of query points at the reference frames
+ coords = query_points.clone().reshape(B, 1, N, 2).repeat(1, S, 1, 1)
+
+ # Sample/extract the features of the query points in the query frame
+ query_track_feat = sample_features4d(fmaps[:, 0], coords[:, 0])
+
+ # init track feats by query feats
+ track_feats = query_track_feat.unsqueeze(1).repeat(1, S, 1, 1) # B, S, N, C
+ # back up the init coords
+ coords_backup = coords.clone()
+
+ fcorr_fn = CorrBlock(fmaps, num_levels=self.corr_levels, radius=self.corr_radius)
+
+ coord_preds = []
+
+ # Iterative Refinement
+ for _ in range(iters):
+ # Detach the gradients from the last iteration
+ # (in my experience, not very important for performance)
+ coords = coords.detach()
+
+ fcorrs = fcorr_fn.corr_sample(track_feats, coords)
+
+ corr_dim = fcorrs.shape[3]
+ fcorrs_ = fcorrs.permute(0, 2, 1, 3).reshape(B * N, S, corr_dim)
+ fcorrs_ = self.corr_mlp(fcorrs_)
+
+ # Movement of current coords relative to query points
+ flows = (coords - coords[:, 0:1]).permute(0, 2, 1, 3).reshape(B * N, S, 2)
+
+ flows_emb = get_2d_embedding(flows, self.flows_emb_dim, cat_coords=False)
+
+ # (In my trials, it is also okay to just add the flows_emb instead of concat)
+ flows_emb = torch.cat([flows_emb, flows / self.max_scale, flows / self.max_scale], dim=-1)
+
+ track_feats_ = track_feats.permute(0, 2, 1, 3).reshape(B * N, S, self.latent_dim)
+
+ # Concatenate them as the input for the transformers
+ transformer_input = torch.cat([flows_emb, fcorrs_, track_feats_], dim=2)
+
+ # 2D positional embed
+ # TODO: this can be much simplified
+ pos_embed = get_2d_sincos_pos_embed(self.transformer_dim, grid_size=(HH, WW)).to(query_points.device)
+ sampled_pos_emb = sample_features4d(pos_embed.expand(B, -1, -1, -1), coords[:, 0])
+
+ sampled_pos_emb = rearrange(sampled_pos_emb, "b n c -> (b n) c").unsqueeze(1)
+
+ x = transformer_input + sampled_pos_emb
+
+ # Add the query ref token to the track feats
+ query_ref_token = torch.cat(
+ [self.query_ref_token[:, 0:1], self.query_ref_token[:, 1:2].expand(-1, S - 1, -1)], dim=1
+ )
+ x = x + query_ref_token.to(x.device).to(x.dtype)
+
+ # B, N, S, C
+ x = rearrange(x, "(b n) s d -> b n s d", b=B)
+
+ # Compute the delta coordinates and delta track features
+ delta, _ = self.updateformer(x)
+
+ # BN, S, C
+ delta = rearrange(delta, " b n s d -> (b n) s d", b=B)
+ delta_coords_ = delta[:, :, :2]
+ delta_feats_ = delta[:, :, 2:]
+
+ track_feats_ = track_feats_.reshape(B * N * S, self.latent_dim)
+ delta_feats_ = delta_feats_.reshape(B * N * S, self.latent_dim)
+
+ # Update the track features
+ track_feats_ = self.ffeat_updater(self.ffeat_norm(delta_feats_)) + track_feats_
+
+ track_feats = track_feats_.reshape(B, N, S, self.latent_dim).permute(0, 2, 1, 3) # BxSxNxC
+
+ # B x S x N x 2
+ coords = coords + delta_coords_.reshape(B, N, S, 2).permute(0, 2, 1, 3)
+
+ # Force coord0 as query
+ # because we assume the query points should not be changed
+ coords[:, 0] = coords_backup[:, 0]
+
+ # The predicted tracks are in the original image scale
+ if down_ratio > 1:
+ coord_preds.append(coords * self.stride * down_ratio)
+ else:
+ coord_preds.append(coords * self.stride)
+
+ # B, S, N
+ vis_e = self.vis_predictor(track_feats.reshape(B * S * N, self.latent_dim)).reshape(B, S, N)
+ if apply_sigmoid:
+ vis_e = torch.sigmoid(vis_e)
+
+ if self.predict_conf:
+ conf_e = self.conf_predictor(track_feats.reshape(B * S * N, self.latent_dim)).reshape(B, S, N)
+ if apply_sigmoid:
+ conf_e = torch.sigmoid(conf_e)
+ else:
+ conf_e = None
+
+ if return_feat:
+ return coord_preds, vis_e, track_feats, query_track_feat, conf_e
+ else:
+ return coord_preds, vis_e, conf_e
diff --git a/vggt/vggt/heads/track_modules/blocks.py b/vggt/vggt/heads/track_modules/blocks.py
new file mode 100644
index 0000000..8e7763f
--- /dev/null
+++ b/vggt/vggt/heads/track_modules/blocks.py
@@ -0,0 +1,246 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+
+# Modified from https://github.com/facebookresearch/co-tracker/
+
+import math
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+from .utils import bilinear_sampler
+from .modules import Mlp, AttnBlock, CrossAttnBlock, ResidualBlock
+
+
+class EfficientUpdateFormer(nn.Module):
+ """
+ Transformer model that updates track estimates.
+ """
+
+ def __init__(
+ self,
+ space_depth=6,
+ time_depth=6,
+ input_dim=320,
+ hidden_size=384,
+ num_heads=8,
+ output_dim=130,
+ mlp_ratio=4.0,
+ add_space_attn=True,
+ num_virtual_tracks=64,
+ ):
+ super().__init__()
+
+ self.out_channels = 2
+ self.num_heads = num_heads
+ self.hidden_size = hidden_size
+ self.add_space_attn = add_space_attn
+
+ # Add input LayerNorm before linear projection
+ self.input_norm = nn.LayerNorm(input_dim)
+ self.input_transform = torch.nn.Linear(input_dim, hidden_size, bias=True)
+
+ # Add output LayerNorm before final projection
+ self.output_norm = nn.LayerNorm(hidden_size)
+ self.flow_head = torch.nn.Linear(hidden_size, output_dim, bias=True)
+ self.num_virtual_tracks = num_virtual_tracks
+
+ if self.add_space_attn:
+ self.virual_tracks = nn.Parameter(torch.randn(1, num_virtual_tracks, 1, hidden_size))
+ else:
+ self.virual_tracks = None
+
+ self.time_blocks = nn.ModuleList(
+ [
+ AttnBlock(
+ hidden_size,
+ num_heads,
+ mlp_ratio=mlp_ratio,
+ attn_class=nn.MultiheadAttention,
+ )
+ for _ in range(time_depth)
+ ]
+ )
+
+ if add_space_attn:
+ self.space_virtual_blocks = nn.ModuleList(
+ [
+ AttnBlock(
+ hidden_size,
+ num_heads,
+ mlp_ratio=mlp_ratio,
+ attn_class=nn.MultiheadAttention,
+ )
+ for _ in range(space_depth)
+ ]
+ )
+ self.space_point2virtual_blocks = nn.ModuleList(
+ [CrossAttnBlock(hidden_size, hidden_size, num_heads, mlp_ratio=mlp_ratio) for _ in range(space_depth)]
+ )
+ self.space_virtual2point_blocks = nn.ModuleList(
+ [CrossAttnBlock(hidden_size, hidden_size, num_heads, mlp_ratio=mlp_ratio) for _ in range(space_depth)]
+ )
+ assert len(self.time_blocks) >= len(self.space_virtual2point_blocks)
+ self.initialize_weights()
+
+ def initialize_weights(self):
+ def _basic_init(module):
+ if isinstance(module, nn.Linear):
+ torch.nn.init.xavier_uniform_(module.weight)
+ if module.bias is not None:
+ nn.init.constant_(module.bias, 0)
+ torch.nn.init.trunc_normal_(self.flow_head.weight, std=0.001)
+
+ self.apply(_basic_init)
+
+ def forward(self, input_tensor, mask=None):
+ # Apply input LayerNorm
+ input_tensor = self.input_norm(input_tensor)
+ tokens = self.input_transform(input_tensor)
+
+ init_tokens = tokens
+
+ B, _, T, _ = tokens.shape
+
+ if self.add_space_attn:
+ virtual_tokens = self.virual_tracks.repeat(B, 1, T, 1)
+ tokens = torch.cat([tokens, virtual_tokens], dim=1)
+
+ _, N, _, _ = tokens.shape
+
+ j = 0
+ for i in range(len(self.time_blocks)):
+ time_tokens = tokens.contiguous().view(B * N, T, -1) # B N T C -> (B N) T C
+
+ time_tokens = self.time_blocks[i](time_tokens)
+
+ tokens = time_tokens.view(B, N, T, -1) # (B N) T C -> B N T C
+ if self.add_space_attn and (i % (len(self.time_blocks) // len(self.space_virtual_blocks)) == 0):
+ space_tokens = tokens.permute(0, 2, 1, 3).contiguous().view(B * T, N, -1) # B N T C -> (B T) N C
+ point_tokens = space_tokens[:, : N - self.num_virtual_tracks]
+ virtual_tokens = space_tokens[:, N - self.num_virtual_tracks :]
+
+ virtual_tokens = self.space_virtual2point_blocks[j](virtual_tokens, point_tokens, mask=mask)
+ virtual_tokens = self.space_virtual_blocks[j](virtual_tokens)
+ point_tokens = self.space_point2virtual_blocks[j](point_tokens, virtual_tokens, mask=mask)
+
+ space_tokens = torch.cat([point_tokens, virtual_tokens], dim=1)
+ tokens = space_tokens.view(B, T, N, -1).permute(0, 2, 1, 3) # (B T) N C -> B N T C
+ j += 1
+
+ if self.add_space_attn:
+ tokens = tokens[:, : N - self.num_virtual_tracks]
+
+ tokens = tokens + init_tokens
+
+ # Apply output LayerNorm before final projection
+ tokens = self.output_norm(tokens)
+ flow = self.flow_head(tokens)
+
+ return flow, None
+
+
+class CorrBlock:
+ def __init__(self, fmaps, num_levels=4, radius=4, multiple_track_feats=False, padding_mode="zeros"):
+ """
+ Build a pyramid of feature maps from the input.
+
+ fmaps: Tensor (B, S, C, H, W)
+ num_levels: number of pyramid levels (each downsampled by factor 2)
+ radius: search radius for sampling correlation
+ multiple_track_feats: if True, split the target features per pyramid level
+ padding_mode: passed to grid_sample / bilinear_sampler
+ """
+ B, S, C, H, W = fmaps.shape
+ self.S, self.C, self.H, self.W = S, C, H, W
+ self.num_levels = num_levels
+ self.radius = radius
+ self.padding_mode = padding_mode
+ self.multiple_track_feats = multiple_track_feats
+
+ # Build pyramid: each level is half the spatial resolution of the previous
+ self.fmaps_pyramid = [fmaps] # level 0 is full resolution
+ current_fmaps = fmaps
+ for i in range(num_levels - 1):
+ B, S, C, H, W = current_fmaps.shape
+ # Merge batch & sequence dimensions
+ current_fmaps = current_fmaps.reshape(B * S, C, H, W)
+ # Avg pool down by factor 2
+ current_fmaps = F.avg_pool2d(current_fmaps, kernel_size=2, stride=2)
+ _, _, H_new, W_new = current_fmaps.shape
+ current_fmaps = current_fmaps.reshape(B, S, C, H_new, W_new)
+ self.fmaps_pyramid.append(current_fmaps)
+
+ # Precompute a delta grid (of shape (2r+1, 2r+1, 2)) for sampling.
+ # This grid is added to the (scaled) coordinate centroids.
+ r = self.radius
+ dx = torch.linspace(-r, r, 2 * r + 1, device=fmaps.device, dtype=fmaps.dtype)
+ dy = torch.linspace(-r, r, 2 * r + 1, device=fmaps.device, dtype=fmaps.dtype)
+ # delta: for every (dy,dx) displacement (i.e. Δx, Δy)
+ self.delta = torch.stack(torch.meshgrid(dy, dx, indexing="ij"), dim=-1) # shape: (2r+1, 2r+1, 2)
+
+ def corr_sample(self, targets, coords):
+ """
+ Instead of storing the entire correlation pyramid, we compute each level's correlation
+ volume, sample it immediately, then discard it. This saves GPU memory.
+
+ Args:
+ targets: Tensor (B, S, N, C) — features for the current targets.
+ coords: Tensor (B, S, N, 2) — coordinates at full resolution.
+
+ Returns:
+ Tensor (B, S, N, L) where L = num_levels * (2*radius+1)**2 (concatenated sampled correlations)
+ """
+ B, S, N, C = targets.shape
+
+ # If you have multiple track features, split them per level.
+ if self.multiple_track_feats:
+ targets_split = torch.split(targets, C // self.num_levels, dim=-1)
+
+ out_pyramid = []
+ for i, fmaps in enumerate(self.fmaps_pyramid):
+ # Get current spatial resolution H, W for this pyramid level.
+ B, S, C, H, W = fmaps.shape
+ # Reshape feature maps for correlation computation:
+ # fmap2s: (B, S, C, H*W)
+ fmap2s = fmaps.view(B, S, C, H * W)
+ # Choose appropriate target features.
+ fmap1 = targets_split[i] if self.multiple_track_feats else targets # shape: (B, S, N, C)
+
+ # Compute correlation directly
+ corrs = compute_corr_level(fmap1, fmap2s, C)
+ corrs = corrs.view(B, S, N, H, W)
+
+ # Prepare sampling grid:
+ # Scale down the coordinates for the current level.
+ centroid_lvl = coords.reshape(B * S * N, 1, 1, 2) / (2**i)
+ # Make sure our precomputed delta grid is on the same device/dtype.
+ delta_lvl = self.delta.to(coords.device).to(coords.dtype)
+ # Now the grid for grid_sample is:
+ # coords_lvl = centroid_lvl + delta_lvl (broadcasted over grid)
+ coords_lvl = centroid_lvl + delta_lvl.view(1, 2 * self.radius + 1, 2 * self.radius + 1, 2)
+
+ # Sample from the correlation volume using bilinear interpolation.
+ # We reshape corrs to (B * S * N, 1, H, W) so grid_sample acts over each target.
+ corrs_sampled = bilinear_sampler(
+ corrs.reshape(B * S * N, 1, H, W), coords_lvl, padding_mode=self.padding_mode
+ )
+ # The sampled output is (B * S * N, 1, 2r+1, 2r+1). Flatten the last two dims.
+ corrs_sampled = corrs_sampled.view(B, S, N, -1) # Now shape: (B, S, N, (2r+1)^2)
+ out_pyramid.append(corrs_sampled)
+
+ # Concatenate all levels along the last dimension.
+ out = torch.cat(out_pyramid, dim=-1).contiguous()
+ return out
+
+
+def compute_corr_level(fmap1, fmap2s, C):
+ # fmap1: (B, S, N, C)
+ # fmap2s: (B, S, C, H*W)
+ corrs = torch.matmul(fmap1, fmap2s) # (B, S, N, H*W)
+ corrs = corrs.view(fmap1.shape[0], fmap1.shape[1], fmap1.shape[2], -1) # (B, S, N, H*W)
+ return corrs / math.sqrt(C)
diff --git a/vggt/vggt/heads/track_modules/modules.py b/vggt/vggt/heads/track_modules/modules.py
new file mode 100644
index 0000000..4b090dd
--- /dev/null
+++ b/vggt/vggt/heads/track_modules/modules.py
@@ -0,0 +1,218 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from functools import partial
+from typing import Callable
+import collections
+from torch import Tensor
+from itertools import repeat
+
+
+# From PyTorch internals
+def _ntuple(n):
+ def parse(x):
+ if isinstance(x, collections.abc.Iterable) and not isinstance(x, str):
+ return tuple(x)
+ return tuple(repeat(x, n))
+
+ return parse
+
+
+def exists(val):
+ return val is not None
+
+
+def default(val, d):
+ return val if exists(val) else d
+
+
+to_2tuple = _ntuple(2)
+
+
+class ResidualBlock(nn.Module):
+ """
+ ResidualBlock: construct a block of two conv layers with residual connections
+ """
+
+ def __init__(self, in_planes, planes, norm_fn="group", stride=1, kernel_size=3):
+ super(ResidualBlock, self).__init__()
+
+ self.conv1 = nn.Conv2d(
+ in_planes,
+ planes,
+ kernel_size=kernel_size,
+ padding=1,
+ stride=stride,
+ padding_mode="zeros",
+ )
+ self.conv2 = nn.Conv2d(
+ planes,
+ planes,
+ kernel_size=kernel_size,
+ padding=1,
+ padding_mode="zeros",
+ )
+ self.relu = nn.ReLU(inplace=True)
+
+ num_groups = planes // 8
+
+ if norm_fn == "group":
+ self.norm1 = nn.GroupNorm(num_groups=num_groups, num_channels=planes)
+ self.norm2 = nn.GroupNorm(num_groups=num_groups, num_channels=planes)
+ if not stride == 1:
+ self.norm3 = nn.GroupNorm(num_groups=num_groups, num_channels=planes)
+
+ elif norm_fn == "batch":
+ self.norm1 = nn.BatchNorm2d(planes)
+ self.norm2 = nn.BatchNorm2d(planes)
+ if not stride == 1:
+ self.norm3 = nn.BatchNorm2d(planes)
+
+ elif norm_fn == "instance":
+ self.norm1 = nn.InstanceNorm2d(planes)
+ self.norm2 = nn.InstanceNorm2d(planes)
+ if not stride == 1:
+ self.norm3 = nn.InstanceNorm2d(planes)
+
+ elif norm_fn == "none":
+ self.norm1 = nn.Sequential()
+ self.norm2 = nn.Sequential()
+ if not stride == 1:
+ self.norm3 = nn.Sequential()
+ else:
+ raise NotImplementedError
+
+ if stride == 1:
+ self.downsample = None
+ else:
+ self.downsample = nn.Sequential(
+ nn.Conv2d(in_planes, planes, kernel_size=1, stride=stride),
+ self.norm3,
+ )
+
+ def forward(self, x):
+ y = x
+ y = self.relu(self.norm1(self.conv1(y)))
+ y = self.relu(self.norm2(self.conv2(y)))
+
+ if self.downsample is not None:
+ x = self.downsample(x)
+
+ return self.relu(x + y)
+
+
+class Mlp(nn.Module):
+ """MLP as used in Vision Transformer, MLP-Mixer and related networks"""
+
+ def __init__(
+ self,
+ in_features,
+ hidden_features=None,
+ out_features=None,
+ act_layer=nn.GELU,
+ norm_layer=None,
+ bias=True,
+ drop=0.0,
+ use_conv=False,
+ ):
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ bias = to_2tuple(bias)
+ drop_probs = to_2tuple(drop)
+ linear_layer = partial(nn.Conv2d, kernel_size=1) if use_conv else nn.Linear
+
+ self.fc1 = linear_layer(in_features, hidden_features, bias=bias[0])
+ self.act = act_layer()
+ self.drop1 = nn.Dropout(drop_probs[0])
+ self.fc2 = linear_layer(hidden_features, out_features, bias=bias[1])
+ self.drop2 = nn.Dropout(drop_probs[1])
+
+ def forward(self, x):
+ x = self.fc1(x)
+ x = self.act(x)
+ x = self.drop1(x)
+ x = self.fc2(x)
+ x = self.drop2(x)
+ return x
+
+
+class AttnBlock(nn.Module):
+ def __init__(
+ self,
+ hidden_size,
+ num_heads,
+ attn_class: Callable[..., nn.Module] = nn.MultiheadAttention,
+ mlp_ratio=4.0,
+ **block_kwargs
+ ):
+ """
+ Self attention block
+ """
+ super().__init__()
+
+ self.norm1 = nn.LayerNorm(hidden_size)
+ self.norm2 = nn.LayerNorm(hidden_size)
+
+ self.attn = attn_class(embed_dim=hidden_size, num_heads=num_heads, batch_first=True, **block_kwargs)
+
+ mlp_hidden_dim = int(hidden_size * mlp_ratio)
+
+ self.mlp = Mlp(in_features=hidden_size, hidden_features=mlp_hidden_dim, drop=0)
+
+ def forward(self, x, mask=None):
+ # Prepare the mask for PyTorch's attention (it expects a different format)
+ # attn_mask = mask if mask is not None else None
+ # Normalize before attention
+ x = self.norm1(x)
+
+ # PyTorch's MultiheadAttention returns attn_output, attn_output_weights
+ # attn_output, _ = self.attn(x, x, x, attn_mask=attn_mask)
+
+ attn_output, _ = self.attn(x, x, x)
+
+ # Add & Norm
+ x = x + attn_output
+ x = x + self.mlp(self.norm2(x))
+ return x
+
+
+class CrossAttnBlock(nn.Module):
+ def __init__(self, hidden_size, context_dim, num_heads=1, mlp_ratio=4.0, **block_kwargs):
+ """
+ Cross attention block
+ """
+ super().__init__()
+
+ self.norm1 = nn.LayerNorm(hidden_size)
+ self.norm_context = nn.LayerNorm(hidden_size)
+ self.norm2 = nn.LayerNorm(hidden_size)
+
+ self.cross_attn = nn.MultiheadAttention(
+ embed_dim=hidden_size, num_heads=num_heads, batch_first=True, **block_kwargs
+ )
+
+ mlp_hidden_dim = int(hidden_size * mlp_ratio)
+
+ self.mlp = Mlp(in_features=hidden_size, hidden_features=mlp_hidden_dim, drop=0)
+
+ def forward(self, x, context, mask=None):
+ # Normalize inputs
+ x = self.norm1(x)
+ context = self.norm_context(context)
+
+ # Apply cross attention
+ # Note: nn.MultiheadAttention returns attn_output, attn_output_weights
+ attn_output, _ = self.cross_attn(x, context, context, attn_mask=mask)
+
+ # Add & Norm
+ x = x + attn_output
+ x = x + self.mlp(self.norm2(x))
+ return x
diff --git a/vggt/vggt/heads/track_modules/utils.py b/vggt/vggt/heads/track_modules/utils.py
new file mode 100644
index 0000000..51d01d3
--- /dev/null
+++ b/vggt/vggt/heads/track_modules/utils.py
@@ -0,0 +1,226 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+# Modified from https://github.com/facebookresearch/vggsfm
+# and https://github.com/facebookresearch/co-tracker/tree/main
+
+
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+
+from typing import Optional, Tuple, Union
+
+
+def get_2d_sincos_pos_embed(embed_dim: int, grid_size: Union[int, Tuple[int, int]], return_grid=False) -> torch.Tensor:
+ """
+ This function initializes a grid and generates a 2D positional embedding using sine and cosine functions.
+ It is a wrapper of get_2d_sincos_pos_embed_from_grid.
+ Args:
+ - embed_dim: The embedding dimension.
+ - grid_size: The grid size.
+ Returns:
+ - pos_embed: The generated 2D positional embedding.
+ """
+ if isinstance(grid_size, tuple):
+ grid_size_h, grid_size_w = grid_size
+ else:
+ grid_size_h = grid_size_w = grid_size
+ grid_h = torch.arange(grid_size_h, dtype=torch.float)
+ grid_w = torch.arange(grid_size_w, dtype=torch.float)
+ grid = torch.meshgrid(grid_w, grid_h, indexing="xy")
+ grid = torch.stack(grid, dim=0)
+ grid = grid.reshape([2, 1, grid_size_h, grid_size_w])
+ pos_embed = get_2d_sincos_pos_embed_from_grid(embed_dim, grid)
+ if return_grid:
+ return (
+ pos_embed.reshape(1, grid_size_h, grid_size_w, -1).permute(0, 3, 1, 2),
+ grid,
+ )
+ return pos_embed.reshape(1, grid_size_h, grid_size_w, -1).permute(0, 3, 1, 2)
+
+
+def get_2d_sincos_pos_embed_from_grid(embed_dim: int, grid: torch.Tensor) -> torch.Tensor:
+ """
+ This function generates a 2D positional embedding from a given grid using sine and cosine functions.
+
+ Args:
+ - embed_dim: The embedding dimension.
+ - grid: The grid to generate the embedding from.
+
+ Returns:
+ - emb: The generated 2D positional embedding.
+ """
+ assert embed_dim % 2 == 0
+
+ # use half of dimensions to encode grid_h
+ emb_h = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[0]) # (H*W, D/2)
+ emb_w = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[1]) # (H*W, D/2)
+
+ emb = torch.cat([emb_h, emb_w], dim=2) # (H*W, D)
+ return emb
+
+
+def get_1d_sincos_pos_embed_from_grid(embed_dim: int, pos: torch.Tensor) -> torch.Tensor:
+ """
+ This function generates a 1D positional embedding from a given grid using sine and cosine functions.
+
+ Args:
+ - embed_dim: The embedding dimension.
+ - pos: The position to generate the embedding from.
+
+ Returns:
+ - emb: The generated 1D positional embedding.
+ """
+ assert embed_dim % 2 == 0
+ omega = torch.arange(embed_dim // 2, dtype=torch.double)
+ omega /= embed_dim / 2.0
+ omega = 1.0 / 10000**omega # (D/2,)
+
+ pos = pos.reshape(-1) # (M,)
+ out = torch.einsum("m,d->md", pos, omega) # (M, D/2), outer product
+
+ emb_sin = torch.sin(out) # (M, D/2)
+ emb_cos = torch.cos(out) # (M, D/2)
+
+ emb = torch.cat([emb_sin, emb_cos], dim=1) # (M, D)
+ return emb[None].float()
+
+
+def get_2d_embedding(xy: torch.Tensor, C: int, cat_coords: bool = True) -> torch.Tensor:
+ """
+ This function generates a 2D positional embedding from given coordinates using sine and cosine functions.
+
+ Args:
+ - xy: The coordinates to generate the embedding from.
+ - C: The size of the embedding.
+ - cat_coords: A flag to indicate whether to concatenate the original coordinates to the embedding.
+
+ Returns:
+ - pe: The generated 2D positional embedding.
+ """
+ B, N, D = xy.shape
+ assert D == 2
+
+ x = xy[:, :, 0:1]
+ y = xy[:, :, 1:2]
+ div_term = (torch.arange(0, C, 2, device=xy.device, dtype=torch.float32) * (1000.0 / C)).reshape(1, 1, int(C / 2))
+
+ pe_x = torch.zeros(B, N, C, device=xy.device, dtype=torch.float32)
+ pe_y = torch.zeros(B, N, C, device=xy.device, dtype=torch.float32)
+
+ pe_x[:, :, 0::2] = torch.sin(x * div_term)
+ pe_x[:, :, 1::2] = torch.cos(x * div_term)
+
+ pe_y[:, :, 0::2] = torch.sin(y * div_term)
+ pe_y[:, :, 1::2] = torch.cos(y * div_term)
+
+ pe = torch.cat([pe_x, pe_y], dim=2) # (B, N, C*3)
+ if cat_coords:
+ pe = torch.cat([xy, pe], dim=2) # (B, N, C*3+3)
+ return pe
+
+
+def bilinear_sampler(input, coords, align_corners=True, padding_mode="border"):
+ r"""Sample a tensor using bilinear interpolation
+
+ `bilinear_sampler(input, coords)` samples a tensor :attr:`input` at
+ coordinates :attr:`coords` using bilinear interpolation. It is the same
+ as `torch.nn.functional.grid_sample()` but with a different coordinate
+ convention.
+
+ The input tensor is assumed to be of shape :math:`(B, C, H, W)`, where
+ :math:`B` is the batch size, :math:`C` is the number of channels,
+ :math:`H` is the height of the image, and :math:`W` is the width of the
+ image. The tensor :attr:`coords` of shape :math:`(B, H_o, W_o, 2)` is
+ interpreted as an array of 2D point coordinates :math:`(x_i,y_i)`.
+
+ Alternatively, the input tensor can be of size :math:`(B, C, T, H, W)`,
+ in which case sample points are triplets :math:`(t_i,x_i,y_i)`. Note
+ that in this case the order of the components is slightly different
+ from `grid_sample()`, which would expect :math:`(x_i,y_i,t_i)`.
+
+ If `align_corners` is `True`, the coordinate :math:`x` is assumed to be
+ in the range :math:`[0,W-1]`, with 0 corresponding to the center of the
+ left-most image pixel :math:`W-1` to the center of the right-most
+ pixel.
+
+ If `align_corners` is `False`, the coordinate :math:`x` is assumed to
+ be in the range :math:`[0,W]`, with 0 corresponding to the left edge of
+ the left-most pixel :math:`W` to the right edge of the right-most
+ pixel.
+
+ Similar conventions apply to the :math:`y` for the range
+ :math:`[0,H-1]` and :math:`[0,H]` and to :math:`t` for the range
+ :math:`[0,T-1]` and :math:`[0,T]`.
+
+ Args:
+ input (Tensor): batch of input images.
+ coords (Tensor): batch of coordinates.
+ align_corners (bool, optional): Coordinate convention. Defaults to `True`.
+ padding_mode (str, optional): Padding mode. Defaults to `"border"`.
+
+ Returns:
+ Tensor: sampled points.
+ """
+ coords = coords.detach().clone()
+ ############################################################
+ # IMPORTANT:
+ coords = coords.to(input.device).to(input.dtype)
+ ############################################################
+
+ sizes = input.shape[2:]
+
+ assert len(sizes) in [2, 3]
+
+ if len(sizes) == 3:
+ # t x y -> x y t to match dimensions T H W in grid_sample
+ coords = coords[..., [1, 2, 0]]
+
+ if align_corners:
+ scale = torch.tensor(
+ [2 / max(size - 1, 1) for size in reversed(sizes)], device=coords.device, dtype=coords.dtype
+ )
+ else:
+ scale = torch.tensor([2 / size for size in reversed(sizes)], device=coords.device, dtype=coords.dtype)
+
+ coords.mul_(scale) # coords = coords * scale
+ coords.sub_(1) # coords = coords - 1
+
+ return F.grid_sample(input, coords, align_corners=align_corners, padding_mode=padding_mode)
+
+
+def sample_features4d(input, coords):
+ r"""Sample spatial features
+
+ `sample_features4d(input, coords)` samples the spatial features
+ :attr:`input` represented by a 4D tensor :math:`(B, C, H, W)`.
+
+ The field is sampled at coordinates :attr:`coords` using bilinear
+ interpolation. :attr:`coords` is assumed to be of shape :math:`(B, R,
+ 2)`, where each sample has the format :math:`(x_i, y_i)`. This uses the
+ same convention as :func:`bilinear_sampler` with `align_corners=True`.
+
+ The output tensor has one feature per point, and has shape :math:`(B,
+ R, C)`.
+
+ Args:
+ input (Tensor): spatial features.
+ coords (Tensor): points.
+
+ Returns:
+ Tensor: sampled features.
+ """
+
+ B, _, _, _ = input.shape
+
+ # B R 2 -> B R 1 2
+ coords = coords.unsqueeze(2)
+
+ # B C R 1
+ feats = bilinear_sampler(input, coords)
+
+ return feats.permute(0, 2, 1, 3).view(B, -1, feats.shape[1] * feats.shape[3]) # B C R 1 -> B R C
diff --git a/vggt/vggt/heads/utils.py b/vggt/vggt/heads/utils.py
new file mode 100644
index 0000000..d7af1f6
--- /dev/null
+++ b/vggt/vggt/heads/utils.py
@@ -0,0 +1,108 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch
+import torch.nn as nn
+
+
+def position_grid_to_embed(pos_grid: torch.Tensor, embed_dim: int, omega_0: float = 100) -> torch.Tensor:
+ """
+ Convert 2D position grid (HxWx2) to sinusoidal embeddings (HxWxC)
+
+ Args:
+ pos_grid: Tensor of shape (H, W, 2) containing 2D coordinates
+ embed_dim: Output channel dimension for embeddings
+
+ Returns:
+ Tensor of shape (H, W, embed_dim) with positional embeddings
+ """
+ H, W, grid_dim = pos_grid.shape
+ assert grid_dim == 2
+ pos_flat = pos_grid.reshape(-1, grid_dim) # Flatten to (H*W, 2)
+
+ # Process x and y coordinates separately
+ emb_x = make_sincos_pos_embed(embed_dim // 2, pos_flat[:, 0], omega_0=omega_0) # [1, H*W, D/2]
+ emb_y = make_sincos_pos_embed(embed_dim // 2, pos_flat[:, 1], omega_0=omega_0) # [1, H*W, D/2]
+
+ # Combine and reshape
+ emb = torch.cat([emb_x, emb_y], dim=-1) # [1, H*W, D]
+
+ return emb.view(H, W, embed_dim) # [H, W, D]
+
+
+def make_sincos_pos_embed(embed_dim: int, pos: torch.Tensor, omega_0: float = 100) -> torch.Tensor:
+ """
+ This function generates a 1D positional embedding from a given grid using sine and cosine functions.
+
+ Args:
+ - embed_dim: The embedding dimension.
+ - pos: The position to generate the embedding from.
+
+ Returns:
+ - emb: The generated 1D positional embedding.
+ """
+ assert embed_dim % 2 == 0
+ omega = torch.arange(embed_dim // 2, dtype=torch.double, device=pos.device)
+ omega /= embed_dim / 2.0
+ omega = 1.0 / omega_0**omega # (D/2,)
+
+ pos = pos.reshape(-1) # (M,)
+ out = torch.einsum("m,d->md", pos, omega) # (M, D/2), outer product
+
+ emb_sin = torch.sin(out) # (M, D/2)
+ emb_cos = torch.cos(out) # (M, D/2)
+
+ emb = torch.cat([emb_sin, emb_cos], dim=1) # (M, D)
+ return emb.float()
+
+
+# Inspired by https://github.com/microsoft/moge
+
+
+def create_uv_grid(
+ width: int, height: int, aspect_ratio: float = None, dtype: torch.dtype = None, device: torch.device = None
+) -> torch.Tensor:
+ """
+ Create a normalized UV grid of shape (width, height, 2).
+
+ The grid spans horizontally and vertically according to an aspect ratio,
+ ensuring the top-left corner is at (-x_span, -y_span) and the bottom-right
+ corner is at (x_span, y_span), normalized by the diagonal of the plane.
+
+ Args:
+ width (int): Number of points horizontally.
+ height (int): Number of points vertically.
+ aspect_ratio (float, optional): Width-to-height ratio. Defaults to width/height.
+ dtype (torch.dtype, optional): Data type of the resulting tensor.
+ device (torch.device, optional): Device on which the tensor is created.
+
+ Returns:
+ torch.Tensor: A (width, height, 2) tensor of UV coordinates.
+ """
+ # Derive aspect ratio if not explicitly provided
+ if aspect_ratio is None:
+ aspect_ratio = float(width) / float(height)
+
+ # Compute normalized spans for X and Y
+ diag_factor = (aspect_ratio**2 + 1.0) ** 0.5
+ span_x = aspect_ratio / diag_factor
+ span_y = 1.0 / diag_factor
+
+ # Establish the linspace boundaries
+ left_x = -span_x * (width - 1) / width
+ right_x = span_x * (width - 1) / width
+ top_y = -span_y * (height - 1) / height
+ bottom_y = span_y * (height - 1) / height
+
+ # Generate 1D coordinates
+ x_coords = torch.linspace(left_x, right_x, steps=width, dtype=dtype, device=device)
+ y_coords = torch.linspace(top_y, bottom_y, steps=height, dtype=dtype, device=device)
+
+ # Create 2D meshgrid (width x height) and stack into UV
+ uu, vv = torch.meshgrid(x_coords, y_coords, indexing="xy")
+ uv_grid = torch.stack((uu, vv), dim=-1)
+
+ return uv_grid
diff --git a/vggt/vggt/layers/__init__.py b/vggt/vggt/layers/__init__.py
new file mode 100644
index 0000000..8120f4b
--- /dev/null
+++ b/vggt/vggt/layers/__init__.py
@@ -0,0 +1,11 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+from .mlp import Mlp
+from .patch_embed import PatchEmbed
+from .swiglu_ffn import SwiGLUFFN, SwiGLUFFNFused
+from .block import NestedTensorBlock
+from .attention import MemEffAttention
diff --git a/vggt/vggt/layers/attention.py b/vggt/vggt/layers/attention.py
new file mode 100644
index 0000000..ab3089c
--- /dev/null
+++ b/vggt/vggt/layers/attention.py
@@ -0,0 +1,98 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/models/vision_transformer.py
+
+import logging
+import os
+import warnings
+
+from torch import Tensor
+from torch import nn
+import torch.nn.functional as F
+
+XFORMERS_AVAILABLE = False
+
+
+class Attention(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ num_heads: int = 8,
+ qkv_bias: bool = True,
+ proj_bias: bool = True,
+ attn_drop: float = 0.0,
+ proj_drop: float = 0.0,
+ norm_layer: nn.Module = nn.LayerNorm,
+ qk_norm: bool = False,
+ fused_attn: bool = True, # use F.scaled_dot_product_attention or not
+ rope=None,
+ ) -> None:
+ super().__init__()
+ assert dim % num_heads == 0, "dim should be divisible by num_heads"
+ self.num_heads = num_heads
+ self.head_dim = dim // num_heads
+ self.scale = self.head_dim**-0.5
+ self.fused_attn = fused_attn
+
+ self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias)
+ self.q_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity()
+ self.k_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity()
+ self.attn_drop = nn.Dropout(attn_drop)
+ self.proj = nn.Linear(dim, dim, bias=proj_bias)
+ self.proj_drop = nn.Dropout(proj_drop)
+ self.rope = rope
+
+ def forward(self, x: Tensor, pos=None) -> Tensor:
+ B, N, C = x.shape
+ qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4)
+ q, k, v = qkv.unbind(0)
+ q, k = self.q_norm(q), self.k_norm(k)
+
+ if self.rope is not None:
+ q = self.rope(q, pos)
+ k = self.rope(k, pos)
+
+ if self.fused_attn:
+ x = F.scaled_dot_product_attention(
+ q,
+ k,
+ v,
+ dropout_p=self.attn_drop.p if self.training else 0.0,
+ )
+ else:
+ q = q * self.scale
+ attn = q @ k.transpose(-2, -1)
+ attn = attn.softmax(dim=-1)
+ attn = self.attn_drop(attn)
+ x = attn @ v
+
+ x = x.transpose(1, 2).reshape(B, N, C)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+ return x
+
+
+class MemEffAttention(Attention):
+ def forward(self, x: Tensor, attn_bias=None, pos=None) -> Tensor:
+ assert pos is None
+ if not XFORMERS_AVAILABLE:
+ if attn_bias is not None:
+ raise AssertionError("xFormers is required for using nested tensors")
+ return super().forward(x)
+
+ B, N, C = x.shape
+ qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, C // self.num_heads)
+
+ q, k, v = unbind(qkv, 2)
+
+ x = memory_efficient_attention(q, k, v, attn_bias=attn_bias)
+ x = x.reshape([B, N, C])
+
+ x = self.proj(x)
+ x = self.proj_drop(x)
+ return x
diff --git a/vggt/vggt/layers/block.py b/vggt/vggt/layers/block.py
new file mode 100644
index 0000000..5f89e4d
--- /dev/null
+++ b/vggt/vggt/layers/block.py
@@ -0,0 +1,259 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/patch_embed.py
+
+import logging
+import os
+from typing import Callable, List, Any, Tuple, Dict
+import warnings
+
+import torch
+from torch import nn, Tensor
+
+from .attention import Attention
+from .drop_path import DropPath
+from .layer_scale import LayerScale
+from .mlp import Mlp
+
+
+XFORMERS_AVAILABLE = False
+
+
+class Block(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ num_heads: int,
+ mlp_ratio: float = 4.0,
+ qkv_bias: bool = True,
+ proj_bias: bool = True,
+ ffn_bias: bool = True,
+ drop: float = 0.0,
+ attn_drop: float = 0.0,
+ init_values=None,
+ drop_path: float = 0.0,
+ act_layer: Callable[..., nn.Module] = nn.GELU,
+ norm_layer: Callable[..., nn.Module] = nn.LayerNorm,
+ attn_class: Callable[..., nn.Module] = Attention,
+ ffn_layer: Callable[..., nn.Module] = Mlp,
+ qk_norm: bool = False,
+ fused_attn: bool = True, # use F.scaled_dot_product_attention or not
+ rope=None,
+ ) -> None:
+ super().__init__()
+
+ self.norm1 = norm_layer(dim)
+
+ self.attn = attn_class(
+ dim,
+ num_heads=num_heads,
+ qkv_bias=qkv_bias,
+ proj_bias=proj_bias,
+ attn_drop=attn_drop,
+ proj_drop=drop,
+ qk_norm=qk_norm,
+ fused_attn=fused_attn,
+ rope=rope,
+ )
+
+ self.ls1 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity()
+ self.drop_path1 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity()
+
+ self.norm2 = norm_layer(dim)
+ mlp_hidden_dim = int(dim * mlp_ratio)
+ self.mlp = ffn_layer(
+ in_features=dim,
+ hidden_features=mlp_hidden_dim,
+ act_layer=act_layer,
+ drop=drop,
+ bias=ffn_bias,
+ )
+ self.ls2 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity()
+ self.drop_path2 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity()
+
+ self.sample_drop_ratio = drop_path
+
+ def forward(self, x: Tensor, pos=None) -> Tensor:
+ def attn_residual_func(x: Tensor, pos=None) -> Tensor:
+ return self.ls1(self.attn(self.norm1(x), pos=pos))
+
+ def ffn_residual_func(x: Tensor) -> Tensor:
+ return self.ls2(self.mlp(self.norm2(x)))
+
+ if self.training and self.sample_drop_ratio > 0.1:
+ # the overhead is compensated only for a drop path rate larger than 0.1
+ x = drop_add_residual_stochastic_depth(
+ x,
+ pos=pos,
+ residual_func=attn_residual_func,
+ sample_drop_ratio=self.sample_drop_ratio,
+ )
+ x = drop_add_residual_stochastic_depth(
+ x,
+ residual_func=ffn_residual_func,
+ sample_drop_ratio=self.sample_drop_ratio,
+ )
+ elif self.training and self.sample_drop_ratio > 0.0:
+ x = x + self.drop_path1(attn_residual_func(x, pos=pos))
+ x = x + self.drop_path1(ffn_residual_func(x)) # FIXME: drop_path2
+ else:
+ x = x + attn_residual_func(x, pos=pos)
+ x = x + ffn_residual_func(x)
+ return x
+
+
+def drop_add_residual_stochastic_depth(
+ x: Tensor,
+ residual_func: Callable[[Tensor], Tensor],
+ sample_drop_ratio: float = 0.0,
+ pos=None,
+) -> Tensor:
+ # 1) extract subset using permutation
+ b, n, d = x.shape
+ sample_subset_size = max(int(b * (1 - sample_drop_ratio)), 1)
+ brange = (torch.randperm(b, device=x.device))[:sample_subset_size]
+ x_subset = x[brange]
+
+ # 2) apply residual_func to get residual
+ if pos is not None:
+ # if necessary, apply rope to the subset
+ pos = pos[brange]
+ residual = residual_func(x_subset, pos=pos)
+ else:
+ residual = residual_func(x_subset)
+
+ x_flat = x.flatten(1)
+ residual = residual.flatten(1)
+
+ residual_scale_factor = b / sample_subset_size
+
+ # 3) add the residual
+ x_plus_residual = torch.index_add(x_flat, 0, brange, residual.to(dtype=x.dtype), alpha=residual_scale_factor)
+ return x_plus_residual.view_as(x)
+
+
+def get_branges_scales(x, sample_drop_ratio=0.0):
+ b, n, d = x.shape
+ sample_subset_size = max(int(b * (1 - sample_drop_ratio)), 1)
+ brange = (torch.randperm(b, device=x.device))[:sample_subset_size]
+ residual_scale_factor = b / sample_subset_size
+ return brange, residual_scale_factor
+
+
+def add_residual(x, brange, residual, residual_scale_factor, scaling_vector=None):
+ if scaling_vector is None:
+ x_flat = x.flatten(1)
+ residual = residual.flatten(1)
+ x_plus_residual = torch.index_add(x_flat, 0, brange, residual.to(dtype=x.dtype), alpha=residual_scale_factor)
+ else:
+ x_plus_residual = scaled_index_add(
+ x, brange, residual.to(dtype=x.dtype), scaling=scaling_vector, alpha=residual_scale_factor
+ )
+ return x_plus_residual
+
+
+attn_bias_cache: Dict[Tuple, Any] = {}
+
+
+def get_attn_bias_and_cat(x_list, branges=None):
+ """
+ this will perform the index select, cat the tensors, and provide the attn_bias from cache
+ """
+ batch_sizes = [b.shape[0] for b in branges] if branges is not None else [x.shape[0] for x in x_list]
+ all_shapes = tuple((b, x.shape[1]) for b, x in zip(batch_sizes, x_list))
+ if all_shapes not in attn_bias_cache.keys():
+ seqlens = []
+ for b, x in zip(batch_sizes, x_list):
+ for _ in range(b):
+ seqlens.append(x.shape[1])
+ attn_bias = fmha.BlockDiagonalMask.from_seqlens(seqlens)
+ attn_bias._batch_sizes = batch_sizes
+ attn_bias_cache[all_shapes] = attn_bias
+
+ if branges is not None:
+ cat_tensors = index_select_cat([x.flatten(1) for x in x_list], branges).view(1, -1, x_list[0].shape[-1])
+ else:
+ tensors_bs1 = tuple(x.reshape([1, -1, *x.shape[2:]]) for x in x_list)
+ cat_tensors = torch.cat(tensors_bs1, dim=1)
+
+ return attn_bias_cache[all_shapes], cat_tensors
+
+
+def drop_add_residual_stochastic_depth_list(
+ x_list: List[Tensor],
+ residual_func: Callable[[Tensor, Any], Tensor],
+ sample_drop_ratio: float = 0.0,
+ scaling_vector=None,
+) -> Tensor:
+ # 1) generate random set of indices for dropping samples in the batch
+ branges_scales = [get_branges_scales(x, sample_drop_ratio=sample_drop_ratio) for x in x_list]
+ branges = [s[0] for s in branges_scales]
+ residual_scale_factors = [s[1] for s in branges_scales]
+
+ # 2) get attention bias and index+concat the tensors
+ attn_bias, x_cat = get_attn_bias_and_cat(x_list, branges)
+
+ # 3) apply residual_func to get residual, and split the result
+ residual_list = attn_bias.split(residual_func(x_cat, attn_bias=attn_bias)) # type: ignore
+
+ outputs = []
+ for x, brange, residual, residual_scale_factor in zip(x_list, branges, residual_list, residual_scale_factors):
+ outputs.append(add_residual(x, brange, residual, residual_scale_factor, scaling_vector).view_as(x))
+ return outputs
+
+
+class NestedTensorBlock(Block):
+ def forward_nested(self, x_list: List[Tensor]) -> List[Tensor]:
+ """
+ x_list contains a list of tensors to nest together and run
+ """
+ assert isinstance(self.attn, MemEffAttention)
+
+ if self.training and self.sample_drop_ratio > 0.0:
+
+ def attn_residual_func(x: Tensor, attn_bias=None) -> Tensor:
+ return self.attn(self.norm1(x), attn_bias=attn_bias)
+
+ def ffn_residual_func(x: Tensor, attn_bias=None) -> Tensor:
+ return self.mlp(self.norm2(x))
+
+ x_list = drop_add_residual_stochastic_depth_list(
+ x_list,
+ residual_func=attn_residual_func,
+ sample_drop_ratio=self.sample_drop_ratio,
+ scaling_vector=self.ls1.gamma if isinstance(self.ls1, LayerScale) else None,
+ )
+ x_list = drop_add_residual_stochastic_depth_list(
+ x_list,
+ residual_func=ffn_residual_func,
+ sample_drop_ratio=self.sample_drop_ratio,
+ scaling_vector=self.ls2.gamma if isinstance(self.ls1, LayerScale) else None,
+ )
+ return x_list
+ else:
+
+ def attn_residual_func(x: Tensor, attn_bias=None) -> Tensor:
+ return self.ls1(self.attn(self.norm1(x), attn_bias=attn_bias))
+
+ def ffn_residual_func(x: Tensor, attn_bias=None) -> Tensor:
+ return self.ls2(self.mlp(self.norm2(x)))
+
+ attn_bias, x = get_attn_bias_and_cat(x_list)
+ x = x + attn_residual_func(x, attn_bias=attn_bias)
+ x = x + ffn_residual_func(x)
+ return attn_bias.split(x)
+
+ def forward(self, x_or_x_list):
+ if isinstance(x_or_x_list, Tensor):
+ return super().forward(x_or_x_list)
+ elif isinstance(x_or_x_list, list):
+ if not XFORMERS_AVAILABLE:
+ raise AssertionError("xFormers is required for using nested tensors")
+ return self.forward_nested(x_or_x_list)
+ else:
+ raise AssertionError
diff --git a/vggt/vggt/layers/drop_path.py b/vggt/vggt/layers/drop_path.py
new file mode 100644
index 0000000..1d640e0
--- /dev/null
+++ b/vggt/vggt/layers/drop_path.py
@@ -0,0 +1,34 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/drop.py
+
+
+from torch import nn
+
+
+def drop_path(x, drop_prob: float = 0.0, training: bool = False):
+ if drop_prob == 0.0 or not training:
+ return x
+ keep_prob = 1 - drop_prob
+ shape = (x.shape[0],) + (1,) * (x.ndim - 1) # work with diff dim tensors, not just 2D ConvNets
+ random_tensor = x.new_empty(shape).bernoulli_(keep_prob)
+ if keep_prob > 0.0:
+ random_tensor.div_(keep_prob)
+ output = x * random_tensor
+ return output
+
+
+class DropPath(nn.Module):
+ """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks)."""
+
+ def __init__(self, drop_prob=None):
+ super(DropPath, self).__init__()
+ self.drop_prob = drop_prob
+
+ def forward(self, x):
+ return drop_path(x, self.drop_prob, self.training)
diff --git a/vggt/vggt/layers/layer_scale.py b/vggt/vggt/layers/layer_scale.py
new file mode 100644
index 0000000..51df0d7
--- /dev/null
+++ b/vggt/vggt/layers/layer_scale.py
@@ -0,0 +1,27 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# Modified from: https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/vision_transformer.py#L103-L110
+
+from typing import Union
+
+import torch
+from torch import Tensor
+from torch import nn
+
+
+class LayerScale(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ init_values: Union[float, Tensor] = 1e-5,
+ inplace: bool = False,
+ ) -> None:
+ super().__init__()
+ self.inplace = inplace
+ self.gamma = nn.Parameter(init_values * torch.ones(dim))
+
+ def forward(self, x: Tensor) -> Tensor:
+ return x.mul_(self.gamma) if self.inplace else x * self.gamma
diff --git a/vggt/vggt/layers/mlp.py b/vggt/vggt/layers/mlp.py
new file mode 100644
index 0000000..bbf9432
--- /dev/null
+++ b/vggt/vggt/layers/mlp.py
@@ -0,0 +1,40 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/mlp.py
+
+
+from typing import Callable, Optional
+
+from torch import Tensor, nn
+
+
+class Mlp(nn.Module):
+ def __init__(
+ self,
+ in_features: int,
+ hidden_features: Optional[int] = None,
+ out_features: Optional[int] = None,
+ act_layer: Callable[..., nn.Module] = nn.GELU,
+ drop: float = 0.0,
+ bias: bool = True,
+ ) -> None:
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.fc1 = nn.Linear(in_features, hidden_features, bias=bias)
+ self.act = act_layer()
+ self.fc2 = nn.Linear(hidden_features, out_features, bias=bias)
+ self.drop = nn.Dropout(drop)
+
+ def forward(self, x: Tensor) -> Tensor:
+ x = self.fc1(x)
+ x = self.act(x)
+ x = self.drop(x)
+ x = self.fc2(x)
+ x = self.drop(x)
+ return x
diff --git a/vggt/vggt/layers/patch_embed.py b/vggt/vggt/layers/patch_embed.py
new file mode 100644
index 0000000..8b7c080
--- /dev/null
+++ b/vggt/vggt/layers/patch_embed.py
@@ -0,0 +1,88 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/patch_embed.py
+
+from typing import Callable, Optional, Tuple, Union
+
+from torch import Tensor
+import torch.nn as nn
+
+
+def make_2tuple(x):
+ if isinstance(x, tuple):
+ assert len(x) == 2
+ return x
+
+ assert isinstance(x, int)
+ return (x, x)
+
+
+class PatchEmbed(nn.Module):
+ """
+ 2D image to patch embedding: (B,C,H,W) -> (B,N,D)
+
+ Args:
+ img_size: Image size.
+ patch_size: Patch token size.
+ in_chans: Number of input image channels.
+ embed_dim: Number of linear projection output channels.
+ norm_layer: Normalization layer.
+ """
+
+ def __init__(
+ self,
+ img_size: Union[int, Tuple[int, int]] = 224,
+ patch_size: Union[int, Tuple[int, int]] = 16,
+ in_chans: int = 3,
+ embed_dim: int = 768,
+ norm_layer: Optional[Callable] = None,
+ flatten_embedding: bool = True,
+ ) -> None:
+ super().__init__()
+
+ image_HW = make_2tuple(img_size)
+ patch_HW = make_2tuple(patch_size)
+ patch_grid_size = (
+ image_HW[0] // patch_HW[0],
+ image_HW[1] // patch_HW[1],
+ )
+
+ self.img_size = image_HW
+ self.patch_size = patch_HW
+ self.patches_resolution = patch_grid_size
+ self.num_patches = patch_grid_size[0] * patch_grid_size[1]
+
+ self.in_chans = in_chans
+ self.embed_dim = embed_dim
+
+ self.flatten_embedding = flatten_embedding
+
+ self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_HW, stride=patch_HW)
+ self.norm = norm_layer(embed_dim) if norm_layer else nn.Identity()
+
+ def forward(self, x: Tensor) -> Tensor:
+ _, _, H, W = x.shape
+ patch_H, patch_W = self.patch_size
+
+ assert H % patch_H == 0, f"Input image height {H} is not a multiple of patch height {patch_H}"
+ assert W % patch_W == 0, f"Input image width {W} is not a multiple of patch width: {patch_W}"
+
+ x = self.proj(x) # B C H W
+ H, W = x.size(2), x.size(3)
+ x = x.flatten(2).transpose(1, 2) # B HW C
+ x = self.norm(x)
+ if not self.flatten_embedding:
+ x = x.reshape(-1, H, W, self.embed_dim) # B H W C
+ return x
+
+ def flops(self) -> float:
+ Ho, Wo = self.patches_resolution
+ flops = Ho * Wo * self.embed_dim * self.in_chans * (self.patch_size[0] * self.patch_size[1])
+ if self.norm is not None:
+ flops += Ho * Wo * self.embed_dim
+ return flops
diff --git a/vggt/vggt/layers/rope.py b/vggt/vggt/layers/rope.py
new file mode 100644
index 0000000..4d5d333
--- /dev/null
+++ b/vggt/vggt/layers/rope.py
@@ -0,0 +1,188 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+
+# Implementation of 2D Rotary Position Embeddings (RoPE).
+
+# This module provides a clean implementation of 2D Rotary Position Embeddings,
+# which extends the original RoPE concept to handle 2D spatial positions.
+
+# Inspired by:
+# https://github.com/meta-llama/codellama/blob/main/llama/model.py
+# https://github.com/naver-ai/rope-vit
+
+
+import numpy as np
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from typing import Dict, Tuple
+
+
+class PositionGetter:
+ """Generates and caches 2D spatial positions for patches in a grid.
+
+ This class efficiently manages the generation of spatial coordinates for patches
+ in a 2D grid, caching results to avoid redundant computations.
+
+ Attributes:
+ position_cache: Dictionary storing precomputed position tensors for different
+ grid dimensions.
+ """
+
+ def __init__(self):
+ """Initializes the position generator with an empty cache."""
+ self.position_cache: Dict[Tuple[int, int], torch.Tensor] = {}
+
+ def __call__(self, batch_size: int, height: int, width: int, device: torch.device) -> torch.Tensor:
+ """Generates spatial positions for a batch of patches.
+
+ Args:
+ batch_size: Number of samples in the batch.
+ height: Height of the grid in patches.
+ width: Width of the grid in patches.
+ device: Target device for the position tensor.
+
+ Returns:
+ Tensor of shape (batch_size, height*width, 2) containing y,x coordinates
+ for each position in the grid, repeated for each batch item.
+ """
+ if (height, width) not in self.position_cache:
+ y_coords = torch.arange(height, device=device)
+ x_coords = torch.arange(width, device=device)
+ positions = torch.cartesian_prod(y_coords, x_coords)
+ self.position_cache[height, width] = positions
+
+ cached_positions = self.position_cache[height, width]
+ return cached_positions.view(1, height * width, 2).expand(batch_size, -1, -1).clone()
+
+
+class RotaryPositionEmbedding2D(nn.Module):
+ """2D Rotary Position Embedding implementation.
+
+ This module applies rotary position embeddings to input tokens based on their
+ 2D spatial positions. It handles the position-dependent rotation of features
+ separately for vertical and horizontal dimensions.
+
+ Args:
+ frequency: Base frequency for the position embeddings. Default: 100.0
+ scaling_factor: Scaling factor for frequency computation. Default: 1.0
+
+ Attributes:
+ base_frequency: Base frequency for computing position embeddings.
+ scaling_factor: Factor to scale the computed frequencies.
+ frequency_cache: Cache for storing precomputed frequency components.
+ """
+
+ def __init__(self, frequency: float = 100.0, scaling_factor: float = 1.0):
+ """Initializes the 2D RoPE module."""
+ super().__init__()
+ self.base_frequency = frequency
+ self.scaling_factor = scaling_factor
+ self.frequency_cache: Dict[Tuple, Tuple[torch.Tensor, torch.Tensor]] = {}
+
+ def _compute_frequency_components(
+ self, dim: int, seq_len: int, device: torch.device, dtype: torch.dtype
+ ) -> Tuple[torch.Tensor, torch.Tensor]:
+ """Computes frequency components for rotary embeddings.
+
+ Args:
+ dim: Feature dimension (must be even).
+ seq_len: Maximum sequence length.
+ device: Target device for computations.
+ dtype: Data type for the computed tensors.
+
+ Returns:
+ Tuple of (cosine, sine) tensors for frequency components.
+ """
+ cache_key = (dim, seq_len, device, dtype)
+ if cache_key not in self.frequency_cache:
+ # Compute frequency bands
+ exponents = torch.arange(0, dim, 2, device=device).float() / dim
+ inv_freq = 1.0 / (self.base_frequency**exponents)
+
+ # Generate position-dependent frequencies
+ positions = torch.arange(seq_len, device=device, dtype=inv_freq.dtype)
+ angles = torch.einsum("i,j->ij", positions, inv_freq)
+
+ # Compute and cache frequency components
+ angles = angles.to(dtype)
+ angles = torch.cat((angles, angles), dim=-1)
+ cos_components = angles.cos().to(dtype)
+ sin_components = angles.sin().to(dtype)
+ self.frequency_cache[cache_key] = (cos_components, sin_components)
+
+ return self.frequency_cache[cache_key]
+
+ @staticmethod
+ def _rotate_features(x: torch.Tensor) -> torch.Tensor:
+ """Performs feature rotation by splitting and recombining feature dimensions.
+
+ Args:
+ x: Input tensor to rotate.
+
+ Returns:
+ Rotated feature tensor.
+ """
+ feature_dim = x.shape[-1]
+ x1, x2 = x[..., : feature_dim // 2], x[..., feature_dim // 2 :]
+ return torch.cat((-x2, x1), dim=-1)
+
+ def _apply_1d_rope(
+ self, tokens: torch.Tensor, positions: torch.Tensor, cos_comp: torch.Tensor, sin_comp: torch.Tensor
+ ) -> torch.Tensor:
+ """Applies 1D rotary position embeddings along one dimension.
+
+ Args:
+ tokens: Input token features.
+ positions: Position indices.
+ cos_comp: Cosine components for rotation.
+ sin_comp: Sine components for rotation.
+
+ Returns:
+ Tokens with applied rotary position embeddings.
+ """
+ # Embed positions with frequency components
+ cos = F.embedding(positions, cos_comp)[:, None, :, :]
+ sin = F.embedding(positions, sin_comp)[:, None, :, :]
+
+ # Apply rotation
+ return (tokens * cos) + (self._rotate_features(tokens) * sin)
+
+ def forward(self, tokens: torch.Tensor, positions: torch.Tensor) -> torch.Tensor:
+ """Applies 2D rotary position embeddings to input tokens.
+
+ Args:
+ tokens: Input tensor of shape (batch_size, n_heads, n_tokens, dim).
+ The feature dimension (dim) must be divisible by 4.
+ positions: Position tensor of shape (batch_size, n_tokens, 2) containing
+ the y and x coordinates for each token.
+
+ Returns:
+ Tensor of same shape as input with applied 2D rotary position embeddings.
+
+ Raises:
+ AssertionError: If input dimensions are invalid or positions are malformed.
+ """
+ # Validate inputs
+ assert tokens.size(-1) % 2 == 0, "Feature dimension must be even"
+ assert positions.ndim == 3 and positions.shape[-1] == 2, "Positions must have shape (batch_size, n_tokens, 2)"
+
+ # Compute feature dimension for each spatial direction
+ feature_dim = tokens.size(-1) // 2
+
+ # Get frequency components
+ max_position = int(positions.max()) + 1
+ cos_comp, sin_comp = self._compute_frequency_components(feature_dim, max_position, tokens.device, tokens.dtype)
+
+ # Split features for vertical and horizontal processing
+ vertical_features, horizontal_features = tokens.chunk(2, dim=-1)
+
+ # Apply RoPE separately for each dimension
+ vertical_features = self._apply_1d_rope(vertical_features, positions[..., 0], cos_comp, sin_comp)
+ horizontal_features = self._apply_1d_rope(horizontal_features, positions[..., 1], cos_comp, sin_comp)
+
+ # Combine processed features
+ return torch.cat((vertical_features, horizontal_features), dim=-1)
diff --git a/vggt/vggt/layers/swiglu_ffn.py b/vggt/vggt/layers/swiglu_ffn.py
new file mode 100644
index 0000000..54fe8e9
--- /dev/null
+++ b/vggt/vggt/layers/swiglu_ffn.py
@@ -0,0 +1,72 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+import os
+from typing import Callable, Optional
+import warnings
+
+from torch import Tensor, nn
+import torch.nn.functional as F
+
+
+class SwiGLUFFN(nn.Module):
+ def __init__(
+ self,
+ in_features: int,
+ hidden_features: Optional[int] = None,
+ out_features: Optional[int] = None,
+ act_layer: Callable[..., nn.Module] = None,
+ drop: float = 0.0,
+ bias: bool = True,
+ ) -> None:
+ super().__init__()
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ self.w12 = nn.Linear(in_features, 2 * hidden_features, bias=bias)
+ self.w3 = nn.Linear(hidden_features, out_features, bias=bias)
+
+ def forward(self, x: Tensor) -> Tensor:
+ x12 = self.w12(x)
+ x1, x2 = x12.chunk(2, dim=-1)
+ hidden = F.silu(x1) * x2
+ return self.w3(hidden)
+
+
+XFORMERS_ENABLED = os.environ.get("XFORMERS_DISABLED") is None
+# try:
+# if XFORMERS_ENABLED:
+# from xformers.ops import SwiGLU
+
+# XFORMERS_AVAILABLE = True
+# warnings.warn("xFormers is available (SwiGLU)")
+# else:
+# warnings.warn("xFormers is disabled (SwiGLU)")
+# raise ImportError
+# except ImportError:
+SwiGLU = SwiGLUFFN
+XFORMERS_AVAILABLE = False
+
+# warnings.warn("xFormers is not available (SwiGLU)")
+
+
+class SwiGLUFFNFused(SwiGLU):
+ def __init__(
+ self,
+ in_features: int,
+ hidden_features: Optional[int] = None,
+ out_features: Optional[int] = None,
+ act_layer: Callable[..., nn.Module] = None,
+ drop: float = 0.0,
+ bias: bool = True,
+ ) -> None:
+ out_features = out_features or in_features
+ hidden_features = hidden_features or in_features
+ hidden_features = (int(hidden_features * 2 / 3) + 7) // 8 * 8
+ super().__init__(
+ in_features=in_features,
+ hidden_features=hidden_features,
+ out_features=out_features,
+ bias=bias,
+ )
diff --git a/vggt/vggt/layers/vision_transformer.py b/vggt/vggt/layers/vision_transformer.py
new file mode 100644
index 0000000..120cbe6
--- /dev/null
+++ b/vggt/vggt/layers/vision_transformer.py
@@ -0,0 +1,407 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+#
+# This source code is licensed under the Apache License, Version 2.0
+# found in the LICENSE file in the root directory of this source tree.
+
+# References:
+# https://github.com/facebookresearch/dino/blob/main/vision_transformer.py
+# https://github.com/rwightman/pytorch-image-models/tree/master/timm/models/vision_transformer.py
+
+from functools import partial
+import math
+import logging
+from typing import Sequence, Tuple, Union, Callable
+
+import torch
+import torch.nn as nn
+from torch.utils.checkpoint import checkpoint
+from torch.nn.init import trunc_normal_
+from . import Mlp, PatchEmbed, SwiGLUFFNFused, MemEffAttention, NestedTensorBlock as Block
+
+logger = logging.getLogger("dinov2")
+
+
+def named_apply(fn: Callable, module: nn.Module, name="", depth_first=True, include_root=False) -> nn.Module:
+ if not depth_first and include_root:
+ fn(module=module, name=name)
+ for child_name, child_module in module.named_children():
+ child_name = ".".join((name, child_name)) if name else child_name
+ named_apply(fn=fn, module=child_module, name=child_name, depth_first=depth_first, include_root=True)
+ if depth_first and include_root:
+ fn(module=module, name=name)
+ return module
+
+
+class BlockChunk(nn.ModuleList):
+ def forward(self, x):
+ for b in self:
+ x = b(x)
+ return x
+
+
+class DinoVisionTransformer(nn.Module):
+ def __init__(
+ self,
+ img_size=224,
+ patch_size=16,
+ in_chans=3,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ mlp_ratio=4.0,
+ qkv_bias=True,
+ ffn_bias=True,
+ proj_bias=True,
+ drop_path_rate=0.0,
+ drop_path_uniform=False,
+ init_values=None, # for layerscale: None or 0 => no layerscale
+ embed_layer=PatchEmbed,
+ act_layer=nn.GELU,
+ block_fn=Block,
+ ffn_layer="mlp",
+ block_chunks=1,
+ num_register_tokens=0,
+ interpolate_antialias=False,
+ interpolate_offset=0.1,
+ qk_norm=False,
+ ):
+ """
+ Args:
+ img_size (int, tuple): input image size
+ patch_size (int, tuple): patch size
+ in_chans (int): number of input channels
+ embed_dim (int): embedding dimension
+ depth (int): depth of transformer
+ num_heads (int): number of attention heads
+ mlp_ratio (int): ratio of mlp hidden dim to embedding dim
+ qkv_bias (bool): enable bias for qkv if True
+ proj_bias (bool): enable bias for proj in attn if True
+ ffn_bias (bool): enable bias for ffn if True
+ drop_path_rate (float): stochastic depth rate
+ drop_path_uniform (bool): apply uniform drop rate across blocks
+ weight_init (str): weight init scheme
+ init_values (float): layer-scale init values
+ embed_layer (nn.Module): patch embedding layer
+ act_layer (nn.Module): MLP activation layer
+ block_fn (nn.Module): transformer block class
+ ffn_layer (str): "mlp", "swiglu", "swiglufused" or "identity"
+ block_chunks: (int) split block sequence into block_chunks units for FSDP wrap
+ num_register_tokens: (int) number of extra cls tokens (so-called "registers")
+ interpolate_antialias: (str) flag to apply anti-aliasing when interpolating positional embeddings
+ interpolate_offset: (float) work-around offset to apply when interpolating positional embeddings
+ """
+ super().__init__()
+ norm_layer = partial(nn.LayerNorm, eps=1e-6)
+
+ # tricky but makes it work
+ self.use_checkpoint = False
+ #
+
+ self.num_features = self.embed_dim = embed_dim # num_features for consistency with other models
+ self.num_tokens = 1
+ self.n_blocks = depth
+ self.num_heads = num_heads
+ self.patch_size = patch_size
+ self.num_register_tokens = num_register_tokens
+ self.interpolate_antialias = interpolate_antialias
+ self.interpolate_offset = interpolate_offset
+
+ self.patch_embed = embed_layer(img_size=img_size, patch_size=patch_size, in_chans=in_chans, embed_dim=embed_dim)
+ num_patches = self.patch_embed.num_patches
+
+ self.cls_token = nn.Parameter(torch.zeros(1, 1, embed_dim))
+ self.pos_embed = nn.Parameter(torch.zeros(1, num_patches + self.num_tokens, embed_dim))
+ assert num_register_tokens >= 0
+ self.register_tokens = (
+ nn.Parameter(torch.zeros(1, num_register_tokens, embed_dim)) if num_register_tokens else None
+ )
+
+ if drop_path_uniform is True:
+ dpr = [drop_path_rate] * depth
+ else:
+ dpr = [x.item() for x in torch.linspace(0, drop_path_rate, depth)] # stochastic depth decay rule
+
+ if ffn_layer == "mlp":
+ logger.info("using MLP layer as FFN")
+ ffn_layer = Mlp
+ elif ffn_layer == "swiglufused" or ffn_layer == "swiglu":
+ logger.info("using SwiGLU layer as FFN")
+ ffn_layer = SwiGLUFFNFused
+ elif ffn_layer == "identity":
+ logger.info("using Identity layer as FFN")
+
+ def f(*args, **kwargs):
+ return nn.Identity()
+
+ ffn_layer = f
+ else:
+ raise NotImplementedError
+
+ blocks_list = [
+ block_fn(
+ dim=embed_dim,
+ num_heads=num_heads,
+ mlp_ratio=mlp_ratio,
+ qkv_bias=qkv_bias,
+ proj_bias=proj_bias,
+ ffn_bias=ffn_bias,
+ drop_path=dpr[i],
+ norm_layer=norm_layer,
+ act_layer=act_layer,
+ ffn_layer=ffn_layer,
+ init_values=init_values,
+ qk_norm=qk_norm,
+ )
+ for i in range(depth)
+ ]
+ if block_chunks > 0:
+ self.chunked_blocks = True
+ chunked_blocks = []
+ chunksize = depth // block_chunks
+ for i in range(0, depth, chunksize):
+ # this is to keep the block index consistent if we chunk the block list
+ chunked_blocks.append([nn.Identity()] * i + blocks_list[i : i + chunksize])
+ self.blocks = nn.ModuleList([BlockChunk(p) for p in chunked_blocks])
+ else:
+ self.chunked_blocks = False
+ self.blocks = nn.ModuleList(blocks_list)
+
+ self.norm = norm_layer(embed_dim)
+ self.head = nn.Identity()
+
+ self.mask_token = nn.Parameter(torch.zeros(1, embed_dim))
+
+ self.init_weights()
+
+ def init_weights(self):
+ trunc_normal_(self.pos_embed, std=0.02)
+ nn.init.normal_(self.cls_token, std=1e-6)
+ if self.register_tokens is not None:
+ nn.init.normal_(self.register_tokens, std=1e-6)
+ named_apply(init_weights_vit_timm, self)
+
+ def interpolate_pos_encoding(self, x, w, h):
+ previous_dtype = x.dtype
+ npatch = x.shape[1] - 1
+ N = self.pos_embed.shape[1] - 1
+ if npatch == N and w == h:
+ return self.pos_embed
+ pos_embed = self.pos_embed.float()
+ class_pos_embed = pos_embed[:, 0]
+ patch_pos_embed = pos_embed[:, 1:]
+ dim = x.shape[-1]
+ w0 = w // self.patch_size
+ h0 = h // self.patch_size
+ M = int(math.sqrt(N)) # Recover the number of patches in each dimension
+ assert N == M * M
+ kwargs = {}
+ if self.interpolate_offset:
+ # Historical kludge: add a small number to avoid floating point error in the interpolation, see https://github.com/facebookresearch/dino/issues/8
+ # Note: still needed for backward-compatibility, the underlying operators are using both output size and scale factors
+ sx = float(w0 + self.interpolate_offset) / M
+ sy = float(h0 + self.interpolate_offset) / M
+ kwargs["scale_factor"] = (sx, sy)
+ else:
+ # Simply specify an output size instead of a scale factor
+ kwargs["size"] = (w0, h0)
+ patch_pos_embed = nn.functional.interpolate(
+ patch_pos_embed.reshape(1, M, M, dim).permute(0, 3, 1, 2),
+ mode="bicubic",
+ antialias=self.interpolate_antialias,
+ **kwargs,
+ )
+ assert (w0, h0) == patch_pos_embed.shape[-2:]
+ patch_pos_embed = patch_pos_embed.permute(0, 2, 3, 1).view(1, -1, dim)
+ return torch.cat((class_pos_embed.unsqueeze(0), patch_pos_embed), dim=1).to(previous_dtype)
+
+ def prepare_tokens_with_masks(self, x, masks=None):
+ B, nc, w, h = x.shape
+ x = self.patch_embed(x)
+ if masks is not None:
+ x = torch.where(masks.unsqueeze(-1), self.mask_token.to(x.dtype).unsqueeze(0), x)
+
+ x = torch.cat((self.cls_token.expand(x.shape[0], -1, -1), x), dim=1)
+ x = x + self.interpolate_pos_encoding(x, w, h)
+
+ if self.register_tokens is not None:
+ x = torch.cat(
+ (
+ x[:, :1],
+ self.register_tokens.expand(x.shape[0], -1, -1),
+ x[:, 1:],
+ ),
+ dim=1,
+ )
+
+ return x
+
+ def forward_features_list(self, x_list, masks_list):
+ x = [self.prepare_tokens_with_masks(x, masks) for x, masks in zip(x_list, masks_list)]
+
+ for blk in self.blocks:
+ if self.use_checkpoint:
+ x = checkpoint(blk, x, use_reentrant=self.use_reentrant)
+ else:
+ x = blk(x)
+
+ all_x = x
+ output = []
+ for x, masks in zip(all_x, masks_list):
+ x_norm = self.norm(x)
+ output.append(
+ {
+ "x_norm_clstoken": x_norm[:, 0],
+ "x_norm_regtokens": x_norm[:, 1 : self.num_register_tokens + 1],
+ "x_norm_patchtokens": x_norm[:, self.num_register_tokens + 1 :],
+ "x_prenorm": x,
+ "masks": masks,
+ }
+ )
+ return output
+
+ def forward_features(self, x, masks=None):
+ if isinstance(x, list):
+ return self.forward_features_list(x, masks)
+
+ x = self.prepare_tokens_with_masks(x, masks)
+
+ for blk in self.blocks:
+ if self.use_checkpoint:
+ x = checkpoint(blk, x, use_reentrant=self.use_reentrant)
+ else:
+ x = blk(x)
+
+ x_norm = self.norm(x)
+ return {
+ "x_norm_clstoken": x_norm[:, 0],
+ "x_norm_regtokens": x_norm[:, 1 : self.num_register_tokens + 1],
+ "x_norm_patchtokens": x_norm[:, self.num_register_tokens + 1 :],
+ "x_prenorm": x,
+ "masks": masks,
+ }
+
+ def _get_intermediate_layers_not_chunked(self, x, n=1):
+ x = self.prepare_tokens_with_masks(x)
+ # If n is an int, take the n last blocks. If it's a list, take them
+ output, total_block_len = [], len(self.blocks)
+ blocks_to_take = range(total_block_len - n, total_block_len) if isinstance(n, int) else n
+ for i, blk in enumerate(self.blocks):
+ x = blk(x)
+ if i in blocks_to_take:
+ output.append(x)
+ assert len(output) == len(blocks_to_take), f"only {len(output)} / {len(blocks_to_take)} blocks found"
+ return output
+
+ def _get_intermediate_layers_chunked(self, x, n=1):
+ x = self.prepare_tokens_with_masks(x)
+ output, i, total_block_len = [], 0, len(self.blocks[-1])
+ # If n is an int, take the n last blocks. If it's a list, take them
+ blocks_to_take = range(total_block_len - n, total_block_len) if isinstance(n, int) else n
+ for block_chunk in self.blocks:
+ for blk in block_chunk[i:]: # Passing the nn.Identity()
+ x = blk(x)
+ if i in blocks_to_take:
+ output.append(x)
+ i += 1
+ assert len(output) == len(blocks_to_take), f"only {len(output)} / {len(blocks_to_take)} blocks found"
+ return output
+
+ def get_intermediate_layers(
+ self,
+ x: torch.Tensor,
+ n: Union[int, Sequence] = 1, # Layers or n last layers to take
+ reshape: bool = False,
+ return_class_token: bool = False,
+ norm=True,
+ ) -> Tuple[Union[torch.Tensor, Tuple[torch.Tensor]]]:
+ if self.chunked_blocks:
+ outputs = self._get_intermediate_layers_chunked(x, n)
+ else:
+ outputs = self._get_intermediate_layers_not_chunked(x, n)
+ if norm:
+ outputs = [self.norm(out) for out in outputs]
+ class_tokens = [out[:, 0] for out in outputs]
+ outputs = [out[:, 1 + self.num_register_tokens :] for out in outputs]
+ if reshape:
+ B, _, w, h = x.shape
+ outputs = [
+ out.reshape(B, w // self.patch_size, h // self.patch_size, -1).permute(0, 3, 1, 2).contiguous()
+ for out in outputs
+ ]
+ if return_class_token:
+ return tuple(zip(outputs, class_tokens))
+ return tuple(outputs)
+
+ def forward(self, *args, is_training=True, **kwargs):
+ ret = self.forward_features(*args, **kwargs)
+ if is_training:
+ return ret
+ else:
+ return self.head(ret["x_norm_clstoken"])
+
+
+def init_weights_vit_timm(module: nn.Module, name: str = ""):
+ """ViT weight initialization, original timm impl (for reproducibility)"""
+ if isinstance(module, nn.Linear):
+ trunc_normal_(module.weight, std=0.02)
+ if module.bias is not None:
+ nn.init.zeros_(module.bias)
+
+
+def vit_small(patch_size=16, num_register_tokens=0, **kwargs):
+ model = DinoVisionTransformer(
+ patch_size=patch_size,
+ embed_dim=384,
+ depth=12,
+ num_heads=6,
+ mlp_ratio=4,
+ block_fn=partial(Block, attn_class=MemEffAttention),
+ num_register_tokens=num_register_tokens,
+ **kwargs,
+ )
+ return model
+
+
+def vit_base(patch_size=16, num_register_tokens=0, **kwargs):
+ model = DinoVisionTransformer(
+ patch_size=patch_size,
+ embed_dim=768,
+ depth=12,
+ num_heads=12,
+ mlp_ratio=4,
+ block_fn=partial(Block, attn_class=MemEffAttention),
+ num_register_tokens=num_register_tokens,
+ **kwargs,
+ )
+ return model
+
+
+def vit_large(patch_size=16, num_register_tokens=0, **kwargs):
+ model = DinoVisionTransformer(
+ patch_size=patch_size,
+ embed_dim=1024,
+ depth=24,
+ num_heads=16,
+ mlp_ratio=4,
+ block_fn=partial(Block, attn_class=MemEffAttention),
+ num_register_tokens=num_register_tokens,
+ **kwargs,
+ )
+ return model
+
+
+def vit_giant2(patch_size=16, num_register_tokens=0, **kwargs):
+ """
+ Close to ViT-giant, with embed-dim 1536 and 24 heads => embed-dim per head 64
+ """
+ model = DinoVisionTransformer(
+ patch_size=patch_size,
+ embed_dim=1536,
+ depth=40,
+ num_heads=24,
+ mlp_ratio=4,
+ block_fn=partial(Block, attn_class=MemEffAttention),
+ num_register_tokens=num_register_tokens,
+ **kwargs,
+ )
+ return model
diff --git a/vggt/vggt/models/aggregator.py b/vggt/vggt/models/aggregator.py
new file mode 100644
index 0000000..50db933
--- /dev/null
+++ b/vggt/vggt/models/aggregator.py
@@ -0,0 +1,331 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import logging
+import torch
+import torch.nn as nn
+import torch.nn.functional as F
+from typing import Optional, Tuple, Union, List, Dict, Any
+
+from ..layers import PatchEmbed
+from ..layers.block import Block
+from ..layers.rope import RotaryPositionEmbedding2D, PositionGetter
+from ..layers.vision_transformer import vit_small, vit_base, vit_large, vit_giant2
+
+logger = logging.getLogger(__name__)
+
+_RESNET_MEAN = [0.485, 0.456, 0.406]
+_RESNET_STD = [0.229, 0.224, 0.225]
+
+
+class Aggregator(nn.Module):
+ """
+ The Aggregator applies alternating-attention over input frames,
+ as described in VGGT: Visual Geometry Grounded Transformer.
+
+
+ Args:
+ img_size (int): Image size in pixels.
+ patch_size (int): Size of each patch for PatchEmbed.
+ embed_dim (int): Dimension of the token embeddings.
+ depth (int): Number of blocks.
+ num_heads (int): Number of attention heads.
+ mlp_ratio (float): Ratio of MLP hidden dim to embedding dim.
+ num_register_tokens (int): Number of register tokens.
+ block_fn (nn.Module): The block type used for attention (Block by default).
+ qkv_bias (bool): Whether to include bias in QKV projections.
+ proj_bias (bool): Whether to include bias in the output projection.
+ ffn_bias (bool): Whether to include bias in MLP layers.
+ patch_embed (str): Type of patch embed. e.g., "conv" or "dinov2_vitl14_reg".
+ aa_order (list[str]): The order of alternating attention, e.g. ["frame", "global"].
+ aa_block_size (int): How many blocks to group under each attention type before switching. If not necessary, set to 1.
+ qk_norm (bool): Whether to apply QK normalization.
+ rope_freq (int): Base frequency for rotary embedding. -1 to disable.
+ init_values (float): Init scale for layer scale.
+ """
+
+ def __init__(
+ self,
+ img_size=518,
+ patch_size=14,
+ embed_dim=1024,
+ depth=24,
+ num_heads=16,
+ mlp_ratio=4.0,
+ num_register_tokens=4,
+ block_fn=Block,
+ qkv_bias=True,
+ proj_bias=True,
+ ffn_bias=True,
+ patch_embed="dinov2_vitl14_reg",
+ aa_order=["frame", "global"],
+ aa_block_size=1,
+ qk_norm=True,
+ rope_freq=100,
+ init_values=0.01,
+ ):
+ super().__init__()
+
+ self.__build_patch_embed__(patch_embed, img_size, patch_size, num_register_tokens, embed_dim=embed_dim)
+
+ # Initialize rotary position embedding if frequency > 0
+ self.rope = RotaryPositionEmbedding2D(frequency=rope_freq) if rope_freq > 0 else None
+ self.position_getter = PositionGetter() if self.rope is not None else None
+
+ self.frame_blocks = nn.ModuleList(
+ [
+ block_fn(
+ dim=embed_dim,
+ num_heads=num_heads,
+ mlp_ratio=mlp_ratio,
+ qkv_bias=qkv_bias,
+ proj_bias=proj_bias,
+ ffn_bias=ffn_bias,
+ init_values=init_values,
+ qk_norm=qk_norm,
+ rope=self.rope,
+ )
+ for _ in range(depth)
+ ]
+ )
+
+ self.global_blocks = nn.ModuleList(
+ [
+ block_fn(
+ dim=embed_dim,
+ num_heads=num_heads,
+ mlp_ratio=mlp_ratio,
+ qkv_bias=qkv_bias,
+ proj_bias=proj_bias,
+ ffn_bias=ffn_bias,
+ init_values=init_values,
+ qk_norm=qk_norm,
+ rope=self.rope,
+ )
+ for _ in range(depth)
+ ]
+ )
+
+ self.depth = depth
+ self.aa_order = aa_order
+ self.patch_size = patch_size
+ self.aa_block_size = aa_block_size
+
+ # Validate that depth is divisible by aa_block_size
+ if self.depth % self.aa_block_size != 0:
+ raise ValueError(f"depth ({depth}) must be divisible by aa_block_size ({aa_block_size})")
+
+ self.aa_block_num = self.depth // self.aa_block_size
+
+ # Note: We have two camera tokens, one for the first frame and one for the rest
+ # The same applies for register tokens
+ self.camera_token = nn.Parameter(torch.randn(1, 2, 1, embed_dim))
+ self.register_token = nn.Parameter(torch.randn(1, 2, num_register_tokens, embed_dim))
+
+ # The patch tokens start after the camera and register tokens
+ self.patch_start_idx = 1 + num_register_tokens
+
+ # Initialize parameters with small values
+ nn.init.normal_(self.camera_token, std=1e-6)
+ nn.init.normal_(self.register_token, std=1e-6)
+
+ # Register normalization constants as buffers
+ for name, value in (
+ ("_resnet_mean", _RESNET_MEAN),
+ ("_resnet_std", _RESNET_STD),
+ ):
+ self.register_buffer(
+ name,
+ torch.FloatTensor(value).view(1, 1, 3, 1, 1),
+ persistent=False,
+ )
+
+ def __build_patch_embed__(
+ self,
+ patch_embed,
+ img_size,
+ patch_size,
+ num_register_tokens,
+ interpolate_antialias=True,
+ interpolate_offset=0.0,
+ block_chunks=0,
+ init_values=1.0,
+ embed_dim=1024,
+ ):
+ """
+ Build the patch embed layer. If 'conv', we use a
+ simple PatchEmbed conv layer. Otherwise, we use a vision transformer.
+ """
+
+ if "conv" in patch_embed:
+ self.patch_embed = PatchEmbed(img_size=img_size, patch_size=patch_size, in_chans=3, embed_dim=embed_dim)
+ else:
+ vit_models = {
+ "dinov2_vitl14_reg": vit_large,
+ "dinov2_vitb14_reg": vit_base,
+ "dinov2_vits14_reg": vit_small,
+ "dinov2_vitg2_reg": vit_giant2,
+ }
+
+ self.patch_embed = vit_models[patch_embed](
+ img_size=img_size,
+ patch_size=patch_size,
+ num_register_tokens=num_register_tokens,
+ interpolate_antialias=interpolate_antialias,
+ interpolate_offset=interpolate_offset,
+ block_chunks=block_chunks,
+ init_values=init_values,
+ )
+
+ # Disable gradient updates for mask token
+ if hasattr(self.patch_embed, "mask_token"):
+ self.patch_embed.mask_token.requires_grad_(False)
+
+ def forward(
+ self,
+ images: torch.Tensor,
+ ) -> Tuple[List[torch.Tensor], int]:
+ """
+ Args:
+ images (torch.Tensor): Input images with shape [B, S, 3, H, W], in range [0, 1].
+ B: batch size, S: sequence length, 3: RGB channels, H: height, W: width
+
+ Returns:
+ (list[torch.Tensor], int):
+ The list of outputs from the attention blocks,
+ and the patch_start_idx indicating where patch tokens begin.
+ """
+ B, S, C_in, H, W = images.shape
+
+ if C_in != 3:
+ raise ValueError(f"Expected 3 input channels, got {C_in}")
+
+ # Normalize images and reshape for patch embed
+ images = (images - self._resnet_mean) / self._resnet_std
+
+ # Reshape to [B*S, C, H, W] for patch embedding
+ images = images.view(B * S, C_in, H, W)
+ patch_tokens = self.patch_embed(images)
+
+ if isinstance(patch_tokens, dict):
+ patch_tokens = patch_tokens["x_norm_patchtokens"]
+
+ _, P, C = patch_tokens.shape
+
+ # Expand camera and register tokens to match batch size and sequence length
+ camera_token = slice_expand_and_flatten(self.camera_token, B, S)
+ register_token = slice_expand_and_flatten(self.register_token, B, S)
+
+ # Concatenate special tokens with patch tokens
+ tokens = torch.cat([camera_token, register_token, patch_tokens], dim=1)
+
+ pos = None
+ if self.rope is not None:
+ pos = self.position_getter(B * S, H // self.patch_size, W // self.patch_size, device=images.device)
+
+ if self.patch_start_idx > 0:
+ # do not use position embedding for special tokens (camera and register tokens)
+ # so set pos to 0 for the special tokens
+ pos = pos + 1
+ pos_special = torch.zeros(B * S, self.patch_start_idx, 2).to(images.device).to(pos.dtype)
+ pos = torch.cat([pos_special, pos], dim=1)
+
+ # update P because we added special tokens
+ _, P, C = tokens.shape
+
+ frame_idx = 0
+ global_idx = 0
+ output_list = []
+
+ for _ in range(self.aa_block_num):
+ for attn_type in self.aa_order:
+ if attn_type == "frame":
+ tokens, frame_idx, frame_intermediates = self._process_frame_attention(
+ tokens, B, S, P, C, frame_idx, pos=pos
+ )
+ elif attn_type == "global":
+ tokens, global_idx, global_intermediates = self._process_global_attention(
+ tokens, B, S, P, C, global_idx, pos=pos
+ )
+ else:
+ raise ValueError(f"Unknown attention type: {attn_type}")
+
+ for i in range(len(frame_intermediates)):
+ # concat frame and global intermediates, [B x S x P x 2C]
+ concat_inter = torch.cat([frame_intermediates[i], global_intermediates[i]], dim=-1)
+ output_list.append(concat_inter)
+
+ del concat_inter
+ del frame_intermediates
+ del global_intermediates
+ return output_list, self.patch_start_idx
+
+ def _process_frame_attention(self, tokens, B, S, P, C, frame_idx, pos=None):
+ """
+ Process frame attention blocks. We keep tokens in shape (B*S, P, C).
+ """
+ # If needed, reshape tokens or positions:
+ if tokens.shape != (B * S, P, C):
+ tokens = tokens.view(B, S, P, C).view(B * S, P, C)
+
+ if pos is not None and pos.shape != (B * S, P, 2):
+ pos = pos.view(B, S, P, 2).view(B * S, P, 2)
+
+ intermediates = []
+
+ # by default, self.aa_block_size=1, which processes one block at a time
+ for _ in range(self.aa_block_size):
+ tokens = self.frame_blocks[frame_idx](tokens, pos=pos)
+ frame_idx += 1
+ intermediates.append(tokens.view(B, S, P, C))
+
+ return tokens, frame_idx, intermediates
+
+ def _process_global_attention(self, tokens, B, S, P, C, global_idx, pos=None):
+ """
+ Process global attention blocks. We keep tokens in shape (B, S*P, C).
+ """
+ if tokens.shape != (B, S * P, C):
+ tokens = tokens.view(B, S, P, C).view(B, S * P, C)
+
+ if pos is not None and pos.shape != (B, S * P, 2):
+ pos = pos.view(B, S, P, 2).view(B, S * P, 2)
+
+ intermediates = []
+
+ # by default, self.aa_block_size=1, which processes one block at a time
+ for _ in range(self.aa_block_size):
+ tokens = self.global_blocks[global_idx](tokens, pos=pos)
+ global_idx += 1
+ intermediates.append(tokens.view(B, S, P, C))
+
+ return tokens, global_idx, intermediates
+
+
+def slice_expand_and_flatten(token_tensor, B, S):
+ """
+ Processes specialized tokens with shape (1, 2, X, C) for multi-frame processing:
+ 1) Uses the first position (index=0) for the first frame only
+ 2) Uses the second position (index=1) for all remaining frames (S-1 frames)
+ 3) Expands both to match batch size B
+ 4) Concatenates to form (B, S, X, C) where each sequence has 1 first-position token
+ followed by (S-1) second-position tokens
+ 5) Flattens to (B*S, X, C) for processing
+
+ Returns:
+ torch.Tensor: Processed tokens with shape (B*S, X, C)
+ """
+
+ # Slice out the "query" tokens => shape (1, 1, ...)
+ query = token_tensor[:, 0:1, ...].expand(B, 1, *token_tensor.shape[2:])
+ # Slice out the "other" tokens => shape (1, S-1, ...)
+ others = token_tensor[:, 1:, ...].expand(B, S - 1, *token_tensor.shape[2:])
+ # Concatenate => shape (B, S, ...)
+ combined = torch.cat([query, others], dim=1)
+
+ # Finally flatten => shape (B*S, ...)
+ combined = combined.view(B * S, *combined.shape[2:])
+ return combined
diff --git a/vggt/vggt/models/vggt.py b/vggt/vggt/models/vggt.py
new file mode 100644
index 0000000..25790f2
--- /dev/null
+++ b/vggt/vggt/models/vggt.py
@@ -0,0 +1,95 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch
+import torch.nn as nn
+from huggingface_hub import PyTorchModelHubMixin # used for model hub
+from .aggregator import Aggregator
+from ..heads.camera_head import CameraHead
+from ..heads.dpt_head import DPTHead
+from ..heads.track_head import TrackHead
+
+
+class VGGT(nn.Module, PyTorchModelHubMixin):
+ def __init__(self, img_size=518, patch_size=14, embed_dim=1024):
+ super().__init__()
+
+ self.aggregator = Aggregator(img_size=img_size, patch_size=patch_size, embed_dim=embed_dim)
+ self.camera_head = CameraHead(dim_in=2 * embed_dim)
+ self.point_head = DPTHead(dim_in=2 * embed_dim, output_dim=4, activation="inv_log", conf_activation="expp1")
+ self.depth_head = DPTHead(dim_in=2 * embed_dim, output_dim=2, activation="exp", conf_activation="expp1")
+ self.track_head = TrackHead(dim_in=2 * embed_dim, patch_size=patch_size)
+
+ def forward(
+ self,
+ images: torch.Tensor,
+ query_points: torch.Tensor = None,
+ ):
+ """
+ Forward pass of the VGGT model.
+
+ Args:
+ images (torch.Tensor): Input images with shape [S, 3, H, W] or [B, S, 3, H, W], in range [0, 1].
+ B: batch size, S: sequence length, 3: RGB channels, H: height, W: width
+ query_points (torch.Tensor, optional): Query points for tracking, in pixel coordinates.
+ Shape: [N, 2] or [B, N, 2], where N is the number of query points.
+ Default: None
+
+ Returns:
+ dict: A dictionary containing the following predictions:
+ - pose_enc (torch.Tensor): Camera pose encoding with shape [B, S, 9] (from the last iteration)
+ - depth (torch.Tensor): Predicted depth maps with shape [B, S, H, W, 1]
+ - depth_conf (torch.Tensor): Confidence scores for depth predictions with shape [B, S, H, W]
+ - world_points (torch.Tensor): 3D world coordinates for each pixel with shape [B, S, H, W, 3]
+ - world_points_conf (torch.Tensor): Confidence scores for world points with shape [B, S, H, W]
+ - images (torch.Tensor): Original input images, preserved for visualization
+
+ If query_points is provided, also includes:
+ - track (torch.Tensor): Point tracks with shape [B, S, N, 2] (from the last iteration), in pixel coordinates
+ - vis (torch.Tensor): Visibility scores for tracked points with shape [B, S, N]
+ - conf (torch.Tensor): Confidence scores for tracked points with shape [B, S, N]
+ """
+
+ # If without batch dimension, add it
+ if len(images.shape) == 4:
+ images = images.unsqueeze(0)
+ if query_points is not None and len(query_points.shape) == 2:
+ query_points = query_points.unsqueeze(0)
+
+ aggregated_tokens_list, patch_start_idx = self.aggregator(images)
+
+ predictions = {}
+
+ with torch.cuda.amp.autocast(enabled=False):
+ if self.camera_head is not None:
+ pose_enc_list = self.camera_head(aggregated_tokens_list)
+ predictions["pose_enc"] = pose_enc_list[-1] # pose encoding of the last iteration
+
+ if self.depth_head is not None:
+ depth, depth_conf = self.depth_head(
+ aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx
+ )
+ predictions["depth"] = depth
+ predictions["depth_conf"] = depth_conf
+
+ if self.point_head is not None:
+ pts3d, pts3d_conf = self.point_head(
+ aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx
+ )
+ predictions["world_points"] = pts3d
+ predictions["world_points_conf"] = pts3d_conf
+
+ if self.track_head is not None and query_points is not None:
+ track_list, vis, conf = self.track_head(
+ aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx, query_points=query_points
+ )
+ predictions["track"] = track_list[-1] # track of the last iteration
+ predictions["vis"] = vis
+ predictions["conf"] = conf
+
+ predictions["images"] = images
+
+ return predictions
diff --git a/vggt/vggt/utils/geometry.py b/vggt/vggt/utils/geometry.py
new file mode 100644
index 0000000..e9c3fde
--- /dev/null
+++ b/vggt/vggt/utils/geometry.py
@@ -0,0 +1,236 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import os
+import torch
+import numpy as np
+
+
+def unproject_depth_map_to_point_map(
+ depth_map: np.ndarray, extrinsics_cam: np.ndarray, intrinsics_cam: np.ndarray
+) -> np.ndarray:
+ """
+ Unproject a batch of depth maps to 3D world coordinates.
+
+ Args:
+ depth_map (np.ndarray): Batch of depth maps of shape (S, H, W, 1) or (S, H, W)
+ extrinsics_cam (np.ndarray): Batch of camera extrinsic matrices of shape (S, 3, 4)
+ intrinsics_cam (np.ndarray): Batch of camera intrinsic matrices of shape (S, 3, 3)
+
+ Returns:
+ np.ndarray: Batch of 3D world coordinates of shape (S, H, W, 3)
+ """
+ if isinstance(depth_map, torch.Tensor):
+ depth_map = depth_map.cpu().numpy()
+ if isinstance(extrinsics_cam, torch.Tensor):
+ extrinsics_cam = extrinsics_cam.cpu().numpy()
+ if isinstance(intrinsics_cam, torch.Tensor):
+ intrinsics_cam = intrinsics_cam.cpu().numpy()
+
+ world_points_list = []
+ for frame_idx in range(depth_map.shape[0]):
+ cur_world_points, _, _ = depth_to_world_coords_points(
+ depth_map[frame_idx].squeeze(-1), extrinsics_cam[frame_idx], intrinsics_cam[frame_idx]
+ )
+ world_points_list.append(cur_world_points)
+ world_points_array = np.stack(world_points_list, axis=0)
+
+ return world_points_array
+
+
+def depth_to_world_coords_points(
+ depth_map: np.ndarray,
+ extrinsic: np.ndarray,
+ intrinsic: np.ndarray,
+ eps=1e-8,
+) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
+ """
+ Convert a depth map to world coordinates.
+
+ Args:
+ depth_map (np.ndarray): Depth map of shape (H, W).
+ intrinsic (np.ndarray): Camera intrinsic matrix of shape (3, 3).
+ extrinsic (np.ndarray): Camera extrinsic matrix of shape (3, 4). OpenCV camera coordinate convention, cam from world.
+
+ Returns:
+ tuple[np.ndarray, np.ndarray]: World coordinates (H, W, 3) and valid depth mask (H, W).
+ """
+ if depth_map is None:
+ return None, None, None
+
+ # Valid depth mask
+ point_mask = depth_map > eps
+
+ # Convert depth map to camera coordinates
+ cam_coords_points = depth_to_cam_coords_points(depth_map, intrinsic)
+
+ # Multiply with the inverse of extrinsic matrix to transform to world coordinates
+ # extrinsic_inv is 4x4 (note closed_form_inverse_OpenCV is batched, the output is (N, 4, 4))
+ cam_to_world_extrinsic = closed_form_inverse_se3(extrinsic[None])[0]
+
+ R_cam_to_world = cam_to_world_extrinsic[:3, :3]
+ t_cam_to_world = cam_to_world_extrinsic[:3, 3]
+
+ # Apply the rotation and translation to the camera coordinates
+ world_coords_points = np.dot(cam_coords_points, R_cam_to_world.T) + t_cam_to_world # HxWx3, 3x3 -> HxWx3
+ # world_coords_points = np.einsum("ij,hwj->hwi", R_cam_to_world, cam_coords_points) + t_cam_to_world
+
+ return world_coords_points, cam_coords_points, point_mask
+
+
+def depth_to_cam_coords_points(depth_map: np.ndarray, intrinsic: np.ndarray) -> tuple[np.ndarray, np.ndarray]:
+ """
+ Convert a depth map to camera coordinates.
+
+ Args:
+ depth_map (np.ndarray): Depth map of shape (H, W).
+ intrinsic (np.ndarray): Camera intrinsic matrix of shape (3, 3).
+
+ Returns:
+ tuple[np.ndarray, np.ndarray]: Camera coordinates (H, W, 3)
+ """
+ H, W = depth_map.shape
+ assert intrinsic.shape == (3, 3), "Intrinsic matrix must be 3x3"
+ assert intrinsic[0, 1] == 0 and intrinsic[1, 0] == 0, "Intrinsic matrix must have zero skew"
+
+ # Intrinsic parameters
+ fu, fv = intrinsic[0, 0], intrinsic[1, 1]
+ cu, cv = intrinsic[0, 2], intrinsic[1, 2]
+
+ # Generate grid of pixel coordinates
+ u, v = np.meshgrid(np.arange(W), np.arange(H))
+
+ # Unproject to camera coordinates
+ x_cam = (u - cu) * depth_map / fu
+ y_cam = (v - cv) * depth_map / fv
+ z_cam = depth_map
+
+ # Stack to form camera coordinates
+ cam_coords = np.stack((x_cam, y_cam, z_cam), axis=-1).astype(np.float32)
+
+ return cam_coords
+
+
+def closed_form_inverse_se3(se3, R=None, T=None):
+ """
+ Compute the inverse of each 4x4 (or 3x4) SE3 matrix in a batch.
+
+ If `R` and `T` are provided, they must correspond to the rotation and translation
+ components of `se3`. Otherwise, they will be extracted from `se3`.
+
+ Args:
+ se3: Nx4x4 or Nx3x4 array or tensor of SE3 matrices.
+ R (optional): Nx3x3 array or tensor of rotation matrices.
+ T (optional): Nx3x1 array or tensor of translation vectors.
+
+ Returns:
+ Inverted SE3 matrices with the same type and device as `se3`.
+
+ Shapes:
+ se3: (N, 4, 4)
+ R: (N, 3, 3)
+ T: (N, 3, 1)
+ """
+ # Check if se3 is a numpy array or a torch tensor
+ is_numpy = isinstance(se3, np.ndarray)
+
+ # Validate shapes
+ if se3.shape[-2:] != (4, 4) and se3.shape[-2:] != (3, 4):
+ raise ValueError(f"se3 must be of shape (N,4,4), got {se3.shape}.")
+
+ # Extract R and T if not provided
+ if R is None:
+ R = se3[:, :3, :3] # (N,3,3)
+ if T is None:
+ T = se3[:, :3, 3:] # (N,3,1)
+
+ # Transpose R
+ if is_numpy:
+ # Compute the transpose of the rotation for NumPy
+ R_transposed = np.transpose(R, (0, 2, 1))
+ # -R^T t for NumPy
+ top_right = -np.matmul(R_transposed, T)
+ inverted_matrix = np.tile(np.eye(4), (len(R), 1, 1))
+ else:
+ R_transposed = R.permute(0, 2, 1) # (N,3,3)
+ top_right = -torch.bmm(R_transposed, T) # (N,3,1)
+ inverted_matrix = torch.eye(4, 4)[None].repeat(len(R), 1, 1)
+ inverted_matrix = inverted_matrix.to(R.dtype).to(R.device)
+
+ inverted_matrix[:, :3, :3] = R_transposed
+ inverted_matrix[:, :3, 3:] = top_right
+
+ return inverted_matrix
+
+def depth_to_cam_coords_points_tensor(depth_map: torch.Tensor, intrinsic: torch.Tensor) -> torch.Tensor:
+ """
+ Convert a depth map to camera coordinates.
+
+ Args:
+ depth_map (torch.Tensor): Depth map of shape (B, H, W).
+ intrinsic (torch.Tensor): Camera intrinsic matrix of shape (B, 3, 3).
+
+ Returns:
+ torch.Tensor: Camera coordinates (B, H, W, 3)
+ """
+ B, H, W = depth_map.shape
+
+ # Intrinsic parameters
+ fu, fv = intrinsic[:, 0, 0], intrinsic[:, 1, 1]
+ cu, cv = intrinsic[:, 0, 2], intrinsic[:, 1, 2]
+
+ # Generate grid of pixel coordinates
+ v, u = torch.meshgrid(torch.arange(W).to(depth_map.device), torch.arange(H).to(depth_map.device))
+
+ # Unproject to camera coordinates
+ x_cam = (u[None] - cu[:, None, None]) * depth_map / fu[:, None, None]
+ y_cam = (v[None] - cv[:, None, None]) * depth_map / fv[:, None, None]
+ z_cam = depth_map
+
+ # Stack to form camera coordinates
+ cam_coords = torch.stack((x_cam, y_cam, z_cam), dim=-1).float()
+
+ return cam_coords
+
+def depth_to_world_coords_points_tensor(
+ depth_map: torch.Tensor,
+ extrinsic: torch.Tensor,
+ intrinsic: torch.Tensor,
+ eps=1e-8,
+) -> torch.Tensor:
+ """
+ Convert a depth map to world coordinates.
+
+ Args:
+ depth_map (torch.Tensor): Depth map of shape (B, H, W, 1).
+ intrinsic (torch.Tensor): Camera intrinsic matrix of shape (B, 3, 3).
+ extrinsic (torch.Tensor): Camera extrinsic matrix of shape (B, 3, 4). OpenCV camera coordinate convention, cam from world.
+
+ Returns:
+ torch.Tensor: World coordinates (B, H, W, 3).
+ """
+ if depth_map is None:
+ return None
+
+ # Valid depth mask
+ point_mask = depth_map > eps
+
+ # Convert depth map to camera coordinates
+ cam_coords_points = depth_to_cam_coords_points_tensor(depth_map, intrinsic)
+
+ # Multiply with the inverse of extrinsic matrix to transform to world coordinates
+ # extrinsic_inv is 4x4 (note closed_form_inverse_OpenCV is batched, the output is (N, 4, 4))
+ cam_to_world_extrinsic = closed_form_inverse_se3(extrinsic)
+
+ R_cam_to_world = cam_to_world_extrinsic[:, :3, :3]
+ t_cam_to_world = cam_to_world_extrinsic[:, :3, 3]
+
+ B, H, W, _ = cam_coords_points.shape
+ # Apply the rotation and translation to the camera coordinates
+ world_coords_points = torch.matmul(cam_coords_points.reshape(B, -1, 3), R_cam_to_world.float().permute(0,2,1)) + t_cam_to_world[:,None] # BxHWx3, Bx3x3 -> HxWx3
+ # world_coords_points = np.einsum("ij,hwj->hwi", R_cam_to_world, cam_coords_points) + t_cam_to_world
+
+ return world_coords_points.reshape(B, H, W, 3)
\ No newline at end of file
diff --git a/vggt/vggt/utils/load_fn.py b/vggt/vggt/utils/load_fn.py
new file mode 100644
index 0000000..1fa765a
--- /dev/null
+++ b/vggt/vggt/utils/load_fn.py
@@ -0,0 +1,118 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch
+from PIL import Image
+from torchvision import transforms as TF
+
+
+def load_and_preprocess_images(image_path_list):
+ """
+ A quick start function to load and preprocess images for model input.
+ This assumes the images should have the same shape for easier batching, but our model can also work well with different shapes.
+
+ Args:
+ image_path_list (list): List of paths to image files
+
+ Returns:
+ torch.Tensor: Batched tensor of preprocessed images with shape (N, 3, H, W)
+
+ Raises:
+ ValueError: If the input list is empty
+
+ Notes:
+ - Images with different dimensions will be padded with white (value=1.0)
+ - A warning is printed when images have different shapes
+ - The function ensures width=518px while maintaining aspect ratio
+ - Height is adjusted to be divisible by 14 for compatibility with model requirements
+ """
+ # Check for empty list
+ if len(image_path_list) == 0:
+ raise ValueError("At least 1 image is required")
+
+ images = []
+ alphas = []
+ shapes = set()
+ to_tensor = TF.ToTensor()
+
+ # First process all images and collect their shapes
+ # for image_path in image_path_list:
+ for img in image_path_list:
+
+ # Open image
+ # img = Image.open(image_path)
+ img = img[0]
+
+ # If there's an alpha channel, blend onto white background:
+ if img.mode == "RGBA":
+ # Create white background
+ alphas.append(to_tensor(img)[3:])
+ # background = Image.new("RGBA", img.size, (255, 255, 255, 255))
+ # Alpha composite onto the white background
+ # img = Image.alpha_composite(background, img)
+
+
+ # Now convert to "RGB" (this step assigns white for transparent areas)
+ img = img.convert("RGB")
+
+ width, height = img.size
+ new_width = 518
+
+ # Calculate height maintaining aspect ratio, divisible by 14
+ new_height = round(height * (new_width / width) / 14) * 14
+
+ # Resize with new dimensions (width, height)
+
+ img = img.resize((new_width, new_height), Image.Resampling.BICUBIC)
+ img = to_tensor(img) # Convert to tensor (0, 1)
+
+ # Center crop height if it's larger than 518
+
+ if new_height > 518:
+ start_y = (new_height - 518) // 2
+ img = img[:, start_y : start_y + 518, :]
+
+ shapes.add((img.shape[1], img.shape[2]))
+ images.append(img)
+
+ # Check if we have different shapes
+ # In theory our model can also work well with different shapes
+
+ if len(shapes) > 1:
+ print(f"Warning: Found images with different shapes: {shapes}")
+ # Find maximum dimensions
+ max_height = max(shape[0] for shape in shapes)
+ max_width = max(shape[1] for shape in shapes)
+
+ # Pad images if necessary
+ padded_images = []
+ for img in images:
+ h_padding = max_height - img.shape[1]
+ w_padding = max_width - img.shape[2]
+
+ if h_padding > 0 or w_padding > 0:
+ pad_top = h_padding // 2
+ pad_bottom = h_padding - pad_top
+ pad_left = w_padding // 2
+ pad_right = w_padding - pad_left
+
+ img = torch.nn.functional.pad(
+ img, (pad_left, pad_right, pad_top, pad_bottom), mode="constant", value=1.0
+ )
+ padded_images.append(img)
+ images = padded_images
+
+ images = torch.stack(images) # concatenate images
+ alphas = torch.stack(alphas) # concatenate images
+
+ # Ensure correct shape when single image
+ if len(image_path_list) == 1:
+ # Verify shape is (1, C, H, W)
+ if images.dim() == 3:
+ images = images.unsqueeze(0)
+ alphas = alphas.unsqueeze(0)
+
+ return images, alphas
diff --git a/vggt/vggt/utils/pose_enc.py b/vggt/vggt/utils/pose_enc.py
new file mode 100644
index 0000000..2f98b08
--- /dev/null
+++ b/vggt/vggt/utils/pose_enc.py
@@ -0,0 +1,130 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import torch
+from .rotation import quat_to_mat, mat_to_quat
+
+
+def extri_intri_to_pose_encoding(
+ extrinsics,
+ intrinsics,
+ image_size_hw=None, # e.g., (256, 512)
+ pose_encoding_type="absT_quaR_FoV",
+):
+ """Convert camera extrinsics and intrinsics to a compact pose encoding.
+
+ This function transforms camera parameters into a unified pose encoding format,
+ which can be used for various downstream tasks like pose prediction or representation.
+
+ Args:
+ extrinsics (torch.Tensor): Camera extrinsic parameters with shape BxSx3x4,
+ where B is batch size and S is sequence length.
+ In OpenCV coordinate system (x-right, y-down, z-forward), representing camera from world transformation.
+ The format is [R|t] where R is a 3x3 rotation matrix and t is a 3x1 translation vector.
+ intrinsics (torch.Tensor): Camera intrinsic parameters with shape BxSx3x3.
+ Defined in pixels, with format:
+ [[fx, 0, cx],
+ [0, fy, cy],
+ [0, 0, 1]]
+ where fx, fy are focal lengths and (cx, cy) is the principal point
+ image_size_hw (tuple): Tuple of (height, width) of the image in pixels.
+ Required for computing field of view values. For example: (256, 512).
+ pose_encoding_type (str): Type of pose encoding to use. Currently only
+ supports "absT_quaR_FoV" (absolute translation, quaternion rotation, field of view).
+
+ Returns:
+ torch.Tensor: Encoded camera pose parameters with shape BxSx9.
+ For "absT_quaR_FoV" type, the 9 dimensions are:
+ - [:3] = absolute translation vector T (3D)
+ - [3:7] = rotation as quaternion quat (4D)
+ - [7:] = field of view (2D)
+ """
+
+ # extrinsics: BxSx3x4
+ # intrinsics: BxSx3x3
+
+ if pose_encoding_type == "absT_quaR_FoV":
+ R = extrinsics[:, :, :3, :3] # BxSx3x3
+ T = extrinsics[:, :, :3, 3] # BxSx3
+
+ quat = mat_to_quat(R)
+ # Note the order of h and w here
+ H, W = image_size_hw
+ fov_h = 2 * torch.atan((H / 2) / intrinsics[..., 1, 1])
+ fov_w = 2 * torch.atan((W / 2) / intrinsics[..., 0, 0])
+ pose_encoding = torch.cat([T, quat, fov_h[..., None], fov_w[..., None]], dim=-1).float()
+ else:
+ raise NotImplementedError
+
+ return pose_encoding
+
+
+def pose_encoding_to_extri_intri(
+ pose_encoding,
+ image_size_hw=None, # e.g., (256, 512)
+ pose_encoding_type="absT_quaR_FoV",
+ build_intrinsics=True,
+):
+ """Convert a pose encoding back to camera extrinsics and intrinsics.
+
+ This function performs the inverse operation of extri_intri_to_pose_encoding,
+ reconstructing the full camera parameters from the compact encoding.
+
+ Args:
+ pose_encoding (torch.Tensor): Encoded camera pose parameters with shape BxSx9,
+ where B is batch size and S is sequence length.
+ For "absT_quaR_FoV" type, the 9 dimensions are:
+ - [:3] = absolute translation vector T (3D)
+ - [3:7] = rotation as quaternion quat (4D)
+ - [7:] = field of view (2D)
+ image_size_hw (tuple): Tuple of (height, width) of the image in pixels.
+ Required for reconstructing intrinsics from field of view values.
+ For example: (256, 512).
+ pose_encoding_type (str): Type of pose encoding used. Currently only
+ supports "absT_quaR_FoV" (absolute translation, quaternion rotation, field of view).
+ build_intrinsics (bool): Whether to reconstruct the intrinsics matrix.
+ If False, only extrinsics are returned and intrinsics will be None.
+
+ Returns:
+ tuple: (extrinsics, intrinsics)
+ - extrinsics (torch.Tensor): Camera extrinsic parameters with shape BxSx3x4.
+ In OpenCV coordinate system (x-right, y-down, z-forward), representing camera from world
+ transformation. The format is [R|t] where R is a 3x3 rotation matrix and t is
+ a 3x1 translation vector.
+ - intrinsics (torch.Tensor or None): Camera intrinsic parameters with shape BxSx3x3,
+ or None if build_intrinsics is False. Defined in pixels, with format:
+ [[fx, 0, cx],
+ [0, fy, cy],
+ [0, 0, 1]]
+ where fx, fy are focal lengths and (cx, cy) is the principal point,
+ assumed to be at the center of the image (W/2, H/2).
+ """
+
+ intrinsics = None
+
+ if pose_encoding_type == "absT_quaR_FoV":
+ T = pose_encoding[..., :3]
+ quat = pose_encoding[..., 3:7]
+ fov_h = pose_encoding[..., 7]
+ fov_w = pose_encoding[..., 8]
+
+ R = quat_to_mat(quat)
+ extrinsics = torch.cat([R, T[..., None]], dim=-1)
+
+ if build_intrinsics:
+ H, W = image_size_hw
+ fy = (H / 2.0) / torch.tan(fov_h / 2.0)
+ fx = (W / 2.0) / torch.tan(fov_w / 2.0)
+ intrinsics = torch.zeros(pose_encoding.shape[:2] + (3, 3), device=pose_encoding.device)
+ intrinsics[..., 0, 0] = fx
+ intrinsics[..., 1, 1] = fy
+ intrinsics[..., 0, 2] = W / 2
+ intrinsics[..., 1, 2] = H / 2
+ intrinsics[..., 2, 2] = 1.0 # Set the homogeneous coordinate to 1
+ else:
+ raise NotImplementedError
+
+ return extrinsics, intrinsics
diff --git a/vggt/vggt/utils/rotation.py b/vggt/vggt/utils/rotation.py
new file mode 100644
index 0000000..657583e
--- /dev/null
+++ b/vggt/vggt/utils/rotation.py
@@ -0,0 +1,138 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+# Modified from PyTorch3D, https://github.com/facebookresearch/pytorch3d
+
+import torch
+import numpy as np
+import torch.nn.functional as F
+
+
+def quat_to_mat(quaternions: torch.Tensor) -> torch.Tensor:
+ """
+ Quaternion Order: XYZW or say ijkr, scalar-last
+
+ Convert rotations given as quaternions to rotation matrices.
+ Args:
+ quaternions: quaternions with real part last,
+ as tensor of shape (..., 4).
+
+ Returns:
+ Rotation matrices as tensor of shape (..., 3, 3).
+ """
+ i, j, k, r = torch.unbind(quaternions, -1)
+ # pyre-fixme[58]: `/` is not supported for operand types `float` and `Tensor`.
+ two_s = 2.0 / (quaternions * quaternions).sum(-1)
+
+ o = torch.stack(
+ (
+ 1 - two_s * (j * j + k * k),
+ two_s * (i * j - k * r),
+ two_s * (i * k + j * r),
+ two_s * (i * j + k * r),
+ 1 - two_s * (i * i + k * k),
+ two_s * (j * k - i * r),
+ two_s * (i * k - j * r),
+ two_s * (j * k + i * r),
+ 1 - two_s * (i * i + j * j),
+ ),
+ -1,
+ )
+ return o.reshape(quaternions.shape[:-1] + (3, 3))
+
+
+def mat_to_quat(matrix: torch.Tensor) -> torch.Tensor:
+ """
+ Convert rotations given as rotation matrices to quaternions.
+
+ Args:
+ matrix: Rotation matrices as tensor of shape (..., 3, 3).
+
+ Returns:
+ quaternions with real part last, as tensor of shape (..., 4).
+ Quaternion Order: XYZW or say ijkr, scalar-last
+ """
+ if matrix.size(-1) != 3 or matrix.size(-2) != 3:
+ raise ValueError(f"Invalid rotation matrix shape {matrix.shape}.")
+
+ batch_dim = matrix.shape[:-2]
+ m00, m01, m02, m10, m11, m12, m20, m21, m22 = torch.unbind(matrix.reshape(batch_dim + (9,)), dim=-1)
+
+ q_abs = _sqrt_positive_part(
+ torch.stack(
+ [
+ 1.0 + m00 + m11 + m22,
+ 1.0 + m00 - m11 - m22,
+ 1.0 - m00 + m11 - m22,
+ 1.0 - m00 - m11 + m22,
+ ],
+ dim=-1,
+ )
+ )
+
+ # we produce the desired quaternion multiplied by each of r, i, j, k
+ quat_by_rijk = torch.stack(
+ [
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
+ # `int`.
+ torch.stack([q_abs[..., 0] ** 2, m21 - m12, m02 - m20, m10 - m01], dim=-1),
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
+ # `int`.
+ torch.stack([m21 - m12, q_abs[..., 1] ** 2, m10 + m01, m02 + m20], dim=-1),
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
+ # `int`.
+ torch.stack([m02 - m20, m10 + m01, q_abs[..., 2] ** 2, m12 + m21], dim=-1),
+ # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and
+ # `int`.
+ torch.stack([m10 - m01, m20 + m02, m21 + m12, q_abs[..., 3] ** 2], dim=-1),
+ ],
+ dim=-2,
+ )
+
+ # We floor here at 0.1 but the exact level is not important; if q_abs is small,
+ # the candidate won't be picked.
+ flr = torch.tensor(0.1).to(dtype=q_abs.dtype, device=q_abs.device)
+ quat_candidates = quat_by_rijk / (2.0 * q_abs[..., None].max(flr))
+
+ # if not for numerical problems, quat_candidates[i] should be same (up to a sign),
+ # forall i; we pick the best-conditioned one (with the largest denominator)
+ out = quat_candidates[F.one_hot(q_abs.argmax(dim=-1), num_classes=4) > 0.5, :].reshape(batch_dim + (4,))
+
+ # Convert from rijk to ijkr
+ out = out[..., [1, 2, 3, 0]]
+
+ out = standardize_quaternion(out)
+
+ return out
+
+
+def _sqrt_positive_part(x: torch.Tensor) -> torch.Tensor:
+ """
+ Returns torch.sqrt(torch.max(0, x))
+ but with a zero subgradient where x is 0.
+ """
+ ret = torch.zeros_like(x)
+ positive_mask = x > 0
+ if torch.is_grad_enabled():
+ ret[positive_mask] = torch.sqrt(x[positive_mask])
+ else:
+ ret = torch.where(positive_mask, torch.sqrt(x), ret)
+ return ret
+
+
+def standardize_quaternion(quaternions: torch.Tensor) -> torch.Tensor:
+ """
+ Convert a unit quaternion to a standard form: one in which the real
+ part is non negative.
+
+ Args:
+ quaternions: Quaternions with real part last,
+ as tensor of shape (..., 4).
+
+ Returns:
+ Standardized quaternions as tensor of shape (..., 4).
+ """
+ return torch.where(quaternions[..., 3:4] < 0, -quaternions, quaternions)
diff --git a/vggt/vggt/utils/visual_track.py b/vggt/vggt/utils/visual_track.py
new file mode 100644
index 0000000..796c114
--- /dev/null
+++ b/vggt/vggt/utils/visual_track.py
@@ -0,0 +1,239 @@
+# Copyright (c) Meta Platforms, Inc. and affiliates.
+# All rights reserved.
+#
+# This source code is licensed under the license found in the
+# LICENSE file in the root directory of this source tree.
+
+import cv2
+import torch
+import numpy as np
+import os
+
+
+def color_from_xy(x, y, W, H, cmap_name="hsv"):
+ """
+ Map (x, y) -> color in (R, G, B).
+ 1) Normalize x,y to [0,1].
+ 2) Combine them into a single scalar c in [0,1].
+ 3) Use matplotlib's colormap to convert c -> (R,G,B).
+
+ You can customize step 2, e.g., c = (x + y)/2, or some function of (x, y).
+ """
+ import matplotlib.cm
+ import matplotlib.colors
+
+ x_norm = x / max(W - 1, 1)
+ y_norm = y / max(H - 1, 1)
+ # Simple combination:
+ c = (x_norm + y_norm) / 2.0
+
+ cmap = matplotlib.cm.get_cmap(cmap_name)
+ # cmap(c) -> (r,g,b,a) in [0,1]
+ rgba = cmap(c)
+ r, g, b = rgba[0], rgba[1], rgba[2]
+ return (r, g, b) # in [0,1], RGB order
+
+
+def get_track_colors_by_position(tracks_b, vis_mask_b=None, image_width=None, image_height=None, cmap_name="hsv"):
+ """
+ Given all tracks in one sample (b), compute a (N,3) array of RGB color values
+ in [0,255]. The color is determined by the (x,y) position in the first
+ visible frame for each track.
+
+ Args:
+ tracks_b: Tensor of shape (S, N, 2). (x,y) for each track in each frame.
+ vis_mask_b: (S, N) boolean mask; if None, assume all are visible.
+ image_width, image_height: used for normalizing (x, y).
+ cmap_name: for matplotlib (e.g., 'hsv', 'rainbow', 'jet').
+
+ Returns:
+ track_colors: np.ndarray of shape (N, 3), each row is (R,G,B) in [0,255].
+ """
+ S, N, _ = tracks_b.shape
+ track_colors = np.zeros((N, 3), dtype=np.uint8)
+
+ if vis_mask_b is None:
+ # treat all as visible
+ vis_mask_b = torch.ones(S, N, dtype=torch.bool, device=tracks_b.device)
+
+ for i in range(N):
+ # Find first visible frame for track i
+ visible_frames = torch.where(vis_mask_b[:, i])[0]
+ if len(visible_frames) == 0:
+ # track is never visible; just assign black or something
+ track_colors[i] = (0, 0, 0)
+ continue
+
+ first_s = int(visible_frames[0].item())
+ # use that frame's (x,y)
+ x, y = tracks_b[first_s, i].tolist()
+
+ # map (x,y) -> (R,G,B) in [0,1]
+ r, g, b = color_from_xy(x, y, W=image_width, H=image_height, cmap_name=cmap_name)
+ # scale to [0,255]
+ r, g, b = int(r * 255), int(g * 255), int(b * 255)
+ track_colors[i] = (r, g, b)
+
+ return track_colors
+
+
+def visualize_tracks_on_images(
+ images,
+ tracks,
+ track_vis_mask=None,
+ out_dir="track_visuals_concat_by_xy",
+ image_format="CHW", # "CHW" or "HWC"
+ normalize_mode="[0,1]",
+ cmap_name="hsv", # e.g. "hsv", "rainbow", "jet"
+ frames_per_row=4, # New parameter for grid layout
+ save_grid=True, # Flag to control whether to save the grid image
+):
+ """
+ Visualizes frames in a grid layout with specified frames per row.
+ Each track's color is determined by its (x,y) position
+ in the first visible frame (or frame 0 if always visible).
+ Finally convert the BGR result to RGB before saving.
+ Also saves each individual frame as a separate PNG file.
+
+ Args:
+ images: torch.Tensor (S, 3, H, W) if CHW or (S, H, W, 3) if HWC.
+ tracks: torch.Tensor (S, N, 2), last dim = (x, y).
+ track_vis_mask: torch.Tensor (S, N) or None.
+ out_dir: folder to save visualizations.
+ image_format: "CHW" or "HWC".
+ normalize_mode: "[0,1]", "[-1,1]", or None for direct raw -> 0..255
+ cmap_name: a matplotlib colormap name for color_from_xy.
+ frames_per_row: number of frames to display in each row of the grid.
+ save_grid: whether to save all frames in one grid image.
+
+ Returns:
+ None (saves images in out_dir).
+ """
+
+ if len(tracks.shape) == 4:
+ tracks = tracks.squeeze(0)
+ images = images.squeeze(0)
+ if track_vis_mask is not None:
+ track_vis_mask = track_vis_mask.squeeze(0)
+
+ import matplotlib
+
+ matplotlib.use("Agg") # for non-interactive (optional)
+
+ os.makedirs(out_dir, exist_ok=True)
+
+ S = images.shape[0]
+ _, N, _ = tracks.shape # (S, N, 2)
+
+ # Move to CPU
+ images = images.cpu().clone()
+ tracks = tracks.cpu().clone()
+ if track_vis_mask is not None:
+ track_vis_mask = track_vis_mask.cpu().clone()
+
+ # Infer H, W from images shape
+ if image_format == "CHW":
+ # e.g. images[s].shape = (3, H, W)
+ H, W = images.shape[2], images.shape[3]
+ else:
+ # e.g. images[s].shape = (H, W, 3)
+ H, W = images.shape[1], images.shape[2]
+
+ # Pre-compute the color for each track i based on first visible position
+ track_colors_rgb = get_track_colors_by_position(
+ tracks, # shape (S, N, 2)
+ vis_mask_b=track_vis_mask if track_vis_mask is not None else None,
+ image_width=W,
+ image_height=H,
+ cmap_name=cmap_name,
+ )
+
+ # We'll accumulate each frame's drawn image in a list
+ frame_images = []
+
+ for s in range(S):
+ # shape => either (3, H, W) or (H, W, 3)
+ img = images[s]
+
+ # Convert to (H, W, 3)
+ if image_format == "CHW":
+ img = img.permute(1, 2, 0) # (H, W, 3)
+ # else "HWC", do nothing
+
+ img = img.numpy().astype(np.float32)
+
+ # Scale to [0,255] if needed
+ if normalize_mode == "[0,1]":
+ img = np.clip(img, 0, 1) * 255.0
+ elif normalize_mode == "[-1,1]":
+ img = (img + 1.0) * 0.5 * 255.0
+ img = np.clip(img, 0, 255.0)
+ # else no normalization
+
+ # Convert to uint8
+ img = img.astype(np.uint8)
+
+ # For drawing in OpenCV, convert to BGR
+ img_bgr = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)
+
+ # Draw each visible track
+ cur_tracks = tracks[s] # shape (N, 2)
+ if track_vis_mask is not None:
+ valid_indices = torch.where(track_vis_mask[s])[0]
+ else:
+ valid_indices = range(N)
+
+ cur_tracks_np = cur_tracks.numpy()
+ for i in valid_indices:
+ x, y = cur_tracks_np[i]
+ pt = (int(round(x)), int(round(y)))
+
+ # track_colors_rgb[i] is (R,G,B). For OpenCV circle, we need BGR
+ R, G, B = track_colors_rgb[i]
+ color_bgr = (int(B), int(G), int(R))
+ cv2.circle(img_bgr, pt, radius=3, color=color_bgr, thickness=-1)
+
+ # Convert back to RGB for consistent final saving:
+ img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB)
+
+ # Save individual frame
+ frame_path = os.path.join(out_dir, f"frame_{s:04d}.png")
+ # Convert to BGR for OpenCV imwrite
+ frame_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(frame_path, frame_bgr)
+
+ frame_images.append(img_rgb)
+
+ # Only create and save the grid image if save_grid is True
+ if save_grid:
+ # Calculate grid dimensions
+ num_rows = (S + frames_per_row - 1) // frames_per_row # Ceiling division
+
+ # Create a grid of images
+ grid_img = None
+ for row in range(num_rows):
+ start_idx = row * frames_per_row
+ end_idx = min(start_idx + frames_per_row, S)
+
+ # Concatenate this row horizontally
+ row_img = np.concatenate(frame_images[start_idx:end_idx], axis=1)
+
+ # If this row has fewer than frames_per_row images, pad with black
+ if end_idx - start_idx < frames_per_row:
+ padding_width = (frames_per_row - (end_idx - start_idx)) * W
+ padding = np.zeros((H, padding_width, 3), dtype=np.uint8)
+ row_img = np.concatenate([row_img, padding], axis=1)
+
+ # Add this row to the grid
+ if grid_img is None:
+ grid_img = row_img
+ else:
+ grid_img = np.concatenate([grid_img, row_img], axis=0)
+
+ out_path = os.path.join(out_dir, "tracks_grid.png")
+ # Convert back to BGR for OpenCV imwrite
+ grid_img_bgr = cv2.cvtColor(grid_img, cv2.COLOR_RGB2BGR)
+ cv2.imwrite(out_path, grid_img_bgr)
+ print(f"[INFO] Saved color-by-XY track visualization grid -> {out_path}")
+
+ print(f"[INFO] Saved {S} individual frames to {out_dir}/frame_*.png")