From f791215b2a608f2fae67722702b06f871af9e5c4 Mon Sep 17 00:00:00 2001 From: Bruno Fargnoli Date: Sat, 4 Apr 2026 21:29:20 +0200 Subject: [PATCH] Added ReconViaGen code with new node "Sparse Generator with ReconViaGen" --- README.md | 1 + example_workflows/Advanced.json | 799 +- example_workflows/Advanced_CustomSteps.json | 1249 +++ .../Advanced_CustomSteps_MeshOnly.json | 1118 +++ example_workflows/MultiViews.json | 1327 ++- example_workflows/MultiViews_MeshOnly.json | 1287 ++- example_workflows/Projection_6Views_Hy20.json | 8578 +++++++++++++++++ example_workflows/ReconViaGen_MeshOnly.json | 1019 ++ example_workflows/RefineMesh.json | 810 ++ example_workflows/RefineMesh_MeshOnly.json | 776 ++ example_workflows/Simple.json | 677 +- nodes.py | 462 +- pyproject.toml | 2 +- reconviagen_pipeline.json | 98 + trellis2/models/__init__.py | 5 +- trellis2/models/sparse_structure_flow.py | 77 +- trellis2/modules/transformer/modulated.py | 72 +- trellis2/pipelines/trellis2_image_to_3d.py | 48 +- vggt/vggt/heads/camera_head.py | 162 + vggt/vggt/heads/dpt_head.py | 497 + vggt/vggt/heads/head_act.py | 125 + vggt/vggt/heads/track_head.py | 108 + vggt/vggt/heads/track_modules/__init__.py | 5 + .../track_modules/base_track_predictor.py | 209 + vggt/vggt/heads/track_modules/blocks.py | 246 + vggt/vggt/heads/track_modules/modules.py | 218 + vggt/vggt/heads/track_modules/utils.py | 226 + vggt/vggt/heads/utils.py | 108 + vggt/vggt/layers/__init__.py | 11 + vggt/vggt/layers/attention.py | 98 + vggt/vggt/layers/block.py | 259 + vggt/vggt/layers/drop_path.py | 34 + vggt/vggt/layers/layer_scale.py | 27 + vggt/vggt/layers/mlp.py | 40 + vggt/vggt/layers/patch_embed.py | 88 + vggt/vggt/layers/rope.py | 188 + vggt/vggt/layers/swiglu_ffn.py | 72 + vggt/vggt/layers/vision_transformer.py | 407 + vggt/vggt/models/aggregator.py | 331 + vggt/vggt/models/vggt.py | 95 + vggt/vggt/utils/geometry.py | 236 + vggt/vggt/utils/load_fn.py | 118 + vggt/vggt/utils/pose_enc.py | 130 + vggt/vggt/utils/rotation.py | 138 + vggt/vggt/utils/visual_track.py | 239 + 45 files changed, 21517 insertions(+), 1303 deletions(-) create mode 100644 example_workflows/Advanced_CustomSteps.json create mode 100644 example_workflows/Advanced_CustomSteps_MeshOnly.json create mode 100644 example_workflows/Projection_6Views_Hy20.json create mode 100644 example_workflows/ReconViaGen_MeshOnly.json create mode 100644 example_workflows/RefineMesh.json create mode 100644 example_workflows/RefineMesh_MeshOnly.json create mode 100644 reconviagen_pipeline.json create mode 100644 vggt/vggt/heads/camera_head.py create mode 100644 vggt/vggt/heads/dpt_head.py create mode 100644 vggt/vggt/heads/head_act.py create mode 100644 vggt/vggt/heads/track_head.py create mode 100644 vggt/vggt/heads/track_modules/__init__.py create mode 100644 vggt/vggt/heads/track_modules/base_track_predictor.py create mode 100644 vggt/vggt/heads/track_modules/blocks.py create mode 100644 vggt/vggt/heads/track_modules/modules.py create mode 100644 vggt/vggt/heads/track_modules/utils.py create mode 100644 vggt/vggt/heads/utils.py create mode 100644 vggt/vggt/layers/__init__.py create mode 100644 vggt/vggt/layers/attention.py create mode 100644 vggt/vggt/layers/block.py create mode 100644 vggt/vggt/layers/drop_path.py create mode 100644 vggt/vggt/layers/layer_scale.py create mode 100644 vggt/vggt/layers/mlp.py create mode 100644 vggt/vggt/layers/patch_embed.py create mode 100644 vggt/vggt/layers/rope.py create mode 100644 vggt/vggt/layers/swiglu_ffn.py create mode 100644 vggt/vggt/layers/vision_transformer.py create mode 100644 vggt/vggt/models/aggregator.py create mode 100644 vggt/vggt/models/vggt.py create mode 100644 vggt/vggt/utils/geometry.py create mode 100644 vggt/vggt/utils/load_fn.py create mode 100644 vggt/vggt/utils/pose_enc.py create mode 100644 vggt/vggt/utils/rotation.py create mode 100644 vggt/vggt/utils/visual_track.py diff --git a/README.md b/README.md index fc77c1e..c355fbc 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,7 @@ | Date | Description | | --- | --- | +| **2026-04-04** | Added node "Sparse Generator with ReconViaGen" | | **2026-04-01** | Added node "Voxel to Mesh"
It replaces Remeshing to make watertight mesh | | **2026-03-21** | Added node "Projection HighPoly to LowPoly"
Added node "Render MultiView" | | **2026-03-17** | Added Inpainting Choice NS and TELEA | diff --git a/example_workflows/Advanced.json b/example_workflows/Advanced.json index 20e4ce3..1f51ae1 100644 --- a/example_workflows/Advanced.json +++ b/example_workflows/Advanced.json @@ -1,22 +1,136 @@ { "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", "revision": 0, - "last_node_id": 55, - "last_link_id": 117, + "last_node_id": 207, + "last_link_id": 416, "nodes": [ { - "id": 10, - "type": "Preview3D", + "id": 69, + "type": "Trellis2LoadImageWithTransparency", "pos": [ - 1739.3745174243877, - 813.1451251994486 + -377.72410571388025, + 328.4870830026766 ], "size": [ - 1216.03125, - 1078.3125 + 717.521556382955, + 915.5313594325766 ], "flags": {}, - "order": 6, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -373.0898096413628, + 163.12227474974895 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 409 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -368.2858630795249, + 34.401714106564214 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 410 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 376.48890096546415, + 331.2515281089069 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 12, "mode": 0, "inputs": [ { @@ -33,111 +147,37 @@ }, { "name": "model_file", - "type": "STRING", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", "widget": { "name": "model_file" }, - "link": 22 + "link": 398 } ], "outputs": [], "properties": { "cnr_id": "comfy-core", - "ver": "0.4.0", - "Node name for S&R": "Preview3D", - "widget_ue_connectable": {}, - "Last Time Model File": "ArmoredWarrior_00002_.glb", - "Scene Config": { - "showGrid": true, - "backgroundColor": "#282828", - "backgroundImage": "", - "backgroundRenderMode": "tiled" - }, - "Camera Config": { - "cameraType": "perspective", - "fov": 35, - "state": { - "position": { - "x": 1.4101977762939673, - "y": 2.397771625867536, - "z": 8.997201229712031 - }, - "target": { - "x": 1.4732208254010746e-177, - "y": 2.5, - "z": 6.912872131937994e-178 - }, - "zoom": 1, - "cameraType": "perspective" - } - }, - "Light Config": { - "intensity": 3 - } + "ver": "0.18.1", + "Node name for S&R": "Preview3D" }, "widgets_values": [ - "ArmoredWarrior_00002_.glb", + "", "" ] }, - { - "id": 6, - "type": "Trellis2LoadImageWithTransparency", - "pos": [ - 138.06663994864374, - 1231.972182577106 - ], - "size": [ - 470.78125, - 439.8125 - ], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": null - }, - { - "name": "mask", - "type": "MASK", - "links": null - }, - { - "name": "image_with_alpha", - "type": "IMAGE", - "links": [ - 116 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", - "Node name for S&R": "Trellis2LoadImageWithTransparency", - "widget_ue_connectable": {} - }, - "widgets_values": [ - "Image_2048_00008_.png", - "image" - ] - }, { "id": 39, "type": "Trellis2LoadModel", "pos": [ - 243.7073730589844, - 824.811803650631 + 404.6117778691903, + -564.6350558283436 ], "size": [ - 362.0625, - 207.328125 + 301.71875, + 202 ], "flags": {}, - "order": 1, + "order": 3, "mode": 0, "inputs": [], "outputs": [ @@ -145,7 +185,7 @@ "name": "pipeline", "type": "TRELLIS2PIPELINE", "links": [ - 87 + 411 ] } ], @@ -156,79 +196,76 @@ "widget_ue_connectable": {} }, "widgets_values": [ - "TRELLIS.2-4B", + "microsoft/TRELLIS.2-4B", "flash_attn", "cuda", true, - false + false, + "flex_gemm", + "flash_attn" ] }, { - "id": 19, - "type": "Trellis2ExportMesh", + "id": 119, + "type": "Trellis2PreProcessImage", "pos": [ - 1583.1015090615297, - 594.4780695351682 + 413.7442818307099, + -185.99180300678682 ], "size": [ - 324, - 146 + 281.8229166666667, + 106 ], "flags": {}, - "order": 5, + "order": 4, "mode": 0, "inputs": [ { - "name": "trimesh", - "type": "TRIMESH", - "link": 115 + "name": "image", + "type": "IMAGE", + "link": 241 } ], "outputs": [ { - "name": "glb_path", - "type": "STRING", + "name": "image", + "type": "IMAGE", "links": [ - 22 + 412 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", - "Node name for S&R": "Trellis2ExportMesh", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", "widget_ue_connectable": {} }, "widgets_values": [ - "Trellis2Mesh", - "glb", - true + 10, + false, + 1024 ] }, { - "id": 45, - "type": "Trellis2MeshWithVoxelAdvancedGenerator", + "id": 196, + "type": "Trellis2FillHolesWithCuMesh", "pos": [ - 687.5583313121791, - 819.8533042235456 + 1148.203832543889, + -691.4362805609004 ], "size": [ - 495.8125, - 866 + 312.4361328125, + 58 ], "flags": {}, - "order": 3, + "order": 6, "mode": 0, "inputs": [ { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", - "link": 87 - }, - { - "name": "image", - "type": "IMAGE", - "link": 117 + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 413 } ], "outputs": [ @@ -236,41 +273,175 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 113 + 392 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 197, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1143.7616665123248, + -576.3871604957823 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 392 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 393 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 198, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1146.6254930330851, + -381.0428638307529 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 393 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 409 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 394 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 205, + "type": "Trellis2MeshWithVoxelAdvancedGenerator", + "pos": [ + 717.5499413662296, + -588.6883997158563 + ], + "size": [ + 413.1841796875, + 702 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 411 + }, + { + "name": "image", + "type": "IMAGE", + "link": 412 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 413 ] }, { "name": "bvh", "type": "BVH", "links": [ - 114 + 415 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a28d6cf93b0cf24aa9e7d3dccfd14f406227fbd0", - "Node name for S&R": "Trellis2MeshWithVoxelAdvancedGenerator", - "widget_ue_connectable": {} + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelAdvancedGenerator" }, "widgets_values": [ 12345, - "randomize", + "fixed", "1024_cascade", + 25, + 7.5, + 0.01, + 5, 12, - 6.5, - 0.2, - 4, - 12, - 6.5, - 0.2, - 4, + 7.5, + 0.01, + 3, 12, 3, - 0.2, + 0.01, 3, 999999, - 4, + 1, 32, true, 0.1, @@ -279,33 +450,129 @@ 1, 0, 0.9, - true + true, + "euler" ] }, { - "id": 54, - "type": "Trellis2PostProcessAndUnWrapAndRasterizer", + "id": 201, + "type": "Trellis2ExportMesh", "pos": [ - 1262.5083613749216, - 819.4992892507565 + 1597.089782460921, + -140.8020848589493 ], "size": [ - 419.125, - 651.328125 + 270, + 102 ], "flags": {}, - "order": 4, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 416 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 410 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 398 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + }, + { + "id": 199, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1151.769446503904, + -245.92060963339878 + ], + "size": [ + 302.4315956250001, + 47.18107670916754 + ], + "flags": {}, + "order": 9, "mode": 0, "inputs": [ { "name": "mesh", "type": "MESHWITHVOXEL", - "link": 113 + "link": 394 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 414 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 207, + "type": "Trellis2UnWrapAndRasterizer", + "pos": [ + 1153.8343873953147, + -150.96148118668327 + ], + "size": [ + 419.15234375, + 314 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 414 }, { "name": "bvh", "type": "BVH", - "link": 114 + "link": 415 } ], "outputs": [ @@ -313,7 +580,7 @@ "name": "trimesh", "type": "TRIMESH", "links": [ - 115 + 416 ] }, { @@ -329,144 +596,170 @@ ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72", - "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer", - "widget_ue_connectable": {} + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2UnWrapAndRasterizer" }, "widgets_values": [ 60, 0, 1, 1, - 4096, - true, - 1, - 0, - 2000000, - "Cumesh", - true, + 2048, "OPAQUE", - "1024", - false, - true, false, false, - true - ] - }, - { - "id": 55, - "type": "Trellis2PreProcessImage", - "pos": [ - 310.038958285416, - 1092.8065963153765 - ], - "size": [ - 281.78125, - 79.328125 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 116 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 117 - ] - } - ], - "properties": { - "widget_ue_connectable": {}, - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72", - "Node name for S&R": "Trellis2PreProcessImage" - }, - "widgets_values": [ - 25 + false, + "telea" ] } ], "links": [ [ - 22, - 19, - 0, - 10, + 241, + 69, 2, - "STRING" + 119, + 0, + "IMAGE" ], [ - 87, - 39, + 392, + 196, 0, - 45, - 0, - "TRELLIS2PIPELINE" - ], - [ - 113, - 45, - 0, - 54, + 197, 0, "MESHWITHVOXEL" ], [ - 114, - 45, + 393, + 197, + 0, + 198, + 0, + "MESHWITHVOXEL" + ], + [ + 394, + 198, + 0, + 199, + 0, + "MESHWITHVOXEL" + ], + [ + 398, + 201, + 0, + 202, + 2, + "STRING" + ], + [ + 409, + 203, + 0, + 198, 1, - 54, + "INT" + ], + [ + 410, + 204, + 0, + 201, + 1, + "STRING" + ], + [ + 411, + 39, + 0, + 205, + 0, + "TRELLIS2PIPELINE" + ], + [ + 412, + 119, + 0, + 205, + 1, + "IMAGE" + ], + [ + 413, + 205, + 0, + 196, + 0, + "MESHWITHVOXEL" + ], + [ + 414, + 199, + 0, + 207, + 0, + "MESHWITHVOXEL" + ], + [ + 415, + 205, + 1, + 207, 1, "BVH" ], [ - 115, - 54, + 416, + 207, 0, - 19, + 201, 0, "TRIMESH" - ], - [ - 116, - 6, - 2, - 55, - 0, - "IMAGE" - ], - [ - 117, - 55, - 0, - 45, - 1, - "IMAGE" ] ], - "groups": [], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -387.72410571388025, + -53.37077389343578, + 737.521556382955, + 1307.389216328689 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 394.6117778691903, + -772.2962805609001, + 1523.2043605195558, + 1035.4596396805657 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], "config": {}, "extra": { - "workflowRendererVersion": "Vue", + "workflowRendererVersion": "LG", "ue_links": [], "ds": { - "scale": 0.520986848192445, + "scale": 0.5644739300537777, "offset": [ - 91.64617544164359, - -275.12756326798984 + 717.24240821608, + 973.4146936919584 ] }, "links_added_by_ue": [], - "frontendVersion": "1.37.11", + "frontendVersion": "1.42.8", "VHS_latentpreview": false, "VHS_latentpreviewrate": 0, "VHS_MetadataImage": true, diff --git a/example_workflows/Advanced_CustomSteps.json b/example_workflows/Advanced_CustomSteps.json new file mode 100644 index 0000000..d5be7db --- /dev/null +++ b/example_workflows/Advanced_CustomSteps.json @@ -0,0 +1,1249 @@ +{ + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", + "revision": 0, + "last_node_id": 223, + "last_link_id": 456, + "nodes": [ + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -377.72410571388025, + 328.4870830026766 + ], + "size": [ + 717.521556382955, + 915.5313594325766 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 213, + "type": "Trellis2ShapeGenerator", + "pos": [ + 783.1875248410059, + -209.1034640968784 + ], + "size": [ + 335.24609375, + 266 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 428 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 429 + }, + { + "name": "coords", + "type": "COORDS", + "link": 430 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 433 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 434 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 431 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeGenerator" + }, + "widgets_values": [ + 512, + 25, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 212, + "type": "Trellis2SparseGenerator", + "pos": [ + 755.5154833297573, + -586.166449201357 + ], + "size": [ + 395.65625, + 314 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 426 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 427 + } + ], + "outputs": [ + { + "name": "coords", + "type": "COORDS", + "links": [ + 430 + ] + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "links": [ + 435 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 428 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2SparseGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + 25, + 7.5, + 0.01, + 5, + "euler", + 32, + 0.1, + 1 + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 385.7151992875252, + -750.0584765375115 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 444 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" + ] + }, + { + "id": 119, + "type": "Trellis2PreProcessImage", + "pos": [ + 424.3737559582122, + -335.9840037159542 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 445 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 208, + "type": "Trellis2ImageCondGenerator", + "pos": [ + 779.7493550843124, + -745.7690930815886 + ], + "size": [ + 308.0748046875, + 98 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 444 + }, + { + "name": "image", + "type": "IMAGE", + "link": 445 + } + ], + "outputs": [ + { + "name": "cond_512", + "type": "IMAGE_COND", + "links": [ + 427, + 429 + ] + }, + { + "name": "cond_1024", + "type": "IMAGE_COND", + "links": [ + 432, + 447 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 426 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ImageCondGenerator" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 214, + "type": "Trellis2ShapeCascadeGenerator", + "pos": [ + 1204.298470222733, + -593.6739172678739 + ], + "size": [ + 335.9791015625, + 338 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 431 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 432 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 433 + }, + { + "name": "from_resolution", + "type": "INT", + "widget": { + "name": "from_resolution" + }, + "link": 434 + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "widget": { + "name": "sparse_structure_resolution" + }, + "link": 435 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 423, + 448 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 424 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 446 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeCascadeGenerator" + }, + "widgets_values": [ + 0, + 1024, + 0, + 999999, + 12, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -373.0898096413628, + 163.12227474974895 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 451 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 216, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1703.393750669334, + -500.59691468161645 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 437 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 451 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 436 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 220, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 1670.7679554530735, + -854.0252749431562 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 443 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 425 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 211, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1658.76843203833, + -712.5207015189599 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 425 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 437 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 215, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1682.2010450446755, + -361.2081091911047 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 436 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 452 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 209, + "type": "Trellis2DecodeLatents", + "pos": [ + 1214.7402059785773, + 139.23732698343224 + ], + "size": [ + 270, + 122 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 450 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 423 + }, + { + "name": "texture_slat", + "shape": 7, + "type": "TEXTURE_SLAT", + "link": 449 + }, + { + "name": "resolution", + "type": "INT", + "widget": { + "name": "resolution" + }, + "link": 424 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 443 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [ + 453 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2DecodeLatents" + }, + "widgets_values": [ + 0, + true + ] + }, + { + "id": 222, + "type": "Trellis2UnWrapAndRasterizer", + "pos": [ + 1627.2488876663126, + -234.41356008052338 + ], + "size": [ + 419.15234375, + 314 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 452 + }, + { + "name": "bvh", + "type": "BVH", + "link": 453 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 454 + ] + }, + { + "name": "base_color_texture", + "type": "IMAGE", + "links": null + }, + { + "name": "metallic_roughness_texture", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2UnWrapAndRasterizer" + }, + "widgets_values": [ + 60, + 0, + 1, + 1, + 2048, + "OPAQUE", + false, + false, + false, + "telea" + ] + }, + { + "id": 223, + "type": "Trellis2ExportMesh", + "pos": [ + 1641.6918045364303, + 146.40641977787487 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 454 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 456 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 455 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 405.0700414873716, + 334.1097337450922 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 455 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "Preview3D" + }, + "widgets_values": [ + "", + "" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -368.2858630795249, + 34.401714106564214 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 456 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 221, + "type": "Trellis2TexSlatGenerator", + "pos": [ + 1188.8316361245397, + -197.22556623205915 + ], + "size": [ + 340.390625, + 266 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 446 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 447 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 448 + } + ], + "outputs": [ + { + "name": "texture_slat", + "type": "TEXTURE_SLAT", + "links": [ + 449 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 450 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2TexSlatGenerator" + }, + "widgets_values": [ + 1024, + 12, + 3, + 0.01, + 3, + "euler", + 0, + 0.9 + ] + } + ], + "links": [ + [ + 241, + 69, + 2, + 119, + 0, + "IMAGE" + ], + [ + 423, + 214, + 0, + 209, + 1, + "SHAPE_SLAT" + ], + [ + 424, + 214, + 1, + 209, + 3, + "INT" + ], + [ + 425, + 220, + 0, + 211, + 0, + "MESHWITHVOXEL" + ], + [ + 426, + 208, + 2, + 212, + 0, + "TRELLIS2PIPELINE" + ], + [ + 427, + 208, + 0, + 212, + 1, + "IMAGE_COND" + ], + [ + 428, + 212, + 2, + 213, + 0, + "TRELLIS2PIPELINE" + ], + [ + 429, + 208, + 0, + 213, + 1, + "IMAGE_COND" + ], + [ + 430, + 212, + 0, + 213, + 2, + "COORDS" + ], + [ + 431, + 213, + 2, + 214, + 0, + "TRELLIS2PIPELINE" + ], + [ + 432, + 208, + 1, + 214, + 1, + "IMAGE_COND" + ], + [ + 433, + 213, + 0, + 214, + 2, + "SHAPE_SLAT" + ], + [ + 434, + 213, + 1, + 214, + 3, + "INT" + ], + [ + 435, + 212, + 1, + 214, + 4, + "INT" + ], + [ + 436, + 216, + 0, + 215, + 0, + "MESHWITHVOXEL" + ], + [ + 437, + 211, + 0, + 216, + 0, + "MESHWITHVOXEL" + ], + [ + 443, + 209, + 0, + 220, + 0, + "MESHWITHVOXEL" + ], + [ + 444, + 39, + 0, + 208, + 0, + "TRELLIS2PIPELINE" + ], + [ + 445, + 119, + 0, + 208, + 1, + "IMAGE" + ], + [ + 446, + 214, + 2, + 221, + 0, + "TRELLIS2PIPELINE" + ], + [ + 447, + 208, + 1, + 221, + 1, + "IMAGE_COND" + ], + [ + 448, + 214, + 0, + 221, + 2, + "SHAPE_SLAT" + ], + [ + 449, + 221, + 0, + 209, + 2, + "TEXTURE_SLAT" + ], + [ + 450, + 221, + 1, + 209, + 0, + "TRELLIS2PIPELINE" + ], + [ + 451, + 203, + 0, + 216, + 1, + "INT" + ], + [ + 452, + 215, + 0, + 222, + 0, + "MESHWITHVOXEL" + ], + [ + 453, + 209, + 1, + 222, + 1, + "BVH" + ], + [ + 454, + 222, + 0, + 223, + 0, + "TRIMESH" + ], + [ + 455, + 223, + 0, + 202, + 2, + "STRING" + ], + [ + 456, + 204, + 0, + 223, + 1, + "STRING" + ] + ], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -387.72410571388025, + -53.37077389343578, + 737.521556382955, + 1307.389216328689 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 375.7151992875252, + -957.7197012700683, + 1860.9819191012211, + 1225.6071509713981 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "workflowRendererVersion": "LG", + "ue_links": [], + "ds": { + "scale": 0.6209213230591558, + "offset": [ + 341.8543937245997, + 855.1515920610424 + ] + }, + "links_added_by_ue": [], + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/Advanced_CustomSteps_MeshOnly.json b/example_workflows/Advanced_CustomSteps_MeshOnly.json new file mode 100644 index 0000000..844feeb --- /dev/null +++ b/example_workflows/Advanced_CustomSteps_MeshOnly.json @@ -0,0 +1,1118 @@ +{ + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", + "revision": 0, + "last_node_id": 224, + "last_link_id": 459, + "nodes": [ + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -377.72410571388025, + 328.4870830026766 + ], + "size": [ + 717.521556382955, + 915.5313594325766 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 213, + "type": "Trellis2ShapeGenerator", + "pos": [ + 783.1875248410059, + -209.1034640968784 + ], + "size": [ + 335.24609375, + 266 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 428 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 429 + }, + { + "name": "coords", + "type": "COORDS", + "link": 430 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 433 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 434 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 431 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeGenerator" + }, + "widgets_values": [ + 512, + 25, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 212, + "type": "Trellis2SparseGenerator", + "pos": [ + 755.5154833297573, + -586.166449201357 + ], + "size": [ + 395.65625, + 314 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 426 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 427 + } + ], + "outputs": [ + { + "name": "coords", + "type": "COORDS", + "links": [ + 430 + ] + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "links": [ + 435 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 428 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2SparseGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + 25, + 7.5, + 0.01, + 5, + "euler", + 32, + 0.1, + 1 + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 385.7151992875252, + -750.0584765375115 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 444 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" + ] + }, + { + "id": 119, + "type": "Trellis2PreProcessImage", + "pos": [ + 424.3737559582122, + -335.9840037159542 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 445 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 208, + "type": "Trellis2ImageCondGenerator", + "pos": [ + 779.7493550843124, + -745.7690930815886 + ], + "size": [ + 308.0748046875, + 98 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 444 + }, + { + "name": "image", + "type": "IMAGE", + "link": 445 + } + ], + "outputs": [ + { + "name": "cond_512", + "type": "IMAGE_COND", + "links": [ + 427, + 429 + ] + }, + { + "name": "cond_1024", + "type": "IMAGE_COND", + "links": [ + 432 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 426 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ImageCondGenerator" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 214, + "type": "Trellis2ShapeCascadeGenerator", + "pos": [ + 1204.298470222733, + -593.6739172678739 + ], + "size": [ + 335.9791015625, + 338 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 431 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 432 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 433 + }, + { + "name": "from_resolution", + "type": "INT", + "widget": { + "name": "from_resolution" + }, + "link": 434 + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "widget": { + "name": "sparse_structure_resolution" + }, + "link": 435 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 423 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 424 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 457 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeCascadeGenerator" + }, + "widgets_values": [ + 0, + 1024, + 0, + 999999, + 12, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -373.0898096413628, + 163.12227474974895 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 451 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 220, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 1670.7679554530735, + -854.0252749431562 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 443 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 425 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 211, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1658.76843203833, + -712.5207015189599 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 425 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 437 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 405.0700414873716, + 334.1097337450922 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 455 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "Preview3D" + }, + "widgets_values": [ + "", + "" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -368.2858630795249, + 34.401714106564214 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 456 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 209, + "type": "Trellis2DecodeLatents", + "pos": [ + 1222.2559520778204, + -168.90688691732427 + ], + "size": [ + 270, + 122 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 457 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 423 + }, + { + "name": "texture_slat", + "shape": 7, + "type": "TEXTURE_SLAT", + "link": null + }, + { + "name": "resolution", + "type": "INT", + "widget": { + "name": "resolution" + }, + "link": 424 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 443 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2DecodeLatents" + }, + "widgets_values": [ + 0, + true + ] + }, + { + "id": 224, + "type": "Trellis2MeshWithVoxelToTrimesh", + "pos": [ + 1658.7016089575882, + -75.3276646144487 + ], + "size": [ + 349.41171875, + 58 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 458 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 459 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + }, + "widgets_values": [ + "90 degrees" + ] + }, + { + "id": 215, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1668.2431606477028, + -224.8516122407264 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 436 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 458 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 216, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1657.2258628678205, + -436.17651468161654 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 437 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 451 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 436 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 223, + "type": "Trellis2ExportMesh", + "pos": [ + 1660.8568735364302, + 90.69364157484758 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 459 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 456 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 455 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + } + ], + "links": [ + [ + 241, + 69, + 2, + 119, + 0, + "IMAGE" + ], + [ + 423, + 214, + 0, + 209, + 1, + "SHAPE_SLAT" + ], + [ + 424, + 214, + 1, + 209, + 3, + "INT" + ], + [ + 425, + 220, + 0, + 211, + 0, + "MESHWITHVOXEL" + ], + [ + 426, + 208, + 2, + 212, + 0, + "TRELLIS2PIPELINE" + ], + [ + 427, + 208, + 0, + 212, + 1, + "IMAGE_COND" + ], + [ + 428, + 212, + 2, + 213, + 0, + "TRELLIS2PIPELINE" + ], + [ + 429, + 208, + 0, + 213, + 1, + "IMAGE_COND" + ], + [ + 430, + 212, + 0, + 213, + 2, + "COORDS" + ], + [ + 431, + 213, + 2, + 214, + 0, + "TRELLIS2PIPELINE" + ], + [ + 432, + 208, + 1, + 214, + 1, + "IMAGE_COND" + ], + [ + 433, + 213, + 0, + 214, + 2, + "SHAPE_SLAT" + ], + [ + 434, + 213, + 1, + 214, + 3, + "INT" + ], + [ + 435, + 212, + 1, + 214, + 4, + "INT" + ], + [ + 436, + 216, + 0, + 215, + 0, + "MESHWITHVOXEL" + ], + [ + 437, + 211, + 0, + 216, + 0, + "MESHWITHVOXEL" + ], + [ + 443, + 209, + 0, + 220, + 0, + "MESHWITHVOXEL" + ], + [ + 444, + 39, + 0, + 208, + 0, + "TRELLIS2PIPELINE" + ], + [ + 445, + 119, + 0, + 208, + 1, + "IMAGE" + ], + [ + 451, + 203, + 0, + 216, + 1, + "INT" + ], + [ + 455, + 223, + 0, + 202, + 2, + "STRING" + ], + [ + 456, + 204, + 0, + 223, + 1, + "STRING" + ], + [ + 457, + 214, + 2, + 209, + 0, + "TRELLIS2PIPELINE" + ], + [ + 458, + 215, + 0, + 224, + 0, + "MESHWITHVOXEL" + ], + [ + 459, + 224, + 0, + 223, + 0, + "TRIMESH" + ] + ], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -387.72410571388025, + -53.37077389343578, + 737.521556382955, + 1307.389216328689 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 375.7151992875252, + -957.7197012700683, + 1860.9819191012211, + 1225.6071509713981 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "workflowRendererVersion": "LG", + "ue_links": [], + "ds": { + "scale": 0.5131581182307072, + "offset": [ + 698.5290381042139, + 1125.1909919152047 + ] + }, + "links_added_by_ue": [], + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/MultiViews.json b/example_workflows/MultiViews.json index 4882eb8..a40a9d3 100644 --- a/example_workflows/MultiViews.json +++ b/example_workflows/MultiViews.json @@ -1,51 +1,28 @@ { - "id": "cb2e6635-37a0-47a0-bf98-1cfad1b842c2", + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", "revision": 0, - "last_node_id": 16, - "last_link_id": 19, + "last_node_id": 243, + "last_link_id": 494, "nodes": [ { - "id": 1, - "type": "Trellis2MeshWithVoxelMultiViewGenerator", + "id": 227, + "type": "Trellis2ReconstructMeshWithQuad", "pos": [ - 1160.6611250957358, - 1551.885281893709 + 1176.3670349681408, + -707.7757475998008 ], "size": [ - 612.71875, - 972.65625 + 331.5878996659427, + 142.11360851901668 ], "flags": {}, - "order": 5, + "order": 13, "mode": 0, "inputs": [ { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", - "link": 8 - }, - { - "name": "front_image", - "type": "IMAGE", - "link": 2 - }, - { - "name": "back_image", - "shape": 7, - "type": "IMAGE", - "link": 10 - }, - { - "name": "left_image", - "shape": 7, - "type": "IMAGE", - "link": null - }, - { - "name": "right_image", - "shape": 7, - "type": "IMAGE", - "link": null + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 470 } ], "outputs": [ @@ -53,63 +30,120 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 16 - ] - }, - { - "name": "bvh", - "type": "BVH", - "links": [ - 17 + 472 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", "widget_ue_connectable": {} }, "widgets_values": [ - 12345, - "fixed", - "1024_cascade", - 25, - 6.5, - 0.2, - 4, - 25, - 6.5, - 0.2, - 4, - 25, - 3, - 0.2, - 3, - 999999, - 32, - true, - 0.1, 1, - 0.1, - 1, - 0, - 0.9, + 1024, true, - "z", - 2 + true ] }, { - "id": 3, - "type": "Trellis2LoadImageWithTransparency", + "id": 231, + "type": "Trellis2FillHolesWithCuMesh", "pos": [ - 189.80670439097503, - 1011.8182611411379 + 1189.269979801219, + -832.1907924780478 ], "size": [ - 453.4375, - 837.1875 + 312.4361328125, + 58 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 487 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 470 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 229, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1181.0098963429245, + -503.10569576014126 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 472 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 475 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 471 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -840.0731525127071, + -237.35696845568904 + ], + "size": [ + 270, + 82 ], "flags": {}, "order": 0, @@ -117,85 +151,34 @@ "inputs": [], "outputs": [ { - "name": "image", - "type": "IMAGE", - "links": null - }, - { - "name": "mask", - "type": "MASK", - "links": null - }, - { - "name": "image_with_alpha", - "type": "IMAGE", + "name": "INT", + "type": "INT", "links": [ - 1 + 475 ] } ], + "title": "Target Face Number", "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2LoadImageWithTransparency", - "widget_ue_connectable": {} + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" }, "widgets_values": [ - "Image_500_00001_.png", - "image" + 300000, + "fixed" ] }, { - "id": 4, - "type": "Trellis2PreProcessImage", + "id": 204, + "type": "PrimitiveString", "pos": [ - 739.4753259488509, - 1571.2899959426204 + -848.2606713054529, + -370.80161968053943 ], "size": [ - 349.09375, - 107.328125 - ], - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 1 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 2 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2PreProcessImage", - "widget_ue_connectable": {} - }, - "widgets_values": [ - 25, - false - ] - }, - { - "id": 10, - "type": "Trellis2LoadModel", - "pos": [ - 720.4210852610131, - 1268.6993659791224 - ], - "size": [ - 355.5, - 207.328125 + 270, + 58 ], "flags": {}, "order": 1, @@ -203,204 +186,50 @@ "inputs": [], "outputs": [ { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", + "name": "STRING", + "type": "STRING", "links": [ - 8 + 479 ] } ], + "title": "Name", "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2LoadModel", - "widget_ue_connectable": {} + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" }, "widgets_values": [ - "TRELLIS.2-4B", - "flash_attn", - "cuda", - true, - false + "Tank" ] }, { - "id": 11, - "type": "Trellis2LoadImageWithTransparency", - "pos": [ - 196.16440063611725, - 1894.1248321979997 - ], - "size": [ - 443.6875, - 849.5 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": null - }, - { - "name": "mask", - "type": "MASK", - "links": null - }, - { - "name": "image_with_alpha", - "type": "IMAGE", - "links": [ - 9 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2LoadImageWithTransparency", - "widget_ue_connectable": {} - }, - "widgets_values": [ - "Image_480_00001_.png", - "image" - ] - }, - { - "id": 12, - "type": "Trellis2PreProcessImage", - "pos": [ - 739.1482171781413, - 1738.5360038621288 - ], - "size": [ - 354.203125, - 107.328125 - ], - "flags": {}, - "order": 4, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 9 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 10 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "Node name for S&R": "Trellis2PreProcessImage", - "widget_ue_connectable": {} - }, - "widgets_values": [ - 25, - false - ] - }, - { - "id": 14, - "type": "Trellis2PostProcessAndUnWrapAndRasterizer", - "pos": [ - 1823.7677651122783, - 1551.9434773806804 - ], - "size": [ - 561.796875, - 674.578125 - ], - "flags": {}, - "order": 6, - "mode": 0, - "inputs": [ - { - "name": "mesh", - "type": "MESHWITHVOXEL", - "link": 16 - }, - { - "name": "bvh", - "type": "BVH", - "link": 17 - } - ], - "outputs": [ - { - "name": "trimesh", - "type": "TRIMESH", - "links": [ - 18 - ] - }, - { - "name": "base_color_texture", - "type": "IMAGE", - "links": null - }, - { - "name": "metallic_roughness_texture", - "type": "IMAGE", - "links": null - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer" - }, - "widgets_values": [ - 60, - 0, - 1, - 1, - 4096, - true, - 1, - 0, - 500000, - "Cumesh", - true, - "OPAQUE", - "1024", - false, - true, - false, - false, - true - ] - }, - { - "id": 15, + "id": 233, "type": "Trellis2ExportMesh", "pos": [ - 2476.893251656926, - 1550.6756032446528 + 1191.3255568422467, + -77.49101003118169 ], "size": [ - 329.8125, - 148.75 + 270, + 102 ], "flags": {}, - "order": 7, + "order": 17, "mode": 0, "inputs": [ { "name": "trimesh", "type": "TRIMESH", - "link": 18 + "link": 494 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 479 } ], "outputs": [ @@ -408,35 +237,254 @@ "name": "glb_path", "type": "STRING", "links": [ - 19 + 480 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 391.3481025606576, + -777.5583507484255 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 483 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2ExportMesh" + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} }, "widgets_values": [ - "Trellis2MV", - "glb", - true + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" ] }, { - "id": 16, - "type": "Preview3D", + "id": 119, + "type": "Trellis2PreProcessImage", "pos": [ - 2420.5741031607986, - 1763.4314920895954 + 379.7979564010223, + -519.6169765183606 ], "size": [ - 842.90625, - 949.015625 + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 484 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 239, + "type": "Trellis2PreProcessImage", + "pos": [ + 386.64212656289783, + -361.6545116326391 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 485 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 486 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 240, + "type": "Trellis2PreProcessImage", + "pos": [ + 384.7958133512735, + -204.2542485606769 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 488 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 489 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 241, + "type": "Trellis2PreProcessImage", + "pos": [ + 384.795813351274, + -50.03583437610661 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 490 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 491 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 360.99306605686377, + 121.97109784511967 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 ], "flags": {}, - "order": 8, + "order": 18, "mode": 0, "inputs": [ { @@ -457,110 +505,601 @@ "widget": { "name": "model_file" }, - "link": 19 + "link": 480 } ], "outputs": [], "properties": { "cnr_id": "comfy-core", - "ver": "0.13.0", - "widget_ue_connectable": {}, + "ver": "0.18.1", "Node name for S&R": "Preview3D" }, "widgets_values": [ "", "" ] + }, + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -847.5380090037122, + -82.9715880857893 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "title": "Front Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 236, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -258.0644919609741, + -81.59760778432084 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 485 + ] + } + ], + "title": "Back Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 237, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -834.3440812871166, + 489.10851591788764 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 488 + ] + } + ], + "title": "Left Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 238, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -256.71749538654217, + 489.176741691081 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 490 + ] + } + ], + "title": "Right Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 228, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1184.3251213630779, + -362.82432619683135 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 471 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 492 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 235, + "type": "Trellis2MeshWithVoxelMultiViewGenerator", + "pos": [ + 730.5993372134623, + -806.3203136622369 + ], + "size": [ + 416.855078125, + 786 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 483 + }, + { + "name": "front_image", + "type": "IMAGE", + "link": 484 + }, + { + "name": "back_image", + "shape": 7, + "type": "IMAGE", + "link": 486 + }, + { + "name": "left_image", + "shape": 7, + "type": "IMAGE", + "link": 489 + }, + { + "name": "right_image", + "shape": 7, + "type": "IMAGE", + "link": 491 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 487 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [ + 493 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + "1024_cascade", + 25, + 7.5, + 0.01, + 5, + 12, + 7.5, + 0.01, + 3, + 12, + 3, + 0.01, + 3, + 999999, + 32, + true, + 0.1, + 1, + 0.1, + 1, + 0, + 0.9, + true, + "z", + 1, + "heun" + ] + }, + { + "id": 242, + "type": "Trellis2UnWrapAndRasterizer", + "pos": [ + 1496.5934369390998, + -290.13985378139296 + ], + "size": [ + 419.15234375, + 314 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 492 + }, + { + "name": "bvh", + "type": "BVH", + "link": 493 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 494 + ] + }, + { + "name": "base_color_texture", + "type": "IMAGE", + "links": null + }, + { + "name": "metallic_roughness_texture", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2UnWrapAndRasterizer" + }, + "widgets_values": [ + 60, + 0, + 1, + 1, + 2048, + "OPAQUE", + false, + false, + false, + "telea" + ] } ], "links": [ [ - 1, - 3, + 241, + 69, 2, - 4, + 119, 0, "IMAGE" ], [ - 2, - 4, + 470, + 231, 0, - 1, - 1, - "IMAGE" - ], - [ - 8, - 10, - 0, - 1, - 0, - "TRELLIS2PIPELINE" - ], - [ - 9, - 11, - 2, - 12, - 0, - "IMAGE" - ], - [ - 10, - 12, - 0, - 1, - 2, - "IMAGE" - ], - [ - 16, - 1, - 0, - 14, + 227, 0, "MESHWITHVOXEL" ], [ - 17, + 471, + 229, + 0, + 228, + 0, + "MESHWITHVOXEL" + ], + [ + 472, + 227, + 0, + 229, + 0, + "MESHWITHVOXEL" + ], + [ + 475, + 203, + 0, + 229, 1, + "INT" + ], + [ + 479, + 204, + 0, + 233, 1, - 14, + "STRING" + ], + [ + 480, + 233, + 0, + 202, + 2, + "STRING" + ], + [ + 483, + 39, + 0, + 235, + 0, + "TRELLIS2PIPELINE" + ], + [ + 484, + 119, + 0, + 235, + 1, + "IMAGE" + ], + [ + 485, + 236, + 2, + 239, + 0, + "IMAGE" + ], + [ + 486, + 239, + 0, + 235, + 2, + "IMAGE" + ], + [ + 487, + 235, + 0, + 231, + 0, + "MESHWITHVOXEL" + ], + [ + 488, + 237, + 2, + 240, + 0, + "IMAGE" + ], + [ + 489, + 240, + 0, + 235, + 3, + "IMAGE" + ], + [ + 490, + 238, + 2, + 241, + 0, + "IMAGE" + ], + [ + 491, + 241, + 0, + 235, + 4, + "IMAGE" + ], + [ + 492, + 228, + 0, + 242, + 0, + "MESHWITHVOXEL" + ], + [ + 493, + 235, + 1, + 242, 1, "BVH" ], [ - 18, - 14, + 494, + 242, 0, - 15, + 233, 0, "TRIMESH" - ], - [ - 19, - 15, - 0, - 16, - 2, - "STRING" ] ], - "groups": [], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -874.7851579398082, + -458.57410768053984, + 1193.2585052367917, + 1492.7062709864938 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 360.7008822276498, + -937.193468065821, + 1585.9419620118415, + 1014.8225405559199 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], "config": {}, "extra": { + "workflowRendererVersion": "LG", "ue_links": [], "ds": { - "scale": 0.45020114458154903, + "scale": 0.5644739300537791, "offset": [ - 307.1190043799142, - -866.3604306862078 + 1098.7675999602163, + 1065.1675662022888 ] }, - "workflowRendererVersion": "Vue", "links_added_by_ue": [], - "frontendVersion": "1.38.13", + "frontendVersion": "1.42.8", "VHS_latentpreview": false, "VHS_latentpreviewrate": 0, "VHS_MetadataImage": true, diff --git a/example_workflows/MultiViews_MeshOnly.json b/example_workflows/MultiViews_MeshOnly.json index 308afc3..202b912 100644 --- a/example_workflows/MultiViews_MeshOnly.json +++ b/example_workflows/MultiViews_MeshOnly.json @@ -1,156 +1,28 @@ { - "id": "440427ce-0c6a-4462-af51-639ed2f16dec", + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", "revision": 0, - "last_node_id": 121, - "last_link_id": 230, + "last_node_id": 241, + "last_link_id": 491, "nodes": [ { - "id": 50, - "type": "Trellis2PreProcessImage", + "id": 227, + "type": "Trellis2ReconstructMeshWithQuad", "pos": [ - -514.210025695023, - 250.92630997998222 + 1176.3670349681408, + -707.7757475998008 ], "size": [ - 297.84375, - 107.328125 + 331.5878996659427, + 142.11360851901668 ], "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 110 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 223 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "15854282a73cf231b81d52ada22e652f414a078b", - "Node name for S&R": "Trellis2PreProcessImage", - "widget_ue_connectable": {} - }, - "widgets_values": [ - 25, - false - ] - }, - { - "id": 6, - "type": "Trellis2LoadImageWithTransparency", - "pos": [ - -1354.11041692169, - -103.28960088005408 - ], - "size": [ - 610.640625, - 788.515625 - ], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": null - }, - { - "name": "mask", - "type": "MASK", - "links": null - }, - { - "name": "image_with_alpha", - "type": "IMAGE", - "links": [ - 110 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", - "Node name for S&R": "Trellis2LoadImageWithTransparency", - "widget_ue_connectable": {} - }, - "widgets_values": [ - "Image_1024_00010_.png", - "image" - ] - }, - { - "id": 39, - "type": "Trellis2LoadModel", - "pos": [ - -502.8470854177922, - -74.32512914672895 - ], - "size": [ - 336.0625, - 207.328125 - ], - "flags": { - "collapsed": false - }, - "order": 1, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", - "links": [ - 222 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", - "Node name for S&R": "Trellis2LoadModel", - "widget_ue_connectable": {} - }, - "widgets_values": [ - "TRELLIS.2-4B", - "flash_attn", - "cuda", - true, - false - ] - }, - { - "id": 103, - "type": "Trellis2RemeshWithQuad", - "pos": [ - 467.2386692680435, - 243.87402592407818 - ], - "size": [ - 455.3125, - 262 - ], - "flags": { - "collapsed": false - }, - "order": 6, + "order": 13, "mode": 0, "inputs": [ { "name": "mesh", "type": "MESHWITHVOXEL", - "link": 225 + "link": 470 } ], "outputs": [ @@ -158,158 +30,42 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 226 + 472 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "258cd607667d64b01b2abdaded4e59016ffd5cb6", - "Node name for S&R": "Trellis2RemeshWithQuad", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", "widget_ue_connectable": {} }, "widgets_values": [ 1, - 0, - false, - 0.03, - "1024", + 1024, true, true ] }, { - "id": 114, - "type": "Trellis2LoadImageWithTransparency", + "id": 231, + "type": "Trellis2FillHolesWithCuMesh", "pos": [ - -1359.0614927706545, - 735.4082082769835 + 1189.269979801219, + -832.1907924780478 ], "size": [ - 609.125, - 759.125 - ], - "flags": { - "collapsed": false - }, - "order": 2, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [] - }, - { - "name": "mask", - "type": "MASK", - "links": [] - }, - { - "name": "image_with_alpha", - "type": "IMAGE", - "links": [ - 221 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2LoadImageWithTransparency" - }, - "widgets_values": [ - "Image_1024_00021_.png", - "image" - ] - }, - { - "id": 115, - "type": "Trellis2PreProcessImage", - "pos": [ - -520.4300253912838, - 427.5271012938298 - ], - "size": [ - 311.921875, - 107.328125 - ], - "flags": { - "collapsed": false - }, - "order": 4, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 221 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 224 - ] - } - ], - "properties": { - "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "15854282a73cf231b81d52ada22e652f414a078b", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2PreProcessImage" - }, - "widgets_values": [ - 50, - false - ] - }, - { - "id": 116, - "type": "Trellis2MeshWithVoxelMultiViewGenerator", - "pos": [ - -76.41514399946482, - 241.60506031220336 - ], - "size": [ - 468.21875, - 1000.53125 + 312.4361328125, + 58 ], "flags": {}, - "order": 5, + "order": 12, "mode": 0, "inputs": [ { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", - "link": 222 - }, - { - "name": "front_image", - "type": "IMAGE", - "link": 223 - }, - { - "name": "back_image", - "shape": 7, - "type": "IMAGE", - "link": 224 - }, - { - "name": "left_image", - "shape": 7, - "type": "IMAGE", - "link": null - }, - { - "name": "right_image", - "shape": 7, - "type": "IMAGE", - "link": null + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 487 } ], "outputs": [ @@ -317,70 +73,46 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 225 + 470 ] - }, - { - "name": "bvh", - "type": "BVH", - "links": null } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator" + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" }, "widgets_values": [ - 12345, - "randomize", - "1024_cascade", - 25, - 6.5, - 0.2, - 4, - 25, - 6.5, - 0.2, - 4, - 12, - 3, - 0.2, - 3, - 999999, - 32, - false, - 0.1, - 1, - 0.1, - 1, - 0, - 0.9, - true, - "z", - 2 + 1 ] }, { - "id": 117, + "id": 229, "type": "Trellis2SimplifyMesh", "pos": [ - 472.141197337784, - 564.9329628175803 + 1181.0098963429245, + -503.10569576014126 ], "size": [ - 317, - 115.046875 + 270, + 82 ], "flags": {}, - "order": 7, + "order": 14, "mode": 0, "inputs": [ { "name": "mesh", "type": "MESHWITHVOXEL", - "link": 226 + "link": 472 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 475 } ], "outputs": [ @@ -388,14 +120,13 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 227 + 471 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", "Node name for S&R": "Trellis2SimplifyMesh" }, "widgets_values": [ @@ -404,24 +135,93 @@ ] }, { - "id": 118, - "type": "Trellis2FillHolesWithMeshlib", + "id": 203, + "type": "PrimitiveInt", "pos": [ - 473.1986723221411, - 743.6571759057757 + -840.0731525127071, + -237.35696845568904 ], "size": [ - 348.71875, - 71.328125 + 270, + 82 ], "flags": {}, - "order": 8, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 475 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -848.2606713054529, + -370.80161968053943 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 479 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 228, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1184.3251213630779, + -362.82432619683135 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 15, "mode": 0, "inputs": [ { "name": "mesh", "type": "MESHWITHVOXEL", - "link": 227 + "link": 471 } ], "outputs": [ @@ -429,7 +229,7 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 228 + 481 ] }, { @@ -440,31 +240,38 @@ ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", "Node name for S&R": "Trellis2FillHolesWithMeshlib" }, "widgets_values": [] }, { - "id": 119, + "id": 233, "type": "Trellis2ExportMesh", "pos": [ - 471.08304460897625, - 1037.6532301953084 + 1191.3255568422467, + -77.49101003118169 ], "size": [ - 348.71875, - 148.109375 + 270, + 102 ], "flags": {}, - "order": 10, + "order": 17, "mode": 0, "inputs": [ { "name": "trimesh", "type": "TRIMESH", - "link": 229 + "link": 482 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 479 } ], "outputs": [ @@ -472,75 +279,254 @@ "name": "glb_path", "type": "STRING", "links": [ - 230 + 480 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 391.3481025606576, + -777.5583507484255 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 483 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2ExportMesh" + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} }, "widgets_values": [ - "Trellis2MV", - "glb", - true + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" ] }, { - "id": 120, - "type": "Trellis2MeshWithVoxelToTrimesh", + "id": 119, + "type": "Trellis2PreProcessImage", "pos": [ - 472.14061641396916, - 874.7920616685096 + 379.7979564010223, + -519.6169765183606 ], "size": [ - 332.859375, - 92.5625 + 281.8229166666667, + 106 ], - "flags": {}, - "order": 9, + "flags": { + "collapsed": false + }, + "order": 7, "mode": 0, "inputs": [ { - "name": "mesh", - "type": "MESHWITHVOXEL", - "link": 228 + "name": "image", + "type": "IMAGE", + "link": 241 } ], "outputs": [ { - "name": "trimesh", - "type": "TRIMESH", + "name": "image", + "type": "IMAGE", "links": [ - 229 + 484 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a19110a28a5c5c434386af0421005dd8edae82db", - "widget_ue_connectable": {}, - "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} }, "widgets_values": [ - "90 degrees" + 10, + false, + 1024 ] }, { - "id": 121, - "type": "Preview3D", + "id": 239, + "type": "Trellis2PreProcessImage", "pos": [ - 960.7538631636339, - 252.95888156302374 + 386.64212656289783, + -361.6545116326391 ], "size": [ - 892.296875, - 936.96875 + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 485 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 486 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 240, + "type": "Trellis2PreProcessImage", + "pos": [ + 384.7958133512735, + -204.2542485606769 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 488 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 489 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 241, + "type": "Trellis2PreProcessImage", + "pos": [ + 384.795813351274, + -50.03583437610661 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": { + "collapsed": false + }, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 490 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 491 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 360.99306605686377, + 121.97109784511967 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 ], "flags": {}, - "order": 11, + "order": 18, "mode": 0, "inputs": [ { @@ -561,126 +547,525 @@ "widget": { "name": "model_file" }, - "link": 230 + "link": 480 } ], "outputs": [], "properties": { "cnr_id": "comfy-core", - "ver": "0.13.0", - "widget_ue_connectable": {}, + "ver": "0.18.1", "Node name for S&R": "Preview3D" }, "widgets_values": [ "", "" ] + }, + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -847.5380090037122, + -82.9715880857893 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "title": "Front Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 236, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -258.0644919609741, + -81.59760778432084 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 485 + ] + } + ], + "title": "Back Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 237, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -834.3440812871166, + 489.10851591788764 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 488 + ] + } + ], + "title": "Left Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 238, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -256.71749538654217, + 489.176741691081 + ], + "size": [ + 496.4332376519267, + 498.75383455750784 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 490 + ] + } + ], + "title": "Right Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 235, + "type": "Trellis2MeshWithVoxelMultiViewGenerator", + "pos": [ + 730.5993372134623, + -806.3203136622369 + ], + "size": [ + 416.855078125, + 786 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 483 + }, + { + "name": "front_image", + "type": "IMAGE", + "link": 484 + }, + { + "name": "back_image", + "shape": 7, + "type": "IMAGE", + "link": 486 + }, + { + "name": "left_image", + "shape": 7, + "type": "IMAGE", + "link": 489 + }, + { + "name": "right_image", + "shape": 7, + "type": "IMAGE", + "link": 491 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 487 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelMultiViewGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + "1024_cascade", + 25, + 7.5, + 0.01, + 5, + 12, + 7.5, + 0.01, + 3, + 12, + 3, + 0.01, + 3, + 999999, + 32, + false, + 0.1, + 1, + 0.1, + 1, + 0, + 0.9, + true, + "z", + 1, + "heun" + ] + }, + { + "id": 234, + "type": "Trellis2MeshWithVoxelToTrimesh", + "pos": [ + 1171.5679242041226, + -219.6291722499163 + ], + "size": [ + 349.41171875, + 58 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 481 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 482 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + }, + "widgets_values": [ + "90 degrees" + ] } ], "links": [ [ - 110, - 6, + 241, + 69, 2, - 50, - 0, - "IMAGE" - ], - [ - 221, - 114, - 2, - 115, - 0, - "IMAGE" - ], - [ - 222, - 39, - 0, - 116, - 0, - "TRELLIS2PIPELINE" - ], - [ - 223, - 50, - 0, - 116, - 1, - "IMAGE" - ], - [ - 224, - 115, - 0, - 116, - 2, - "IMAGE" - ], - [ - 225, - 116, - 0, - 103, - 0, - "MESHWITHVOXEL" - ], - [ - 226, - 103, - 0, - 117, - 0, - "MESHWITHVOXEL" - ], - [ - 227, - 117, - 0, - 118, - 0, - "MESHWITHVOXEL" - ], - [ - 228, - 118, - 0, - 120, - 0, - "MESHWITHVOXEL" - ], - [ - 229, - 120, - 0, 119, 0, + "IMAGE" + ], + [ + 470, + 231, + 0, + 227, + 0, + "MESHWITHVOXEL" + ], + [ + 471, + 229, + 0, + 228, + 0, + "MESHWITHVOXEL" + ], + [ + 472, + 227, + 0, + 229, + 0, + "MESHWITHVOXEL" + ], + [ + 475, + 203, + 0, + 229, + 1, + "INT" + ], + [ + 479, + 204, + 0, + 233, + 1, + "STRING" + ], + [ + 480, + 233, + 0, + 202, + 2, + "STRING" + ], + [ + 481, + 228, + 0, + 234, + 0, + "MESHWITHVOXEL" + ], + [ + 482, + 234, + 0, + 233, + 0, "TRIMESH" ], [ - 230, + 483, + 39, + 0, + 235, + 0, + "TRELLIS2PIPELINE" + ], + [ + 484, 119, 0, - 121, + 235, + 1, + "IMAGE" + ], + [ + 485, + 236, 2, - "STRING" + 239, + 0, + "IMAGE" + ], + [ + 486, + 239, + 0, + 235, + 2, + "IMAGE" + ], + [ + 487, + 235, + 0, + 231, + 0, + "MESHWITHVOXEL" + ], + [ + 488, + 237, + 2, + 240, + 0, + "IMAGE" + ], + [ + 489, + 240, + 0, + 235, + 3, + "IMAGE" + ], + [ + 490, + 238, + 2, + 241, + 0, + "IMAGE" + ], + [ + 491, + 241, + 0, + 235, + 4, + "IMAGE" ] ], - "groups": [], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -874.7851579398082, + -458.57410768053984, + 1193.2585052367917, + 1492.7062709864938 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 360.7008822276498, + -937.193468065821, + 1227.3780156118423, + 1005.7286067160883 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], "config": {}, "extra": { - "workflowRendererVersion": "Vue", + "workflowRendererVersion": "LG", "ue_links": [], "ds": { - "scale": 0.4736244074476824, + "scale": 0.513158118230708, "offset": [ - 1666.544822537185, - 301.4636979965617 + 1358.4460295288495, + 1222.5907408893986 ] }, "links_added_by_ue": [], - "frontendVersion": "1.38.13", + "frontendVersion": "1.42.8", "VHS_latentpreview": false, "VHS_latentpreviewrate": 0, "VHS_MetadataImage": true, diff --git a/example_workflows/Projection_6Views_Hy20.json b/example_workflows/Projection_6Views_Hy20.json new file mode 100644 index 0000000..b6e70e1 --- /dev/null +++ b/example_workflows/Projection_6Views_Hy20.json @@ -0,0 +1,8578 @@ +{ + "id": "d8f6c381-6a87-4599-b2c1-0b140608c3ce", + "revision": 0, + "last_node_id": 1620, + "last_link_id": 2048, + "nodes": [ + { + "id": 419, + "type": "SetNode", + "pos": [ + -745.6584936204468, + -1151.5145189856826 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 39, + "mode": 0, + "inputs": [ + { + "name": "TRELLIS2PIPELINE", + "type": "TRELLIS2PIPELINE", + "link": 638 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_Trellis2Pipeline", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "Trellis2Pipeline" + }, + "widgets_values": [ + "Trellis2Pipeline" + ] + }, + { + "id": 418, + "type": "SetNode", + "pos": [ + -746.8312372724462, + -1195.6482856970033 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 38, + "mode": 0, + "inputs": [ + { + "name": "TRIMESH", + "type": "TRIMESH", + "link": 637 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_WhiteMesh", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "WhiteMesh" + }, + "widgets_values": [ + "WhiteMesh" + ] + }, + { + "id": 428, + "type": "GetNode", + "pos": [ + -1314.190670687656, + -1187.3587574508188 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 658 + ] + } + ], + "title": "Get_NormalImage", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "NormalImage" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 417, + "type": "GetNode", + "pos": [ + -1310.2904954152943, + -1139.8984090506763 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 657 + ] + } + ], + "title": "Get_Prefix", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 363, + "type": "GetNode", + "pos": [ + -545.6263878998996, + 76.88764947131784 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 566 + ] + } + ], + "title": "Get_Prefix", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 267, + "type": "StringConcatenate", + "pos": [ + -377.11763534068075, + 75.22708582112858 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 27, + "mode": 0, + "inputs": [ + { + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 566 + } + ], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 480 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_Textured_MV", + "" + ] + }, + { + "id": 429, + "type": "GetNode", + "pos": [ + -1277.085676114816, + -573.0082510434125 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "TRELLIS2PIPELINE", + "type": "TRELLIS2PIPELINE", + "links": [ + 1193 + ] + } + ], + "title": "Get_Trellis2Pipeline", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Trellis2Pipeline" + ] + }, + { + "id": 266, + "type": "Preview3D", + "pos": [ + -183.09944061302795, + 64.39555879074847 + ], + "size": [ + 1171.9726831120965, + 1226.3422371338038 + ], + "flags": {}, + "order": 51, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 478 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "Preview3D", + "Last Time Model File": "C:/Git/ComfyUI/output/3D/Tank_4View_Hy20/Tank_4View_Hy20_Textured_MV_00002_.glb", + "Resource Folder": "Git/ComfyUI/output/3D/Tank_4View_Hy20", + "Scene Config": { + "showGrid": true, + "backgroundColor": "#282828", + "backgroundImage": "", + "backgroundRenderMode": "tiled" + }, + "Camera Config": { + "cameraType": "perspective", + "fov": 35, + "state": { + "position": { + "x": 6.116219723662718, + "y": 2.598047886467756, + "z": 8.399990125907564 + }, + "target": { + "x": 0, + "y": 1.61200424353802, + "z": 0 + }, + "zoom": 1, + "cameraType": "perspective" + } + }, + "Light Config": { + "intensity": 5 + }, + "Model Config": { + "upDirection": "original", + "materialMode": "original", + "showSkeleton": false + } + }, + "widgets_values": [ + "C:/Git/ComfyUI/output/3D/Tank_4View_Hy20/Tank_4View_Hy20_Textured_MV_00002_.glb", + "" + ] + }, + { + "id": 259, + "type": "StringConcatenate", + "pos": [ + -1767.7760695000165, + -1014.1318223903585 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 37, + "mode": 0, + "inputs": [ + { + "name": "string_b", + "type": "STRING", + "widget": { + "name": "string_b" + }, + "link": 464 + } + ], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 466 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "3D/", + "", + "" + ] + }, + { + "id": 358, + "type": "SetNode", + "pos": [ + -1588.7674516597126, + -1042.3594981549759 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 53, + "mode": 0, + "inputs": [ + { + "name": "STRING", + "type": "STRING", + "link": 561 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_Prefix", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "Prefix" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 260, + "type": "StringConcatenate", + "pos": [ + -1592.981372429068, + -1002.3323455162318 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 50, + "mode": 0, + "inputs": [ + { + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 466 + }, + { + "name": "string_b", + "type": "STRING", + "widget": { + "name": "string_b" + }, + "link": 465 + } + ], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 561 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "/", + "", + "/" + ] + }, + { + "id": 361, + "type": "SetNode", + "pos": [ + -1752.955634745735, + -780.3113555830523 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 33, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 564 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_TargetFaceNumber", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "TargetFaceNumber" + }, + "widgets_values": [ + "TargetFaceNumber" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 1477, + "type": "SetNode", + "pos": [ + -1725.8334697028015, + -900.73932618393 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 29, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 1694 + } + ], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": null + } + ], + "title": "Set_Seed", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "Seed" + }, + "widgets_values": [ + "Seed" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 436, + "type": "GetNode", + "pos": [ + -1311.7255852250182, + -1044.7680632967733 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 671 + ] + } + ], + "title": "Get_TargetFaceNumber", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "TargetFaceNumber" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 1478, + "type": "GetNode", + "pos": [ + -1307.0333223276834, + -1098.482779160294 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 1695 + ] + } + ], + "title": "Get_Seed", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Seed" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 432, + "type": "GetNode", + "pos": [ + -1310.6835334127045, + -929.1947034503972 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 661 + ] + } + ], + "title": "Get_Prefix", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 433, + "type": "GetNode", + "pos": [ + -1317.8991439004014, + -883.1004178376324 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "TRIMESH", + "type": "TRIMESH", + "links": [ + 663 + ] + } + ], + "title": "Get_WhiteMesh", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "WhiteMesh" + ] + }, + { + "id": 431, + "type": "GetNode", + "pos": [ + -1291.2007962110185, + 75.26010684547481 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "*", + "type": "*", + "links": [ + 660 + ] + } + ], + "title": "Get_TexturedMesh", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "TexturedMesh" + ] + }, + { + "id": 465, + "type": "GetNode", + "pos": [ + -1297.939277518252, + 192.06573820666307 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 745 + ] + } + ], + "title": "Get_BackView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "BackView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 464, + "type": "GetNode", + "pos": [ + -1305.6961669906457, + 124.51320291334166 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 744 + ] + } + ], + "title": "Get_FrontView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "FrontView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 466, + "type": "GetNode", + "pos": [ + -1308.041506018605, + 246.7257610683613 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 11, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1456 + ] + } + ], + "title": "Get_LeftView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "LeftView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 467, + "type": "GetNode", + "pos": [ + -1306.0647206290878, + 319.1279643724372 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 12, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1457 + ] + } + ], + "title": "Get_RightView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "RightView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1476, + "type": "PrimitiveInt", + "pos": [ + -2047.6461070066043, + -929.5690492212133 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 1694 + ] + } + ], + "title": "Seed", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 12345, + "fixed" + ] + }, + { + "id": 1597, + "type": "GetNode", + "pos": [ + -1324.8554794577053, + -1442.423960365636 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 14, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1953 + ] + } + ], + "title": "Get_SourceImage", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "SourceImage" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1598, + "type": "GetNode", + "pos": [ + -1310.500842466565, + -1318.9820466669423 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 15, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 1955 + ] + } + ], + "title": "Get_Seed", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "kijai/ComfyUI-KJNodes" + }, + "widgets_values": [ + "Seed" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 1599, + "type": "GetNode", + "pos": [ + -1312.4383327980986, + -1379.7206754514782 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 16, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 1954 + ] + } + ], + "title": "Get_Prefix", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 1600, + "type": "SetNode", + "pos": [ + -690.0219522319078, + -1367.462466851463 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 41, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 1956 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "title": "Set_NormalImage", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "NormalImage" + }, + "widgets_values": [ + "NormalImage" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1601, + "type": "SetNode", + "pos": [ + -706.4131438977086, + -1428.9185602519058 + ], + "size": [ + 232.4, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 40, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 1957 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "title": "Set_SourceImage_Processed", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "SourceImage_Processed" + }, + "widgets_values": [ + "SourceImage_Processed" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 435, + "type": "GetNode", + "pos": [ + -1329.7904832679303, + -822.2830888092468 + ], + "size": [ + 233.862890625, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 17, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 669 + ] + } + ], + "title": "Get_SourceImage_Processed", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "SourceImage_Processed" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 384, + "type": "GetNode", + "pos": [ + -1294.3135844229482, + -513.6867133233288 + ], + "size": [ + 242.2095703125, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 18, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1972 + ] + } + ], + "title": "Get_SourceImage_Processed", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "SourceImage_Processed" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 388, + "type": "GetNode", + "pos": [ + -1281.730037434195, + -428.48056829609936 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 19, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "TRIMESH", + "type": "TRIMESH", + "links": [ + 1192 + ] + } + ], + "title": "Get_WhiteMesh", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "WhiteMesh" + ] + }, + { + "id": 1481, + "type": "GetNode", + "pos": [ + -1274.4385233484818, + -330.08116352707134 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 20, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 1702 + ] + } + ], + "title": "Get_Seed", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Seed" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 1027, + "type": "SetNode", + "pos": [ + -1583.8192421073943, + -711.6703867862403 + ], + "size": [ + 240.7466796875, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 36, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 2013 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "title": "Set_SourceImage", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "SourceImage" + }, + "widgets_values": [ + "SourceImage" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 209, + "type": "PrimitiveInt", + "pos": [ + -2051.623505404044, + -793.4184503551801 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 21, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 564 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 1025, + "type": "Trellis2MeshTexturing", + "pos": [ + -849.9080887367477, + -575.6484113326054 + ], + "size": [ + 419.15234375, + 506 + ], + "flags": {}, + "order": 32, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 1193 + }, + { + "name": "image", + "type": "IMAGE", + "link": 1972 + }, + { + "name": "trimesh", + "type": "TRIMESH", + "link": 1192 + }, + { + "name": "seed", + "type": "INT", + "widget": { + "name": "seed" + }, + "link": 1702 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 1188, + 1189 + ] + }, + { + "name": "base_color_texture", + "type": "IMAGE", + "links": null + }, + { + "name": "metallic_roughness_texture", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "41a6ddb2d48c408ead711846dbbfe587d4ff7102", + "Node name for S&R": "Trellis2MeshTexturing" + }, + "widgets_values": [ + 12345, + "fixed", + 12, + 3, + 0.01, + 3, + 1024, + 1024, + "OPAQUE", + false, + 0, + 0.9, + 1, + false, + false, + 60, + "heun", + "telea" + ] + }, + { + "id": 1618, + "type": "SetNode", + "pos": [ + -703.7275023648066, + -696.3227770155813 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 47, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 2041 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "title": "Set_BottomView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "kijai/ComfyUI-KJNodes", + "previousName": "BottomView" + }, + "widgets_values": [ + "BottomView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 422, + "type": "SetNode", + "pos": [ + -711.2137626890823, + -933.2452185469294 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 42, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 680 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_FrontView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "FrontView" + }, + "widgets_values": [ + "FrontView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 423, + "type": "SetNode", + "pos": [ + -702.1733471504639, + -885.4573667051945 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 43, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 681 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_LeftView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "LeftView" + }, + "widgets_values": [ + "LeftView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 424, + "type": "SetNode", + "pos": [ + -707.5068883874511, + -840.4019794879641 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 44, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 682 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_BackView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "BackView" + }, + "widgets_values": [ + "BackView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 425, + "type": "SetNode", + "pos": [ + -707.0424164713548, + -793.9808483962397 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 45, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 683 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_RightView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "RightView" + }, + "widgets_values": [ + "RightView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1617, + "type": "SetNode", + "pos": [ + -698.1475594478394, + -749.7298950583769 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 46, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 2040 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + } + ], + "title": "Set_TopView", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "kijai/ComfyUI-KJNodes", + "previousName": "TopView" + }, + "widgets_values": [ + "TopView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1619, + "type": "GetNode", + "pos": [ + -1301.478855605668, + 388.23914525504057 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 22, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2042 + ] + } + ], + "title": "Get_TopView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "kijai/ComfyUI-KJNodes" + }, + "widgets_values": [ + "TopView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 1620, + "type": "GetNode", + "pos": [ + -1295.1019135179424, + 462.37137981688676 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 23, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2043 + ] + } + ], + "title": "Get_BottomView", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "kijai/ComfyUI-KJNodes" + }, + "widgets_values": [ + "BottomView" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 360, + "type": "GetNode", + "pos": [ + -363.8827042196567, + -585.5953164158891 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 24, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 563 + ] + } + ], + "title": "Get_Prefix", + "properties": { + "Node name for S&R": "GetNode", + "aux_id": "GetNode" + }, + "widgets_values": [ + "Prefix" + ] + }, + { + "id": 246, + "type": "StringConcatenate", + "pos": [ + -196.5772097982741, + -581.9708677305606 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 35, + "mode": 0, + "inputs": [ + { + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 563 + } + ], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 447 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_Textured", + "" + ] + }, + { + "id": 242, + "type": "Trellis2ExportMesh", + "pos": [ + -193.40411998539903, + -508.7875293978828 + ], + "size": [ + 270, + 102 + ], + "flags": { + "collapsed": true + }, + "order": 49, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 1188 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 447 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 809 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "Tavern_Textured", + "glb" + ] + }, + { + "id": 482, + "type": "Trellis2Continue", + "pos": [ + -189.39491765604424, + -442.4017886046022 + ], + "size": [ + 161.33359375, + 46 + ], + "flags": { + "collapsed": true + }, + "order": 52, + "mode": 0, + "inputs": [ + { + "name": "input_1", + "type": "*", + "link": 1189 + }, + { + "name": "input_2", + "type": "*", + "link": 809 + } + ], + "outputs": [ + { + "name": "output_1", + "type": "*", + "links": [ + 810 + ] + }, + { + "name": "output_2", + "type": "*", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "b7ae30e26a4c8ab3e73d661668b80a8024465a40", + "Node name for S&R": "Trellis2Continue" + }, + "widgets_values": [] + }, + { + "id": 389, + "type": "SetNode", + "pos": [ + 25.12412753026314, + -441.9412931629978 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 54, + "mode": 0, + "inputs": [ + { + "name": "*", + "type": "*", + "link": 810 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_TexturedMesh", + "properties": { + "Node name for S&R": "SetNode", + "aux_id": "SetNode", + "previousName": "TexturedMesh" + }, + "widgets_values": [ + "TexturedMesh" + ] + }, + { + "id": 265, + "type": "Trellis2ExportMesh", + "pos": [ + -531.7753807563762, + 135.57688531430873 + ], + "size": [ + 270, + 102 + ], + "flags": { + "collapsed": true + }, + "order": 48, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 476 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 480 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 478 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "Test_v2", + "glb" + ] + }, + { + "id": 264, + "type": "Trellis2MultiViewTexturing", + "pos": [ + -974.3758998631852, + 75.46500664790204 + ], + "size": [ + 364.6066617624932, + 633.3109375310778 + ], + "flags": {}, + "order": 34, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 660 + }, + { + "name": "front_image", + "shape": 7, + "type": "IMAGE", + "link": 744 + }, + { + "name": "back_image", + "shape": 7, + "type": "IMAGE", + "link": 745 + }, + { + "name": "left_image", + "shape": 7, + "type": "IMAGE", + "link": 1456 + }, + { + "name": "right_image", + "shape": 7, + "type": "IMAGE", + "link": 1457 + }, + { + "name": "top_image", + "shape": 7, + "type": "IMAGE", + "link": 2042 + }, + { + "name": "bottom_image", + "shape": 7, + "type": "IMAGE", + "link": 2043 + }, + { + "name": "custom_images", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "camera_config", + "shape": 7, + "type": "HY3DCAMERA", + "link": null + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 476 + ] + }, + { + "name": "base_color", + "type": "IMAGE", + "links": null + }, + { + "name": "metallic_roughness", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2MultiViewTexturing" + }, + "widgets_values": [ + 1024, + true, + 1, + 1.2, + 1.15, + true, + 20, + false, + 0.01, + 1, + 1, + 0.5, + 0.5, + 0.1, + 0.1, + "", + "", + "" + ] + }, + { + "id": 416, + "type": "a7c4d8ae-5500-4f9b-9601-eafdb97e7a68", + "pos": [ + -1058.861926280772, + -1190.3382989836646 + ], + "size": [ + 210, + 98 + ], + "flags": {}, + "order": 28, + "mode": 0, + "inputs": [ + { + "label": "Normal_Image", + "name": "image", + "type": "IMAGE", + "link": 658 + }, + { + "label": "Prefix", + "name": "", + "type": "*", + "link": 657 + }, + { + "name": "target_face_num", + "type": "INT", + "link": 671 + }, + { + "label": "seed", + "name": "_1", + "type": "*", + "link": 1695 + } + ], + "outputs": [ + { + "label": "white_mesh", + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 637 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 638 + ] + } + ], + "properties": { + "proxyWidgets": [], + "cnr_id": "comfy-core", + "ver": "0.16.4" + }, + "widgets_values": [] + }, + { + "id": 239, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -2051.601088773679, + -652.112967692624 + ], + "size": [ + 627.638044195148, + 672.4931923313359 + ], + "flags": {}, + "order": 25, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 2013 + ] + }, + { + "name": "mask", + "type": "MASK", + "links": null + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [] + } + ], + "title": "Color Image", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2LoadImageWithTransparency" + }, + "widgets_values": [ + "stylized_ogre_tavern_3d_game_asset_gray (2).jpeg", + "image" + ] + }, + { + "id": 219, + "type": "PrimitiveString", + "pos": [ + -2058.146657853287, + -1035.357925598092 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 26, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 464, + 465 + ] + } + ], + "title": "Name -> output in 3D folder", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "OgreTavern_6View_Hy20" + ] + }, + { + "id": 1596, + "type": "e62a2ed8-84aa-4855-9c23-e444bcf27791", + "pos": [ + -1029.0444894493137, + -1435.6564661781129 + ], + "size": [ + 274.330078125, + 66 + ], + "flags": {}, + "order": 30, + "mode": 0, + "inputs": [ + { + "label": "source_image", + "name": "image", + "type": "IMAGE", + "link": 1953 + }, + { + "label": "prefix", + "name": "", + "type": "*", + "link": 1954 + }, + { + "label": "seed", + "name": "_1", + "type": "*", + "link": 1955 + } + ], + "outputs": [ + { + "label": "source_image", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1957 + ] + }, + { + "label": "normal_image_transparent", + "name": "IMAGE_1", + "type": "IMAGE", + "links": [ + 1956 + ] + } + ], + "properties": { + "proxyWidgets": [], + "cnr_id": "comfy-core", + "ver": "0.18.1" + }, + "widgets_values": [] + }, + { + "id": 420, + "type": "bf1edf2f-ccb7-45ac-882e-55bdd5f0be86", + "pos": [ + -1079.3175301918616, + -938.6000286674437 + ], + "size": [ + 210, + 148 + ], + "flags": {}, + "order": 31, + "mode": 0, + "inputs": [ + { + "label": "prefix", + "name": "", + "type": "*", + "link": 661 + }, + { + "label": "white_mesh", + "name": "trimesh", + "type": "TRIMESH", + "link": 663 + }, + { + "label": "color_image", + "name": "_1", + "type": "*", + "link": 669 + } + ], + "outputs": [ + { + "label": "FrontView", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 680 + ] + }, + { + "label": "LeftView", + "name": "IMAGE_1", + "type": "IMAGE", + "links": [ + 681 + ] + }, + { + "label": "BackView", + "name": "IMAGE_2", + "type": "IMAGE", + "links": [ + 682 + ] + }, + { + "label": "RightView", + "name": "IMAGE_3", + "type": "IMAGE", + "links": [ + 683 + ] + }, + { + "label": "TopView", + "name": "IMAGE_4", + "type": "IMAGE", + "links": [ + 2040 + ] + }, + { + "label": "BottomView", + "name": "IMAGE_5", + "type": "IMAGE", + "links": [ + 2041 + ] + } + ], + "properties": { + "proxyWidgets": [], + "cnr_id": "comfy-core", + "ver": "0.16.4" + }, + "widgets_values": [] + } + ], + "links": [ + [ + 447, + 246, + 0, + 242, + 1, + "STRING" + ], + [ + 464, + 219, + 0, + 259, + 0, + "STRING" + ], + [ + 465, + 219, + 0, + 260, + 1, + "STRING" + ], + [ + 466, + 259, + 0, + 260, + 0, + "STRING" + ], + [ + 476, + 264, + 0, + 265, + 0, + "TRIMESH" + ], + [ + 478, + 265, + 0, + 266, + 2, + "STRING" + ], + [ + 480, + 267, + 0, + 265, + 1, + "STRING" + ], + [ + 561, + 260, + 0, + 358, + 0, + "STRING" + ], + [ + 563, + 360, + 0, + 246, + 0, + "STRING" + ], + [ + 564, + 209, + 0, + 361, + 0, + "INT" + ], + [ + 566, + 363, + 0, + 267, + 0, + "STRING" + ], + [ + 637, + 416, + 0, + 418, + 0, + "TRIMESH" + ], + [ + 638, + 416, + 1, + 419, + 0, + "TRELLIS2PIPELINE" + ], + [ + 657, + 417, + 0, + 416, + 1, + "STRING" + ], + [ + 658, + 428, + 0, + 416, + 0, + "IMAGE" + ], + [ + 660, + 431, + 0, + 264, + 0, + "TRIMESH" + ], + [ + 661, + 432, + 0, + 420, + 0, + "STRING" + ], + [ + 663, + 433, + 0, + 420, + 1, + "TRIMESH" + ], + [ + 669, + 435, + 0, + 420, + 2, + "IMAGE" + ], + [ + 671, + 436, + 0, + 416, + 2, + "INT" + ], + [ + 680, + 420, + 0, + 422, + 0, + "IMAGE" + ], + [ + 681, + 420, + 1, + 423, + 0, + "IMAGE" + ], + [ + 682, + 420, + 2, + 424, + 0, + "IMAGE" + ], + [ + 683, + 420, + 3, + 425, + 0, + "IMAGE" + ], + [ + 744, + 464, + 0, + 264, + 1, + "IMAGE" + ], + [ + 745, + 465, + 0, + 264, + 2, + "IMAGE" + ], + [ + 809, + 242, + 0, + 482, + 1, + "STRING" + ], + [ + 810, + 482, + 0, + 389, + 0, + "TRIMESH" + ], + [ + 1188, + 1025, + 0, + 242, + 0, + "TRIMESH" + ], + [ + 1189, + 1025, + 0, + 482, + 0, + "TRIMESH" + ], + [ + 1192, + 388, + 0, + 1025, + 2, + "TRIMESH" + ], + [ + 1193, + 429, + 0, + 1025, + 0, + "TRELLIS2PIPELINE" + ], + [ + 1456, + 466, + 0, + 264, + 3, + "IMAGE" + ], + [ + 1457, + 467, + 0, + 264, + 4, + "IMAGE" + ], + [ + 1694, + 1476, + 0, + 1477, + 0, + "INT" + ], + [ + 1695, + 1478, + 0, + 416, + 3, + "INT" + ], + [ + 1702, + 1481, + 0, + 1025, + 3, + "INT" + ], + [ + 1953, + 1597, + 0, + 1596, + 0, + "IMAGE" + ], + [ + 1954, + 1599, + 0, + 1596, + 1, + "STRING" + ], + [ + 1955, + 1598, + 0, + 1596, + 2, + "INT" + ], + [ + 1956, + 1596, + 1, + 1600, + 0, + "IMAGE" + ], + [ + 1957, + 1596, + 0, + 1601, + 0, + "IMAGE" + ], + [ + 1972, + 384, + 0, + 1025, + 1, + "IMAGE" + ], + [ + 2013, + 239, + 0, + 1027, + 0, + "IMAGE" + ], + [ + 2040, + 420, + 4, + 1617, + 0, + "IMAGE" + ], + [ + 2041, + 420, + 5, + 1618, + 0, + "IMAGE" + ], + [ + 2042, + 1619, + 0, + 264, + 5, + "IMAGE" + ], + [ + 2043, + 1620, + 0, + 264, + 6, + "IMAGE" + ] + ], + "groups": [ + { + "id": 1, + "title": "Texturing", + "bounding": [ + -1331.067657728365, + -673.7818277278924, + 1553.6673346627758, + 624.0973323799652 + ], + "color": "#A88", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Projection", + "bounding": [ + -1330.3992143893013, + -23.42961664340983, + 2426.370430804953, + 1351.1024465541664 + ], + "color": "#b58b2a", + "font_size": 24, + "flags": {} + }, + { + "id": 5, + "title": "Configuration", + "bounding": [ + -2068.3222779518146, + -1128.221840574841, + 669.1786918496364, + 1186.8963044768402 + ], + "color": "#b06634", + "font_size": 24, + "flags": {} + }, + { + "id": 8, + "title": "Mesh Generation", + "bounding": [ + -1334.1541087377145, + -1274.931709235975, + 783.6469210441065, + 249.0021636156074 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 9, + "title": "MultiView Generation with Hy2.0", + "bounding": [ + -1335.908517983496, + -1011.8851719087663, + 838.352169354825, + 319.25838307895276 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 32, + "title": "Normal Generation", + "bounding": [ + -1334.8554794577053, + -1521.5982778096845, + 881.6773546690334, + 226.160224658576 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "definitions": { + "subgraphs": [ + { + "id": "a7c4d8ae-5500-4f9b-9601-eafdb97e7a68", + "version": 1, + "state": { + "lastGroupId": 32, + "lastNodeId": 1620, + "lastLinkId": 2048, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "Mesh Generation", + "inputNode": { + "id": -10, + "bounding": [ + -1429.953594419078, + -690.030564252835, + 134.212890625, + 120 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 423.7021994103568, + -690.030564252835, + 120, + 80 + ] + }, + "inputs": [ + { + "id": "8f27facf-c8f6-47f3-804e-db19f864d21f", + "name": "image", + "type": "IMAGE", + "linkIds": [ + 631 + ], + "label": "Normal_Image", + "pos": [ + -1315.740703794078, + -670.030564252835 + ] + }, + { + "id": "ea5d96ce-c832-43b3-bf6b-63a09c722abe", + "name": "", + "type": "*", + "linkIds": [ + 655 + ], + "label": "Prefix", + "pos": [ + -1315.740703794078, + -650.030564252835 + ] + }, + { + "id": "ae30eada-3ab4-491d-b92a-14a0627aaa34", + "name": "target_face_num", + "type": "INT", + "linkIds": [ + 1964 + ], + "pos": [ + -1315.740703794078, + -630.030564252835 + ] + }, + { + "id": "f87e02a8-d1a3-48cd-b57c-5a1fa2c5d563", + "name": "_1", + "type": "*", + "linkIds": [ + 1692 + ], + "label": "seed", + "pos": [ + -1315.740703794078, + -610.030564252835 + ] + } + ], + "outputs": [ + { + "id": "0f9ae109-484c-495b-9e82-7f454a26b6d5", + "name": "trimesh", + "type": "TRIMESH", + "linkIds": [ + 750 + ], + "label": "white_mesh", + "pos": [ + 443.7021994103568, + -670.030564252835 + ] + }, + { + "id": "d2aad993-ce3c-4f5e-a2ba-e1ce7bef92d3", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "linkIds": [ + 635 + ], + "pos": [ + 443.7021994103568, + -650.030564252835 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 397, + "type": "Trellis2ImageCondGenerator", + "pos": [ + -903.9659631794001, + -978.2409763036111 + ], + "size": [ + 308.0748046875, + 98 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 602 + }, + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 603 + } + ], + "outputs": [ + { + "localized_name": "cond_512", + "name": "cond_512", + "type": "IMAGE_COND", + "links": [ + 605, + 607 + ] + }, + { + "localized_name": "cond_1024", + "name": "cond_1024", + "type": "IMAGE_COND", + "links": [ + 610 + ] + }, + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 604 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ImageCondGenerator" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 401, + "type": "Trellis2DecodeLatents", + "pos": [ + -444.17325828513566, + -415.65865823859036 + ], + "size": [ + 270, + 122 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 614 + }, + { + "localized_name": "shape_slat", + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 615 + }, + { + "localized_name": "texture_slat", + "name": "texture_slat", + "shape": 7, + "type": "TEXTURE_SLAT", + "link": null + }, + { + "localized_name": "resolution", + "name": "resolution", + "type": "INT", + "widget": { + "name": "resolution" + }, + "link": 616 + } + ], + "outputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 618 + ] + }, + { + "localized_name": "bvh", + "name": "bvh", + "type": "BVH", + "links": [] + }, + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2DecodeLatents" + }, + "widgets_values": [ + 0, + true + ] + }, + { + "id": 410, + "type": "Trellis2PreProcessImage", + "pos": [ + -1249.953594419078, + -751.3094691431603 + ], + "size": [ + 281.7837890625, + 106 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 631 + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 603 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "7ce26deae425d114f3deb78e802d6196e5987a4a", + "Node name for S&R": "Trellis2PreProcessImage" + }, + "widgets_values": [ + 5, + false, + 1024 + ] + }, + { + "id": 1606, + "type": "Reroute", + "pos": [ + -361.2137915568719, + -976.5651884005764 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 1964 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 2046 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 404, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + -1.3688675741572178, + -1197.2135353903182 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 619 + } + ], + "outputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 1966 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 396, + "type": "Trellis2LoadModel", + "pos": [ + -1248.8098695647218, + -1061.4257138783487 + ], + "size": [ + 280.05208333333337, + 202 + ], + "flags": { + "collapsed": false + }, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 602, + 635 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" + ] + }, + { + "id": 398, + "type": "Trellis2SparseGenerator", + "pos": [ + -909.3031482247874, + -824.5435537779637 + ], + "size": [ + 395.65625, + 314 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 604 + }, + { + "localized_name": "image_cond", + "name": "image_cond", + "type": "IMAGE_COND", + "link": 605 + }, + { + "localized_name": "seed", + "name": "seed", + "type": "INT", + "widget": { + "name": "seed" + }, + "link": 1693 + } + ], + "outputs": [ + { + "localized_name": "coords", + "name": "coords", + "type": "COORDS", + "links": [ + 608 + ] + }, + { + "localized_name": "sparse_structure_resolution", + "name": "sparse_structure_resolution", + "type": "INT", + "links": [ + 613 + ] + }, + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 606 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2SparseGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + 25, + 7.5, + 0.01, + 5, + "euler", + 32, + 0.1, + 1 + ] + }, + { + "id": 399, + "type": "Trellis2ShapeGenerator", + "pos": [ + -892.2604727135389, + -441.57534731890115 + ], + "size": [ + 335.24609375, + 266 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 606 + }, + { + "localized_name": "image_cond", + "name": "image_cond", + "type": "IMAGE_COND", + "link": 607 + }, + { + "localized_name": "coords", + "name": "coords", + "type": "COORDS", + "link": 608 + } + ], + "outputs": [ + { + "localized_name": "shape_slat", + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 611 + ] + }, + { + "localized_name": "resolution", + "name": "resolution", + "type": "INT", + "links": [ + 612 + ] + }, + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 609 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeGenerator" + }, + "widgets_values": [ + 512, + 25, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 400, + "type": "Trellis2ShapeCascadeGenerator", + "pos": [ + -465.2443600409803, + -820.2405791353129 + ], + "size": [ + 335.9791015625, + 338 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 609 + }, + { + "localized_name": "image_cond", + "name": "image_cond", + "type": "IMAGE_COND", + "link": 610 + }, + { + "localized_name": "shape_slat", + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 611 + }, + { + "localized_name": "from_resolution", + "name": "from_resolution", + "type": "INT", + "widget": { + "name": "from_resolution" + }, + "link": 612 + }, + { + "localized_name": "sparse_structure_resolution", + "name": "sparse_structure_resolution", + "type": "INT", + "widget": { + "name": "sparse_structure_resolution" + }, + "link": 613 + } + ], + "outputs": [ + { + "localized_name": "shape_slat", + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 615 + ] + }, + { + "localized_name": "resolution", + "name": "resolution", + "type": "INT", + "links": [ + 616 + ] + }, + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 614 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeCascadeGenerator" + }, + "widgets_values": [ + 0, + 1024, + 0, + 999999, + 12, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 414, + "type": "StringConcatenate", + "pos": [ + -879.1728621391709, + -81.2258234995071 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 12, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 656 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 629 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_WhiteMesh", + "" + ] + }, + { + "id": 406, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 50.28786765744911, + -792.0290399873467 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 9, + "mode": 4, + "inputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 2047 + } + ], + "outputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 1971 + ] + }, + { + "localized_name": "holes_filled", + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 1605, + "type": "Trellis2SimplifyMesh", + "pos": [ + 44.610272963955815, + -972.4658282598242 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 1966 + }, + { + "localized_name": "target_face_num", + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 2046 + } + ], + "outputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 2047 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 402, + "type": "Trellis2MeshWithVoxelToTrimesh", + "pos": [ + -19.274159374003318, + -597.7545005308027 + ], + "size": [ + 349.41171875, + 58 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 1971 + } + ], + "outputs": [ + { + "localized_name": "trimesh", + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 2044, + 2045 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + }, + "widgets_values": [ + "90 degrees" + ] + }, + { + "id": 468, + "type": "Trellis2Continue", + "pos": [ + 242.36320497352165, + -327.9179781939218 + ], + "size": [ + 161.33359375, + 46 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "localized_name": "input_1", + "name": "input_1", + "type": "*", + "link": 2045 + }, + { + "localized_name": "input_2", + "name": "input_2", + "type": "*", + "link": 749 + } + ], + "outputs": [ + { + "localized_name": "output_1", + "name": "output_1", + "type": "*", + "links": [ + 750 + ] + }, + { + "localized_name": "output_2", + "name": "output_2", + "type": "*", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "6c3771938ad833fd840e489b02c839cd3b018e9f", + "Node name for S&R": "Trellis2Continue" + }, + "widgets_values": [] + }, + { + "id": 409, + "type": "Trellis2ExportMesh", + "pos": [ + -85.89777477147005, + -241.56574480172577 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "localized_name": "trimesh", + "name": "trimesh", + "type": "TRIMESH", + "link": 2044 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 629 + } + ], + "outputs": [ + { + "localized_name": "glb_path", + "name": "glb_path", + "type": "STRING", + "links": [ + 749 + ] + }, + { + "localized_name": "relative_path", + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "LaserPistol", + "glb" + ] + }, + { + "id": 1475, + "type": "Reroute", + "pos": [ + -1203.4611355198667, + -616.859229905566 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 1692 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 1693 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 427, + "type": "Reroute", + "pos": [ + -1209.3136872322395, + -527.7593742715582 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 655 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 656 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 403, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 11.534077258920945, + -1321.6285802685652 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 618 + } + ], + "outputs": [ + { + "localized_name": "mesh", + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 619 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + } + ], + "groups": [], + "links": [ + { + "id": 602, + "origin_id": 396, + "origin_slot": 0, + "target_id": 397, + "target_slot": 0, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 603, + "origin_id": 410, + "origin_slot": 0, + "target_id": 397, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 604, + "origin_id": 397, + "origin_slot": 2, + "target_id": 398, + "target_slot": 0, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 605, + "origin_id": 397, + "origin_slot": 0, + "target_id": 398, + "target_slot": 1, + "type": "IMAGE_COND" + }, + { + "id": 606, + "origin_id": 398, + "origin_slot": 2, + "target_id": 399, + "target_slot": 0, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 607, + "origin_id": 397, + "origin_slot": 0, + "target_id": 399, + "target_slot": 1, + "type": "IMAGE_COND" + }, + { + "id": 608, + "origin_id": 398, + "origin_slot": 0, + "target_id": 399, + "target_slot": 2, + "type": "COORDS" + }, + { + "id": 609, + "origin_id": 399, + "origin_slot": 2, + "target_id": 400, + "target_slot": 0, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 610, + "origin_id": 397, + "origin_slot": 1, + "target_id": 400, + "target_slot": 1, + "type": "IMAGE_COND" + }, + { + "id": 611, + "origin_id": 399, + "origin_slot": 0, + "target_id": 400, + "target_slot": 2, + "type": "SHAPE_SLAT" + }, + { + "id": 612, + "origin_id": 399, + "origin_slot": 1, + "target_id": 400, + "target_slot": 3, + "type": "INT" + }, + { + "id": 613, + "origin_id": 398, + "origin_slot": 1, + "target_id": 400, + "target_slot": 4, + "type": "INT" + }, + { + "id": 614, + "origin_id": 400, + "origin_slot": 2, + "target_id": 401, + "target_slot": 0, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 615, + "origin_id": 400, + "origin_slot": 0, + "target_id": 401, + "target_slot": 1, + "type": "SHAPE_SLAT" + }, + { + "id": 616, + "origin_id": 400, + "origin_slot": 1, + "target_id": 401, + "target_slot": 3, + "type": "INT" + }, + { + "id": 618, + "origin_id": 401, + "origin_slot": 0, + "target_id": 403, + "target_slot": 0, + "type": "MESHWITHVOXEL" + }, + { + "id": 619, + "origin_id": 403, + "origin_slot": 0, + "target_id": 404, + "target_slot": 0, + "type": "MESHWITHVOXEL" + }, + { + "id": 629, + "origin_id": 414, + "origin_slot": 0, + "target_id": 409, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 631, + "origin_id": -10, + "origin_slot": 0, + "target_id": 410, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 635, + "origin_id": 396, + "origin_slot": 0, + "target_id": -20, + "target_slot": 1, + "type": "TRELLIS2PIPELINE" + }, + { + "id": 655, + "origin_id": -10, + "origin_slot": 1, + "target_id": 427, + "target_slot": 0, + "type": "*" + }, + { + "id": 656, + "origin_id": 427, + "origin_slot": 0, + "target_id": 414, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 749, + "origin_id": 409, + "origin_slot": 0, + "target_id": 468, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 750, + "origin_id": 468, + "origin_slot": 0, + "target_id": -20, + "target_slot": 0, + "type": "*" + }, + { + "id": 1692, + "origin_id": -10, + "origin_slot": 3, + "target_id": 1475, + "target_slot": 0, + "type": "*" + }, + { + "id": 1693, + "origin_id": 1475, + "origin_slot": 0, + "target_id": 398, + "target_slot": 2, + "type": "INT" + }, + { + "id": 1964, + "origin_id": -10, + "origin_slot": 2, + "target_id": 1606, + "target_slot": 0, + "type": "*" + }, + { + "id": 1966, + "origin_id": 404, + "origin_slot": 0, + "target_id": 1605, + "target_slot": 0, + "type": "MESHWITHVOXEL" + }, + { + "id": 1971, + "origin_id": 406, + "origin_slot": 0, + "target_id": 402, + "target_slot": 0, + "type": "MESHWITHVOXEL" + }, + { + "id": 2044, + "origin_id": 402, + "origin_slot": 0, + "target_id": 409, + "target_slot": 0, + "type": "TRIMESH" + }, + { + "id": 2045, + "origin_id": 402, + "origin_slot": 0, + "target_id": 468, + "target_slot": 0, + "type": "TRIMESH" + }, + { + "id": 2046, + "origin_id": 1606, + "origin_slot": 0, + "target_id": 1605, + "target_slot": 1, + "type": "INT" + }, + { + "id": 2047, + "origin_id": 1605, + "origin_slot": 0, + "target_id": 406, + "target_slot": 0, + "type": "MESHWITHVOXEL" + } + ], + "extra": { + "workflowRendererVersion": "LG" + } + }, + { + "id": "bf1edf2f-ccb7-45ac-882e-55bdd5f0be86", + "version": 1, + "state": { + "lastGroupId": 32, + "lastNodeId": 1620, + "lastLinkId": 2048, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "MultiView Generation", + "inputNode": { + "id": -10, + "bounding": [ + -1746.4835931485231, + -470.3084268847315, + 120, + 100 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 820.8030880744058, + -593.0368552327766, + 120, + 160 + ] + }, + "inputs": [ + { + "id": "e34c41cc-50b6-4907-9f86-5027585da336", + "name": "", + "type": "*", + "linkIds": [ + 649 + ], + "label": "prefix", + "pos": [ + -1646.4835931485231, + -450.3084268847315 + ] + }, + { + "id": "9e8d195b-2c4c-4818-ab9a-1c87355b8e8f", + "name": "trimesh", + "type": "TRIMESH", + "linkIds": [ + 662 + ], + "label": "white_mesh", + "pos": [ + -1646.4835931485231, + -430.3084268847315 + ] + }, + { + "id": "ccdf7db0-19ec-4036-b246-e0b04038529e", + "name": "_1", + "type": "*", + "linkIds": [ + 1505 + ], + "label": "color_image", + "pos": [ + -1646.4835931485231, + -410.3084268847315 + ] + } + ], + "outputs": [ + { + "id": "9a3500ea-af9b-4260-a413-5960a0ed36d9", + "name": "IMAGE", + "type": "IMAGE", + "linkIds": [ + 2034 + ], + "label": "FrontView", + "pos": [ + 840.8030880744058, + -573.0368552327766 + ] + }, + { + "id": "844113bb-886c-47a9-bf6a-3dc09a5e800e", + "name": "IMAGE_1", + "type": "IMAGE", + "linkIds": [ + 2035 + ], + "label": "LeftView", + "pos": [ + 840.8030880744058, + -553.0368552327766 + ] + }, + { + "id": "fb5df4c8-d61e-48f5-b282-2e2da64f7a2e", + "name": "IMAGE_2", + "type": "IMAGE", + "linkIds": [ + 2036 + ], + "label": "BackView", + "pos": [ + 840.8030880744058, + -533.0368552327766 + ] + }, + { + "id": "26886f31-ca69-429e-8829-abce39d19d0a", + "name": "IMAGE_3", + "type": "IMAGE", + "linkIds": [ + 2037 + ], + "label": "RightView", + "pos": [ + 840.8030880744058, + -513.0368552327766 + ] + }, + { + "id": "f9ce56a7-2b8e-4eea-b2b6-0dbf9ff9f427", + "name": "IMAGE_4", + "type": "IMAGE", + "linkIds": [ + 2038 + ], + "label": "TopView", + "pos": [ + 840.8030880744058, + -493.0368552327766 + ] + }, + { + "id": "0aa5299d-43ac-40ee-ad48-b4cf12470877", + "name": "IMAGE_5", + "type": "IMAGE", + "linkIds": [ + 2039 + ], + "label": "BottomView", + "pos": [ + 840.8030880744058, + -473.0368552327766 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 1328, + "type": "Hy3DDiffusersSchedulerConfig", + "pos": [ + -862.8999634654654, + -581.3271020103193 + ], + "size": [ + 390.5999755859375, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "HY3DDIFFUSERSPIPE", + "link": 1502 + } + ], + "outputs": [ + { + "localized_name": "diffusers_scheduler", + "name": "diffusers_scheduler", + "type": "NOISESCHEDULER", + "slot_index": 0, + "links": [ + 1503 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper", + "ver": "6090a9051f109d2774d309cf6a8a22d4a11a1b15", + "Node name for S&R": "Hy3DDiffusersSchedulerConfig", + "widget_ue_connectable": { + "scheduler": true, + "sigmas": true + }, + "cnr_id": "comfyui-hunyan3dwrapper" + }, + "widgets_values": [ + "Euler", + "default" + ] + }, + { + "id": 1327, + "type": "DownloadAndLoadHy3DPaintModel", + "pos": [ + -1274.0598329076045, + -714.5061282030463 + ], + "size": [ + 327.5999755859375, + 58 + ], + "flags": { + "collapsed": false + }, + "order": 0, + "mode": 0, + "inputs": [ + { + "localized_name": "compile_args", + "name": "compile_args", + "shape": 7, + "type": "HY3DCOMPILEARGS", + "link": null + } + ], + "outputs": [ + { + "localized_name": "multiview_pipe", + "name": "multiview_pipe", + "type": "HY3DDIFFUSERSPIPE", + "slot_index": 0, + "links": [ + 1502, + 1504 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper", + "ver": "6090a9051f109d2774d309cf6a8a22d4a11a1b15", + "Node name for S&R": "DownloadAndLoadHy3DPaintModel", + "widget_ue_connectable": { + "model": true + }, + "cnr_id": "comfyui-hunyan3dwrapper" + }, + "widgets_values": [ + "hunyuan3d-paint-v2-0" + ] + }, + { + "id": 296, + "type": "Hy3DRenderMultiView", + "pos": [ + -1239.8610699824203, + -428.39700113665066 + ], + "size": [ + 303.787109375, + 234 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "trimesh", + "name": "trimesh", + "type": "TRIMESH", + "link": 662 + }, + { + "localized_name": "camera_config", + "name": "camera_config", + "shape": 7, + "type": "HY3DCAMERA", + "link": 546 + } + ], + "outputs": [ + { + "localized_name": "normal_maps", + "name": "normal_maps", + "type": "IMAGE", + "links": [ + 1507 + ] + }, + { + "localized_name": "position_maps", + "name": "position_maps", + "type": "IMAGE", + "links": [ + 1508 + ] + }, + { + "localized_name": "renderer", + "name": "renderer", + "type": "MESHRENDER", + "links": null + }, + { + "localized_name": "masks", + "name": "masks", + "type": "MASK", + "links": [] + }, + { + "localized_name": "textured_maps", + "name": "textured_maps", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper", + "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d", + "Node name for S&R": "Hy3DRenderMultiView" + }, + "widgets_values": [ + 1024, + 1024, + false, + "", + "world" + ] + }, + { + "id": 1035, + "type": "StringConcatenate", + "pos": [ + -990.4672803385581, + 51.17410543902827 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 14, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 1204 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1873 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_Ortho", + "" + ] + }, + { + "id": 298, + "type": "Hy3DCameraConfig", + "pos": [ + -1542.2009936232482, + -594.7805122274469 + ], + "size": [ + 270, + 154 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "camera_config", + "name": "camera_config", + "type": "HY3DCAMERA", + "links": [ + 546, + 1506 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper", + "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d", + "Node name for S&R": "Hy3DCameraConfig" + }, + "widgets_values": [ + "0, 90, 180, 270, 0, 0", + "0, 0, 0, 0, 90, -90", + "1,1,1,1,1,1", + 1.1, + 1.2 + ] + }, + { + "id": 299, + "type": "VHS_SelectImages", + "pos": [ + -435.22855283205126, + -757.9872683426315 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1510 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1499, + 1514, + 2021 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "1", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 318, + "type": "VHS_SelectImages", + "pos": [ + -435.6953598313359, + -574.8401846327595 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1511 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1500, + 1518, + 2022 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "2", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 1543, + "type": "Trellis2SaveImage", + "pos": [ + -543.7388509308294, + 95.01661567145857 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 19, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1872 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1873 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 355, + "type": "VHS_SelectImages", + "pos": [ + -430.8103719299532, + -406.66851210910005 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1512 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1501, + 1519, + 2023 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "3", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 1326, + "type": "Hy3DSampleMultiView", + "pos": [ + -883.0149101311823, + -469.245600284044 + ], + "size": [ + 270, + 274 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "localized_name": "pipeline", + "name": "pipeline", + "type": "HY3DDIFFUSERSPIPE", + "link": 1504 + }, + { + "localized_name": "ref_image", + "name": "ref_image", + "type": "IMAGE", + "link": 1505 + }, + { + "localized_name": "normal_maps", + "name": "normal_maps", + "type": "IMAGE", + "link": 1507 + }, + { + "localized_name": "position_maps", + "name": "position_maps", + "type": "IMAGE", + "link": 1508 + }, + { + "localized_name": "camera_config", + "name": "camera_config", + "shape": 7, + "type": "HY3DCAMERA", + "link": 1506 + }, + { + "localized_name": "scheduler", + "name": "scheduler", + "shape": 7, + "type": "NOISESCHEDULER", + "link": 1503 + }, + { + "localized_name": "samples", + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 1509, + 1510, + 1511, + 1512, + 1872, + 2024, + 2025 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Hunyuan3DWrapper", + "ver": "6e5e9710e5f2f4e37cb214e871c923a9d8db423d", + "Node name for S&R": "Hy3DSampleMultiView" + }, + "widgets_values": [ + 1024, + 12, + 12345, + "fixed", + 1 + ] + }, + { + "id": 426, + "type": "Reroute", + "pos": [ + -1496.3532269059936, + -389.05347596216643 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 649 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 650, + 651, + 652, + 653, + 1204, + 1491, + 1492, + 1493, + 1515, + 1516, + 1517, + 2028, + 2031 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 297, + "type": "VHS_SelectImages", + "pos": [ + -436.1636888919487, + -938.3479012181282 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1509 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2020 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "0", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 1610, + "type": "VHS_SelectImages", + "pos": [ + -424.2401697968086, + -67.4260309811222 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 25, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 2025 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2027 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "5", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 1609, + "type": "VHS_SelectImages", + "pos": [ + -435.6599296933568, + -241.75265413133576 + ], + "size": [ + 212.5712890625, + 106 + ], + "flags": {}, + "order": 24, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 2024 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2026 + ] + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "993082e4f2473bf4acaf06f51e33877a7eb38960", + "Node name for S&R": "VHS_SelectImages" + }, + "widgets_values": { + "indexes": "4", + "err_if_missing": true, + "err_if_empty": true + } + }, + { + "id": 364, + "type": "StringConcatenate", + "pos": [ + 35.77617179788515, + -846.0155885271691 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 8, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 650 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1875 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_FrontView", + "" + ] + }, + { + "id": 367, + "type": "StringConcatenate", + "pos": [ + 34.080342327046154, + -666.1024115080415 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 9, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 651 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1877 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_LeftView", + "" + ] + }, + { + "id": 371, + "type": "StringConcatenate", + "pos": [ + 35.71309824482861, + -503.67579172607446 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 10, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 652 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1879 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_BackView", + "" + ] + }, + { + "id": 372, + "type": "StringConcatenate", + "pos": [ + 40.46595031847387, + -359.967454329042 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 11, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 653 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1881 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_RightView", + "" + ] + }, + { + "id": 1613, + "type": "StringConcatenate", + "pos": [ + 42.9445878581309, + -219.80206489015328 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 28, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 2028 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 2030 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_TopView", + "" + ] + }, + { + "id": 1615, + "type": "StringConcatenate", + "pos": [ + 35.09400401574455, + -19.439852278110624 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": true + }, + "order": 30, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 2031 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 2033 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.16.4", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_BottomView", + "" + ] + }, + { + "id": 494, + "type": "RMBG", + "pos": [ + -19.26089667273442, + -908.2866484721291 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 13, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2020 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1874, + 2034 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1500, + "type": "RMBG", + "pos": [ + -15.770746233324541, + -722.8726246922832 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 16, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2021 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1876, + 2035 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1501, + "type": "RMBG", + "pos": [ + -11.102746066987152, + -556.9884037168202 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 17, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2022 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1878, + 2036 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1502, + "type": "RMBG", + "pos": [ + -5.297787896089057, + -425.97143469459684 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 18, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2023 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1880, + 2037 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1611, + "type": "RMBG", + "pos": [ + 0.23025981432448983, + -277.57154720273326 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 26, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2026 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2029, + 2038 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1612, + "type": "RMBG", + "pos": [ + -9.806598738150342, + -89.90771460473213 + ], + "size": [ + 320, + 320.546875 + ], + "flags": { + "collapsed": true + }, + "order": 27, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 2027 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 2032, + 2039 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1545, + "type": "Trellis2SaveImage", + "pos": [ + 432.2244570200339, + -721.078003096569 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 21, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1876 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1877 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1546, + "type": "Trellis2SaveImage", + "pos": [ + 444.4266599842291, + -570.5252945485121 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 22, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1878 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1879 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1547, + "type": "Trellis2SaveImage", + "pos": [ + 430.51678393065424, + -407.6644013260513 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 23, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1880 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1881 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1614, + "type": "Trellis2SaveImage", + "pos": [ + 439.89666728876546, + -267.60335440135475 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 29, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 2029 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 2030 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1616, + "type": "Trellis2SaveImage", + "pos": [ + 444.9208198224709, + -109.84120169377852 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 31, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 2032 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 2033 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1544, + "type": "Trellis2SaveImage", + "pos": [ + 430.0252087743253, + -899.771306242578 + ], + "size": [ + 270, + 82 + ], + "flags": { + "collapsed": true + }, + "order": 20, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1874 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1875 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "86f063655a00bcb3a325273ec941899cc06fd115", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + } + ], + "groups": [], + "links": [ + { + "id": 546, + "origin_id": 298, + "origin_slot": 0, + "target_id": 296, + "target_slot": 1, + "type": "HY3DCAMERA" + }, + { + "id": 649, + "origin_id": -10, + "origin_slot": 0, + "target_id": 426, + "target_slot": 0, + "type": "*" + }, + { + "id": 650, + "origin_id": 426, + "origin_slot": 0, + "target_id": 364, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 651, + "origin_id": 426, + "origin_slot": 0, + "target_id": 367, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 652, + "origin_id": 426, + "origin_slot": 0, + "target_id": 371, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 653, + "origin_id": 426, + "origin_slot": 0, + "target_id": 372, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 662, + "origin_id": -10, + "origin_slot": 1, + "target_id": 296, + "target_slot": 0, + "type": "TRIMESH" + }, + { + "id": 1204, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1035, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 1491, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1279, + "target_slot": 3, + "type": "*" + }, + { + "id": 1492, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1302, + "target_slot": 3, + "type": "*" + }, + { + "id": 1493, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1325, + "target_slot": 3, + "type": "*" + }, + { + "id": 1499, + "origin_id": 299, + "origin_slot": 0, + "target_id": 1279, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1500, + "origin_id": 318, + "origin_slot": 0, + "target_id": 1302, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1501, + "origin_id": 355, + "origin_slot": 0, + "target_id": 1325, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1502, + "origin_id": 1327, + "origin_slot": 0, + "target_id": 1328, + "target_slot": 0, + "type": "HY3DDIFFUSERSPIPE" + }, + { + "id": 1503, + "origin_id": 1328, + "origin_slot": 0, + "target_id": 1326, + "target_slot": 5, + "type": "NOISESCHEDULER" + }, + { + "id": 1504, + "origin_id": 1327, + "origin_slot": 0, + "target_id": 1326, + "target_slot": 0, + "type": "HY3DDIFFUSERSPIPE" + }, + { + "id": 1505, + "origin_id": -10, + "origin_slot": 2, + "target_id": 1326, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 1506, + "origin_id": 298, + "origin_slot": 0, + "target_id": 1326, + "target_slot": 4, + "type": "HY3DCAMERA" + }, + { + "id": 1507, + "origin_id": 296, + "origin_slot": 0, + "target_id": 1326, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 1508, + "origin_id": 296, + "origin_slot": 1, + "target_id": 1326, + "target_slot": 3, + "type": "IMAGE" + }, + { + "id": 1509, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 297, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1510, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 299, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1511, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 318, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1512, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 355, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1514, + "origin_id": 299, + "origin_slot": 0, + "target_id": 1351, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1515, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1351, + "target_slot": 3, + "type": "*" + }, + { + "id": 1516, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1374, + "target_slot": 3, + "type": "*" + }, + { + "id": 1517, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1397, + "target_slot": 3, + "type": "*" + }, + { + "id": 1518, + "origin_id": 318, + "origin_slot": 0, + "target_id": 1374, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1519, + "origin_id": 355, + "origin_slot": 0, + "target_id": 1397, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 1872, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 1543, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1873, + "origin_id": 1035, + "origin_slot": 0, + "target_id": 1543, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1874, + "origin_id": 494, + "origin_slot": 0, + "target_id": 1544, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1875, + "origin_id": 364, + "origin_slot": 0, + "target_id": 1544, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1876, + "origin_id": 1500, + "origin_slot": 0, + "target_id": 1545, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1877, + "origin_id": 367, + "origin_slot": 0, + "target_id": 1545, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1878, + "origin_id": 1501, + "origin_slot": 0, + "target_id": 1546, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1879, + "origin_id": 371, + "origin_slot": 0, + "target_id": 1546, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1880, + "origin_id": 1502, + "origin_slot": 0, + "target_id": 1547, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1881, + "origin_id": 372, + "origin_slot": 0, + "target_id": 1547, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 2020, + "origin_id": 297, + "origin_slot": 0, + "target_id": 494, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2021, + "origin_id": 299, + "origin_slot": 0, + "target_id": 1500, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2022, + "origin_id": 318, + "origin_slot": 0, + "target_id": 1501, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2023, + "origin_id": 355, + "origin_slot": 0, + "target_id": 1502, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2024, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 1609, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2025, + "origin_id": 1326, + "origin_slot": 0, + "target_id": 1610, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2026, + "origin_id": 1609, + "origin_slot": 0, + "target_id": 1611, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2027, + "origin_id": 1610, + "origin_slot": 0, + "target_id": 1612, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2028, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1613, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 2029, + "origin_id": 1611, + "origin_slot": 0, + "target_id": 1614, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2030, + "origin_id": 1613, + "origin_slot": 0, + "target_id": 1614, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 2031, + "origin_id": 426, + "origin_slot": 0, + "target_id": 1615, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 2032, + "origin_id": 1612, + "origin_slot": 0, + "target_id": 1616, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2033, + "origin_id": 1615, + "origin_slot": 0, + "target_id": 1616, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 2034, + "origin_id": 494, + "origin_slot": 0, + "target_id": -20, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 2035, + "origin_id": 1500, + "origin_slot": 0, + "target_id": -20, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 2036, + "origin_id": 1501, + "origin_slot": 0, + "target_id": -20, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 2037, + "origin_id": 1502, + "origin_slot": 0, + "target_id": -20, + "target_slot": 3, + "type": "IMAGE" + }, + { + "id": 2038, + "origin_id": 1611, + "origin_slot": 0, + "target_id": -20, + "target_slot": 4, + "type": "IMAGE" + }, + { + "id": 2039, + "origin_id": 1612, + "origin_slot": 0, + "target_id": -20, + "target_slot": 5, + "type": "IMAGE" + } + ], + "extra": { + "workflowRendererVersion": "LG" + } + }, + { + "id": "e62a2ed8-84aa-4855-9c23-e444bcf27791", + "version": 1, + "state": { + "lastGroupId": 32, + "lastNodeId": 1620, + "lastLinkId": 2048, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "Normal Generation", + "inputNode": { + "id": -10, + "bounding": [ + -3817.1162130023004, + -1898.1700046856838, + 120, + 100 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 1856.3953075122079, + -2035.1527997843789, + 190.146484375, + 80 + ] + }, + "inputs": [ + { + "id": "13fd56fd-b271-45d9-a15e-422894e3c54a", + "name": "image", + "type": "IMAGE", + "linkIds": [ + 1900 + ], + "label": "source_image", + "pos": [ + -3717.1162130023004, + -1878.1700046856838 + ] + }, + { + "id": "4d41f897-1b58-4f8c-8ea3-53dca49bbade", + "name": "", + "type": "*", + "linkIds": [ + 1876 + ], + "label": "prefix", + "pos": [ + -3717.1162130023004, + -1858.1700046856838 + ] + }, + { + "id": "cf851b38-1b59-43f9-8094-f0f28b7804ec", + "name": "_1", + "type": "*", + "linkIds": [ + 1947 + ], + "label": "seed", + "pos": [ + -3717.1162130023004, + -1838.1700046856838 + ] + } + ], + "outputs": [ + { + "id": "89cda40d-3878-45a6-bc24-ef146cc3731d", + "name": "IMAGE", + "type": "IMAGE", + "linkIds": [ + 1898 + ], + "label": "source_image", + "pos": [ + 1876.3953075122079, + -2015.1527997843789 + ] + }, + { + "id": "8cc5e680-c415-4b4e-803e-5d5115fe5847", + "name": "IMAGE_1", + "type": "IMAGE", + "linkIds": [ + 1917 + ], + "label": "normal_image_transparent", + "pos": [ + 1876.3953075122079, + -1995.1527997843789 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 1569, + "type": "ModelSamplingAuraFlow", + "pos": [ + -115.00971403135668, + -2337.6245076749606 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "MODEL", + "link": 1841 + } + ], + "outputs": [ + { + "localized_name": "MODEL", + "name": "MODEL", + "type": "MODEL", + "links": [ + 1854 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "ModelSamplingAuraFlow", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 3.1 + ] + }, + { + "id": 1570, + "type": "KSampler", + "pos": [ + 202.66715685192432, + -2335.7053574496863 + ], + "size": [ + 280, + 280 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "MODEL", + "link": 1842 + }, + { + "localized_name": "positive", + "name": "positive", + "type": "CONDITIONING", + "link": 1879 + }, + { + "localized_name": "negative", + "name": "negative", + "type": "CONDITIONING", + "link": 1880 + }, + { + "localized_name": "latent_image", + "name": "latent_image", + "type": "LATENT", + "link": 1845 + }, + { + "localized_name": "seed", + "name": "seed", + "type": "INT", + "widget": { + "name": "seed" + }, + "link": 1948 + } + ], + "outputs": [ + { + "localized_name": "LATENT", + "name": "LATENT", + "type": "LATENT", + "links": [ + 1848 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "KSampler", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 12345, + "fixed", + 4, + 1, + "euler", + "simple", + 1 + ] + }, + { + "id": 1571, + "type": "CFGNorm", + "pos": [ + -118.33919486966624, + -2215.139128869213 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "MODEL", + "link": 1854 + } + ], + "outputs": [ + { + "localized_name": "patched_model", + "name": "patched_model", + "type": "MODEL", + "links": [ + 1842 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "CFGNorm", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 1572, + "type": "EmptyLatentImage", + "pos": [ + -111.33960518501158, + -2102.8599395388537 + ], + "size": [ + 270, + 106 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "LATENT", + "name": "LATENT", + "type": "LATENT", + "links": [ + 1845 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.14.1", + "Node name for S&R": "EmptyLatentImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1024, + 1024, + 1 + ] + }, + { + "id": 1573, + "type": "TextEncodeQwenImageEditPlus", + "pos": [ + -637.9784327451256, + -2027.7176557548157 + ], + "size": [ + 420, + 168 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "localized_name": "clip", + "name": "clip", + "type": "CLIP", + "link": 1859 + }, + { + "localized_name": "vae", + "name": "vae", + "shape": 7, + "type": "VAE", + "link": 1860 + }, + { + "localized_name": "image1", + "name": "image1", + "shape": 7, + "type": "IMAGE", + "link": 1875 + }, + { + "localized_name": "image2", + "name": "image2", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "localized_name": "image3", + "name": "image3", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "localized_name": "CONDITIONING", + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 1850 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "TextEncodeQwenImageEditPlus", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + "" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 1574, + "type": "CLIPLoader", + "pos": [ + -1124.3007943593586, + -1970.878331573208 + ], + "size": [ + 396.1197916666667, + 106 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "CLIP", + "name": "CLIP", + "type": "CLIP", + "links": [ + 1857, + 1859 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "CLIPLoader", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "qwen_2.5_vl_7b_fp8_scaled.safetensors", + "url": "https://huggingface.co/Comfy-Org/HunyuanVideo_1.5_repackaged/resolve/main/split_files/text_encoders/qwen_2.5_vl_7b_fp8_scaled.safetensors", + "directory": "text_encoders" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "qwen_2.5_vl_7b_fp8_scaled.safetensors", + "qwen_image", + "default" + ] + }, + { + "id": 1575, + "type": "VAELoader", + "pos": [ + -1125.7132689453608, + -1784.1974997982275 + ], + "size": [ + 396.1197916666667, + 58 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "VAE", + "name": "VAE", + "type": "VAE", + "slot_index": 0, + "links": [ + 1849, + 1858, + 1860, + 1865 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "VAELoader", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "qwen_image_vae.safetensors", + "url": "https://huggingface.co/Comfy-Org/Qwen-Image_ComfyUI/resolve/main/split_files/vae/qwen_image_vae.safetensors", + "directory": "vae" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "qwen_image_vae.safetensors" + ] + }, + { + "id": 1576, + "type": "ImageResizeKJv2", + "pos": [ + -1489.802570145589, + -2136.853995293102 + ], + "size": [ + 278.5893539007568, + 288.0639521984865 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1903 + }, + { + "localized_name": "mask", + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1874, + 1875, + 1878 + ] + }, + { + "localized_name": "width", + "name": "width", + "type": "INT", + "links": null + }, + { + "localized_name": "height", + "name": "height", + "type": "INT", + "links": null + }, + { + "localized_name": "mask", + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "1.1.7", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 1024, + 1024, + "lanczos", + "stretch", + "0, 0, 0", + "center", + 2, + "cpu" + ] + }, + { + "id": 1577, + "type": "VAEEncode", + "pos": [ + -426.07544431121744, + -1752.8109264596974 + ], + "size": [ + 190, + 46 + ], + "flags": { + "collapsed": false + }, + "order": 10, + "mode": 0, + "inputs": [ + { + "localized_name": "pixels", + "name": "pixels", + "type": "IMAGE", + "link": 1878 + }, + { + "localized_name": "vae", + "name": "vae", + "type": "VAE", + "link": 1865 + } + ], + "outputs": [ + { + "localized_name": "LATENT", + "name": "LATENT", + "type": "LATENT", + "links": [ + 1851, + 1856 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.8.2", + "Node name for S&R": "VAEEncode", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [] + }, + { + "id": 1578, + "type": "ReferenceLatent", + "pos": [ + -76.2171515196161, + -1834.0309760866128 + ], + "size": [ + 210, + 46 + ], + "flags": { + "collapsed": false + }, + "order": 11, + "mode": 0, + "inputs": [ + { + "localized_name": "conditioning", + "name": "conditioning", + "type": "CONDITIONING", + "link": 1855 + }, + { + "localized_name": "latent", + "name": "latent", + "shape": 7, + "type": "LATENT", + "link": 1856 + } + ], + "outputs": [ + { + "localized_name": "CONDITIONING", + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 1879 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.8.2", + "Node name for S&R": "ReferenceLatent", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [] + }, + { + "id": 1579, + "type": "ReferenceLatent", + "pos": [ + -74.0707456357083, + -1733.6745960744763 + ], + "size": [ + 204.134765625, + 46 + ], + "flags": { + "collapsed": false + }, + "order": 12, + "mode": 0, + "inputs": [ + { + "localized_name": "conditioning", + "name": "conditioning", + "type": "CONDITIONING", + "link": 1850 + }, + { + "localized_name": "latent", + "name": "latent", + "shape": 7, + "type": "LATENT", + "link": 1851 + } + ], + "outputs": [ + { + "localized_name": "CONDITIONING", + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 1880 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.8.2", + "Node name for S&R": "ReferenceLatent", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65 + }, + "widgets_values": [] + }, + { + "id": 1580, + "type": "TextEncodeQwenImageEditPlus", + "pos": [ + -635.9353251063575, + -2326.2691568206615 + ], + "size": [ + 431.9022487170264, + 239.91270427521232 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "localized_name": "clip", + "name": "clip", + "type": "CLIP", + "link": 1857 + }, + { + "localized_name": "vae", + "name": "vae", + "shape": 7, + "type": "VAE", + "link": 1858 + }, + { + "localized_name": "image1", + "name": "image1", + "shape": 7, + "type": "IMAGE", + "link": 1874 + }, + { + "localized_name": "image2", + "name": "image2", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "localized_name": "image3", + "name": "image3", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "localized_name": "CONDITIONING", + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 1855 + ] + } + ], + "title": "TextEncodeQwenImageEditPlus (Positive)", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "TextEncodeQwenImageEditPlus", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Generate a highly detailed normal map of the image" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 1581, + "type": "VAEDecodeTiled", + "pos": [ + 526.7607544549602, + -2532.6820137931068 + ], + "size": [ + 225, + 150 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "link": 1848 + }, + { + "localized_name": "vae", + "name": "vae", + "type": "VAE", + "link": 1849 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1881 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.9.1", + "Node name for S&R": "VAEDecodeTiled", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 128, + 64, + 64, + 8 + ] + }, + { + "id": 1582, + "type": "Reroute", + "pos": [ + -2190.0488823440746, + -1559.9544752931013 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 1876 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 1877, + 1883 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 1583, + "type": "UnloadAllModels", + "pos": [ + 1447.9470179112147, + -2042.9135859133257 + ], + "size": [ + 151.941015625, + 26 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "localized_name": "value", + "name": "value", + "type": "*", + "link": 1916 + } + ], + "outputs": [ + { + "localized_name": "*", + "name": "*", + "type": "*", + "links": [ + 1897 + ] + } + ], + "properties": { + "cnr_id": "comfyui-unload-model", + "ver": "ac5ffb4ed05546545ce7cf38e7b69b5152714eed", + "Node name for S&R": "UnloadAllModels" + }, + "widgets_values": [] + }, + { + "id": 1584, + "type": "RMBG", + "pos": [ + -2391.9355550426676, + -2025.8660535759227 + ], + "size": [ + 320, + 320.546875 + ], + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 1900 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1901, + 1908 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1585, + "type": "StringConcatenate", + "pos": [ + -1905.0159643022744, + -1418.6616489185215 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": false + }, + "order": 18, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 1877 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1909 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.17.0", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_SourceImage", + "" + ] + }, + { + "id": 1586, + "type": "RMBG", + "pos": [ + 801.1227972789899, + -2357.1958642043965 + ], + "size": [ + 320, + 320.546875 + ], + "flags": {}, + "order": 19, + "mode": 0, + "inputs": [ + { + "localized_name": "Image", + "name": "image", + "type": "IMAGE", + "link": 1881 + } + ], + "outputs": [ + { + "localized_name": "IMAGE", + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1910, + 1913 + ] + }, + { + "localized_name": "MASK", + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "localized_name": "MASK IMAGE", + "name": "MASK_IMAGE", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-rmbg", + "ver": "2.3.2", + "Node name for S&R": "RMBG", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RMBG-2.0", + 1, + 1024, + 0, + 0, + false, + false, + "Alpha", + "#000000" + ], + "color": "#222e40", + "bgcolor": "#364254" + }, + { + "id": 1587, + "type": "Trellis2SaveImage", + "pos": [ + -1393.5546159526318, + -1449.641997359841 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 20, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1908 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1909 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [ + 1914 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1589, + "type": "Trellis2SaveImage", + "pos": [ + 1221.4525526830735, + -1497.1011446614482 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 22, + "mode": 0, + "inputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "link": 1910 + }, + { + "localized_name": "filename_prefix", + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 1911 + } + ], + "outputs": [ + { + "localized_name": "images_path", + "name": "images_path", + "type": "STRING", + "links": [ + 1915 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268", + "Node name for S&R": "Trellis2SaveImage" + }, + "widgets_values": [ + "ComfyUI", + 1 + ] + }, + { + "id": 1590, + "type": "StringConcatenate", + "pos": [ + 714.1324446088805, + -1512.6780555515054 + ], + "size": [ + 400, + 200 + ], + "flags": { + "collapsed": false + }, + "order": 23, + "mode": 0, + "inputs": [ + { + "localized_name": "string_a", + "name": "string_a", + "type": "STRING", + "widget": { + "name": "string_a" + }, + "link": 1883 + } + ], + "outputs": [ + { + "localized_name": "STRING", + "name": "STRING", + "type": "STRING", + "links": [ + 1911 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.17.0", + "Node name for S&R": "StringConcatenate" + }, + "widgets_values": [ + "", + "_NormalImage", + "" + ] + }, + { + "id": 1591, + "type": "Trellis2CudaReset", + "pos": [ + 1644.461580167075, + -2046.46286508813 + ], + "size": [ + 177.7330078125, + 26 + ], + "flags": {}, + "order": 24, + "mode": 0, + "inputs": [ + { + "localized_name": "input_1", + "name": "input_1", + "type": "*", + "link": 1897 + } + ], + "outputs": [ + { + "localized_name": "output_1", + "name": "output_1", + "type": "*", + "links": [ + 1898 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "63fe7ecf14006a8f4f2c91e7e0b1d39feaba4a74", + "Node name for S&R": "Trellis2CudaReset" + }, + "widgets_values": [] + }, + { + "id": 1592, + "type": "Trellis2PreProcessImage", + "pos": [ + -1864.8219949814027, + -2026.3949704551314 + ], + "size": [ + 281.7837890625, + 106 + ], + "flags": {}, + "order": 25, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 1901 + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 1903, + 1912 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2PreProcessImage" + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 1593, + "type": "LoraLoaderModelOnly", + "pos": [ + -1120.8732878670087, + -2178.199053568686 + ], + "size": [ + 396.1197916666667, + 82 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "MODEL", + "link": 1866 + } + ], + "outputs": [ + { + "localized_name": "MODEL", + "name": "MODEL", + "type": "MODEL", + "links": [ + 1841 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.5.1", + "Node name for S&R": "LoraLoaderModelOnly", + "enableTabs": false, + "tabWidth": 65, + "tabXOffset": 10, + "hasSecondTab": false, + "secondTabText": "Send Back", + "secondTabOffset": 80, + "secondTabWidth": 65, + "models": [ + { + "name": "Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors", + "url": "https://huggingface.co/lightx2v/Qwen-Image-Edit-2511-Lightning/resolve/main/Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors", + "directory": "loras" + } + ], + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Qwen-Image-Edit-2511-Lightning-4steps-V1.0-bf16.safetensors", + 1 + ] + }, + { + "id": 1594, + "type": "UnetLoaderGGUF", + "pos": [ + -1124.495826042935, + -2331.0673251834373 + ], + "size": [ + 364.97395833333337, + 58 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "localized_name": "MODEL", + "name": "MODEL", + "type": "MODEL", + "links": [ + 1866 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-GGUF", + "ver": "795e45156ece99afbc3efef911e63fcb46e6a20d", + "Node name for S&R": "UnetLoaderGGUF", + "aux_id": "city96/ComfyUI-GGUF", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "qwen-image-edit-2511-Q4_K_M.gguf" + ] + }, + { + "id": 1595, + "type": "Reroute", + "pos": [ + -1457.9227051676305, + -1678.5900903360498 + ], + "size": [ + 140, + 60 + ], + "flags": {}, + "order": 26, + "mode": 0, + "inputs": [ + { + "name": "", + "type": "*", + "link": 1947 + } + ], + "outputs": [ + { + "name": "", + "type": "*", + "links": [ + 1948 + ] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 1588, + "type": "Trellis2Continue4", + "pos": [ + 1244.0426972330154, + -1996.538284106618 + ], + "size": [ + 174.3150390625, + 86 + ], + "flags": {}, + "order": 21, + "mode": 0, + "inputs": [ + { + "localized_name": "input_1", + "name": "input_1", + "type": "*", + "link": 1912 + }, + { + "localized_name": "input_2", + "name": "input_2", + "type": "*", + "link": 1913 + }, + { + "localized_name": "input_3", + "name": "input_3", + "type": "*", + "link": 1914 + }, + { + "localized_name": "input_4", + "name": "input_4", + "type": "*", + "link": 1915 + } + ], + "outputs": [ + { + "localized_name": "output_1", + "name": "output_1", + "type": "*", + "links": [ + 1916 + ] + }, + { + "localized_name": "output_2", + "name": "output_2", + "type": "*", + "links": [ + 1917 + ] + }, + { + "localized_name": "output_3", + "name": "output_3", + "type": "*", + "links": null + }, + { + "localized_name": "output_4", + "name": "output_4", + "type": "*", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "0afaab3723ed18c15303f44c7aaea1bc8df61268", + "Node name for S&R": "Trellis2Continue4" + }, + "widgets_values": [] + } + ], + "groups": [ + { + "id": 30, + "title": "Resize and Remove Background", + "bounding": [ + -2444.357644543909, + -2123.9220262537956, + 907.8307755616227, + 445.7473139007566 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "links": [ + { + "id": 1841, + "origin_id": 1593, + "origin_slot": 0, + "target_id": 1569, + "target_slot": 0, + "type": "MODEL" + }, + { + "id": 1842, + "origin_id": 1571, + "origin_slot": 0, + "target_id": 1570, + "target_slot": 0, + "type": "MODEL" + }, + { + "id": 1845, + "origin_id": 1572, + "origin_slot": 0, + "target_id": 1570, + "target_slot": 3, + "type": "LATENT" + }, + { + "id": 1848, + "origin_id": 1570, + "origin_slot": 0, + "target_id": 1581, + "target_slot": 0, + "type": "LATENT" + }, + { + "id": 1849, + "origin_id": 1575, + "origin_slot": 0, + "target_id": 1581, + "target_slot": 1, + "type": "VAE" + }, + { + "id": 1850, + "origin_id": 1573, + "origin_slot": 0, + "target_id": 1579, + "target_slot": 0, + "type": "CONDITIONING" + }, + { + "id": 1851, + "origin_id": 1577, + "origin_slot": 0, + "target_id": 1579, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 1854, + "origin_id": 1569, + "origin_slot": 0, + "target_id": 1571, + "target_slot": 0, + "type": "MODEL" + }, + { + "id": 1855, + "origin_id": 1580, + "origin_slot": 0, + "target_id": 1578, + "target_slot": 0, + "type": "CONDITIONING" + }, + { + "id": 1856, + "origin_id": 1577, + "origin_slot": 0, + "target_id": 1578, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 1857, + "origin_id": 1574, + "origin_slot": 0, + "target_id": 1580, + "target_slot": 0, + "type": "CLIP" + }, + { + "id": 1858, + "origin_id": 1575, + "origin_slot": 0, + "target_id": 1580, + "target_slot": 1, + "type": "VAE" + }, + { + "id": 1859, + "origin_id": 1574, + "origin_slot": 0, + "target_id": 1573, + "target_slot": 0, + "type": "CLIP" + }, + { + "id": 1860, + "origin_id": 1575, + "origin_slot": 0, + "target_id": 1573, + "target_slot": 1, + "type": "VAE" + }, + { + "id": 1865, + "origin_id": 1575, + "origin_slot": 0, + "target_id": 1577, + "target_slot": 1, + "type": "VAE" + }, + { + "id": 1866, + "origin_id": 1594, + "origin_slot": 0, + "target_id": 1593, + "target_slot": 0, + "type": "MODEL" + }, + { + "id": 1874, + "origin_id": 1576, + "origin_slot": 0, + "target_id": 1580, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 1875, + "origin_id": 1576, + "origin_slot": 0, + "target_id": 1573, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 1876, + "origin_id": -10, + "origin_slot": 1, + "target_id": 1582, + "target_slot": 0, + "type": "*" + }, + { + "id": 1877, + "origin_id": 1582, + "origin_slot": 0, + "target_id": 1585, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 1878, + "origin_id": 1576, + "origin_slot": 0, + "target_id": 1577, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1879, + "origin_id": 1578, + "origin_slot": 0, + "target_id": 1570, + "target_slot": 1, + "type": "CONDITIONING" + }, + { + "id": 1880, + "origin_id": 1579, + "origin_slot": 0, + "target_id": 1570, + "target_slot": 2, + "type": "CONDITIONING" + }, + { + "id": 1881, + "origin_id": 1581, + "origin_slot": 0, + "target_id": 1586, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1883, + "origin_id": 1582, + "origin_slot": 0, + "target_id": 1590, + "target_slot": 0, + "type": "STRING" + }, + { + "id": 1897, + "origin_id": 1583, + "origin_slot": 0, + "target_id": 1591, + "target_slot": 0, + "type": "*" + }, + { + "id": 1898, + "origin_id": 1591, + "origin_slot": 0, + "target_id": -20, + "target_slot": 0, + "type": "*" + }, + { + "id": 1900, + "origin_id": -10, + "origin_slot": 0, + "target_id": 1584, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1901, + "origin_id": 1584, + "origin_slot": 0, + "target_id": 1592, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1903, + "origin_id": 1592, + "origin_slot": 0, + "target_id": 1576, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1908, + "origin_id": 1584, + "origin_slot": 0, + "target_id": 1587, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1909, + "origin_id": 1585, + "origin_slot": 0, + "target_id": 1587, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1910, + "origin_id": 1586, + "origin_slot": 0, + "target_id": 1589, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1911, + "origin_id": 1590, + "origin_slot": 0, + "target_id": 1589, + "target_slot": 1, + "type": "STRING" + }, + { + "id": 1912, + "origin_id": 1592, + "origin_slot": 0, + "target_id": 1588, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 1913, + "origin_id": 1586, + "origin_slot": 0, + "target_id": 1588, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 1914, + "origin_id": 1587, + "origin_slot": 0, + "target_id": 1588, + "target_slot": 2, + "type": "STRING" + }, + { + "id": 1915, + "origin_id": 1589, + "origin_slot": 0, + "target_id": 1588, + "target_slot": 3, + "type": "STRING" + }, + { + "id": 1916, + "origin_id": 1588, + "origin_slot": 0, + "target_id": 1583, + "target_slot": 0, + "type": "*" + }, + { + "id": 1917, + "origin_id": 1588, + "origin_slot": 1, + "target_id": -20, + "target_slot": 1, + "type": "*" + }, + { + "id": 1947, + "origin_id": -10, + "origin_slot": 2, + "target_id": 1595, + "target_slot": 0, + "type": "*" + }, + { + "id": 1948, + "origin_id": 1595, + "origin_slot": 0, + "target_id": 1570, + "target_slot": 4, + "type": "INT" + } + ], + "extra": {} + } + ] + }, + "config": {}, + "extra": { + "workflowRendererVersion": "LG", + "ue_links": [], + "links_added_by_ue": [], + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true, + "ds": { + "scale": 0.7603121192389539, + "offset": [ + 2009.2822506587647, + 1733.2788535228347 + ] + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/ReconViaGen_MeshOnly.json b/example_workflows/ReconViaGen_MeshOnly.json new file mode 100644 index 0000000..afdae1c --- /dev/null +++ b/example_workflows/ReconViaGen_MeshOnly.json @@ -0,0 +1,1019 @@ +{ + "id": "6eb9772a-caf3-4cd2-a1f3-8e21b2fcb356", + "revision": 0, + "last_node_id": 19, + "last_link_id": 36, + "nodes": [ + { + "id": 2, + "type": "Trellis2PreProcessImage", + "pos": [ + 860.5615222195892, + 320.28189087243953 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 1 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 3, + 7 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 1, + "type": "Trellis2LoadModel", + "pos": [ + 818.1451367332586, + -43.32986975063045 + ], + "size": [ + 301.71875, + 226 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 2, + 6 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn", + true + ] + }, + { + "id": 6, + "type": "Trellis2ImageCondGenerator", + "pos": [ + 1283.125269298477, + -44.305212193867426 + ], + "size": [ + 308.0748046875, + 98 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 6 + }, + { + "name": "image", + "type": "IMAGE", + "link": 7 + } + ], + "outputs": [ + { + "name": "cond_512", + "type": "IMAGE_COND", + "links": null + }, + { + "name": "cond_1024", + "type": "IMAGE_COND", + "links": [ + 8, + 17 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ImageCondGenerator" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 4, + "type": "Trellis2SparseGeneratorWithReconViaGen", + "pos": [ + 1257.1961548750526, + 122.64558954014163 + ], + "size": [ + 402.6431640625, + 314 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 2 + }, + { + "name": "images", + "type": "IMAGE", + "link": 3 + } + ], + "outputs": [ + { + "name": "coords", + "type": "COORDS", + "links": [ + 32 + ] + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "links": [ + 34 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 33 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2SparseGeneratorWithReconViaGen" + }, + "widgets_values": [ + 12345, + "fixed", + 25, + 7.5, + 0.01, + 5, + "euler", + 32, + 0.1, + 1 + ] + }, + { + "id": 5, + "type": "Trellis2ShapeGenerator", + "pos": [ + 1741.2710306870335, + 171.79218707401608 + ], + "size": [ + 335.24609375, + 266 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 33 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 8 + }, + { + "name": "coords", + "type": "COORDS", + "link": 32 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 18 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 19 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 16 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ShapeGenerator" + }, + "widgets_values": [ + 512, + 25, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 8, + "type": "Trellis2ShapeCascadeGenerator", + "pos": [ + 2144.5184531017253, + 89.08064130521795 + ], + "size": [ + 335.9791015625, + 338 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 16 + }, + { + "name": "image_cond", + "type": "IMAGE_COND", + "link": 17 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 18 + }, + { + "name": "from_resolution", + "type": "INT", + "widget": { + "name": "from_resolution" + }, + "link": 19 + }, + { + "name": "sparse_structure_resolution", + "type": "INT", + "widget": { + "name": "sparse_structure_resolution" + }, + "link": 34 + } + ], + "outputs": [ + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "links": [ + 11 + ] + }, + { + "name": "resolution", + "type": "INT", + "links": [ + 12 + ] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 10 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ShapeCascadeGenerator" + }, + "widgets_values": [ + 0, + 1024, + 0, + 999999, + 12, + 7.5, + 0.01, + 3, + "heun", + 0.1, + 1 + ] + }, + { + "id": 9, + "type": "Trellis2DecodeLatents", + "pos": [ + 2550.5271755896983, + 85.2758108930415 + ], + "size": [ + 270, + 122 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 10 + }, + { + "name": "shape_slat", + "type": "SHAPE_SLAT", + "link": 11 + }, + { + "name": "texture_slat", + "shape": 7, + "type": "TEXTURE_SLAT", + "link": null + }, + { + "name": "resolution", + "type": "INT", + "widget": { + "name": "resolution" + }, + "link": 12 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 13 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [] + }, + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2DecodeLatents" + }, + "widgets_values": [ + 0, + true + ] + }, + { + "id": 10, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 2861.8616589898866, + 88.51255217697037 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 13 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 14 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 11, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 2847.559963718577, + 257.899269352995 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 14 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 15 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 12, + "type": "Trellis2SimplifyMesh", + "pos": [ + 2886.0029502911757, + 486.99086892176337 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 15 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 35 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 13, + "type": "Trellis2MeshWithVoxelToTrimesh", + "pos": [ + 2840.8121187391353, + 764.4960805898208 + ], + "size": [ + 349.41171875, + 58 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 36 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 21 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + }, + "widgets_values": [ + "90 degrees" + ] + }, + { + "id": 19, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 2889.6686029432403, + 637.1515820571416 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 11, + "mode": 4, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 35 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 36 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + } + }, + { + "id": 14, + "type": "Trellis2ExportMesh", + "pos": [ + 2878.801482335511, + 893.0809816705423 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 21 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 22 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "ef2286b47ce0ef0e681dffe0209aadd96de50f72", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "Ogre_VGGT_32", + "glb" + ] + }, + { + "id": 15, + "type": "Preview3D", + "pos": [ + 1271.3057615922355, + 522.7384991788329 + ], + "size": [ + 1051.7218022911484, + 1141.158091689083 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 22 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "Preview3D", + "Camera Config": { + "cameraType": "perspective", + "fov": 75, + "state": { + "position": { + "x": 1.0315283207512576, + "y": 2.985009308965137, + "z": 4.0662863439166825 + }, + "target": { + "x": 0.20776096785142306, + "y": 2.467924964322487, + "z": -0.017876789746178047 + }, + "zoom": 1, + "cameraType": "perspective" + } + }, + "Last Time Model File": "C:/Git/ComfyUI/output/Ogre_VGGT_32_00001_.glb", + "Resource Folder": "Git/ComfyUI/output", + "Scene Config": { + "showGrid": true, + "backgroundColor": "#282828", + "backgroundImage": "", + "backgroundRenderMode": "tiled" + }, + "Light Config": { + "intensity": 3 + }, + "Model Config": { + "upDirection": "original", + "materialMode": "normal", + "showSkeleton": false + } + }, + "widgets_values": [ + "C:/Git/ComfyUI/output/Ogre_VGGT_32_00001_.glb", + "" + ] + }, + { + "id": 3, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + 207.10417104607322, + 536.1883465110252 + ], + "size": [ + 1025.7731703829552, + 1112.7651147234092 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 1 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_2048_00106_.png", + "image" + ] + } + ], + "links": [ + [ + 1, + 3, + 2, + 2, + 0, + "IMAGE" + ], + [ + 2, + 1, + 0, + 4, + 0, + "TRELLIS2PIPELINE" + ], + [ + 3, + 2, + 0, + 4, + 1, + "IMAGE" + ], + [ + 6, + 1, + 0, + 6, + 0, + "TRELLIS2PIPELINE" + ], + [ + 7, + 2, + 0, + 6, + 1, + "IMAGE" + ], + [ + 8, + 6, + 1, + 5, + 1, + "IMAGE_COND" + ], + [ + 10, + 8, + 2, + 9, + 0, + "TRELLIS2PIPELINE" + ], + [ + 11, + 8, + 0, + 9, + 1, + "SHAPE_SLAT" + ], + [ + 12, + 8, + 1, + 9, + 3, + "INT" + ], + [ + 13, + 9, + 0, + 10, + 0, + "MESHWITHVOXEL" + ], + [ + 14, + 10, + 0, + 11, + 0, + "MESHWITHVOXEL" + ], + [ + 15, + 11, + 0, + 12, + 0, + "MESHWITHVOXEL" + ], + [ + 16, + 5, + 2, + 8, + 0, + "TRELLIS2PIPELINE" + ], + [ + 17, + 6, + 1, + 8, + 1, + "IMAGE_COND" + ], + [ + 18, + 5, + 0, + 8, + 2, + "SHAPE_SLAT" + ], + [ + 19, + 5, + 1, + 8, + 3, + "INT" + ], + [ + 21, + 13, + 0, + 14, + 0, + "TRIMESH" + ], + [ + 22, + 14, + 0, + 15, + 2, + "STRING" + ], + [ + 32, + 4, + 0, + 5, + 2, + "COORDS" + ], + [ + 33, + 4, + 2, + 5, + 0, + "TRELLIS2PIPELINE" + ], + [ + 34, + 4, + 1, + 8, + 4, + "INT" + ], + [ + 35, + 12, + 0, + 19, + 0, + "MESHWITHVOXEL" + ], + [ + 36, + 19, + 0, + 13, + 0, + "MESHWITHVOXEL" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.5644739300537774, + "offset": [ + -319.70457827331774, + 353.8321991591369 + ] + }, + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/RefineMesh.json b/example_workflows/RefineMesh.json new file mode 100644 index 0000000..b604831 --- /dev/null +++ b/example_workflows/RefineMesh.json @@ -0,0 +1,810 @@ +{ + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", + "revision": 0, + "last_node_id": 233, + "last_link_id": 480, + "nodes": [ + { + "id": 226, + "type": "Trellis2LoadMesh", + "pos": [ + -385.0735510204828, + 158.88137472879367 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 460 + ] + } + ], + "title": "Original Mesh", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2LoadMesh" + }, + "widgets_values": [ + "" + ] + }, + { + "id": 119, + "type": "Trellis2PreProcessImage", + "pos": [ + 432.4981118440155, + -182.40504275448276 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 461 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 418.8462464575081, + -492.4195525023097 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 469 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" + ] + }, + { + "id": 227, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1190.2002753188488, + -496.8180011267565 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 470 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 472 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 231, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 1203.103220151927, + -621.2330460050034 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 474 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 470 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 229, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1194.8431366936325, + -292.14794928709534 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 472 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 475 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 471 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 228, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1198.1583617137858, + -151.8665797237854 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 471 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 476 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 232, + "type": "Trellis2UnWrapAndRasterizer", + "pos": [ + 1195.383895062337, + -50.23863923208595 + ], + "size": [ + 419.15234375, + 314 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 476 + }, + { + "name": "bvh", + "type": "BVH", + "link": 477 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 478 + ] + }, + { + "name": "base_color_texture", + "type": "IMAGE", + "links": null + }, + { + "name": "metallic_roughness_texture", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2UnWrapAndRasterizer" + }, + "widgets_values": [ + 60, + 0, + 1, + 1, + 2048, + "OPAQUE", + false, + false, + false, + "telea" + ] + }, + { + "id": 233, + "type": "Trellis2ExportMesh", + "pos": [ + 1672.6947901006665, + -49.057670650420995 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 478 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 479 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 480 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 369.6388214873717, + 332.92876516342716 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 480 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "Preview3D" + }, + "widgets_values": [ + "", + "" + ] + }, + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -383.4145219550909, + 310.9862093076388 + ], + "size": [ + 717.521556382955, + 915.5313594325766 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -379.85391723715725, + 7.869022281792564 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 475 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -388.041436029903, + -125.57562894305721 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 479 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 225, + "type": "Trellis2MeshRefiner", + "pos": [ + 753.899038005419, + -389.13958558695583 + ], + "size": [ + 340.390625, + 578 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 469 + }, + { + "name": "trimesh", + "type": "TRIMESH", + "link": 460 + }, + { + "name": "image", + "type": "IMAGE", + "link": 461 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 474 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [ + 477 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshRefiner" + }, + "widgets_values": [ + 12345, + "fixed", + 1024, + 12, + 7.5, + 0.01, + 3, + 12, + 3, + 0.01, + 3, + 999999, + true, + 16, + 0.1, + 1, + 0, + 0.9, + true, + 1, + "heun" + ] + } + ], + "links": [ + [ + 241, + 69, + 2, + 119, + 0, + "IMAGE" + ], + [ + 460, + 226, + 0, + 225, + 1, + "TRIMESH" + ], + [ + 461, + 119, + 0, + 225, + 2, + "IMAGE" + ], + [ + 469, + 39, + 0, + 225, + 0, + "TRELLIS2PIPELINE" + ], + [ + 470, + 231, + 0, + 227, + 0, + "MESHWITHVOXEL" + ], + [ + 471, + 229, + 0, + 228, + 0, + "MESHWITHVOXEL" + ], + [ + 472, + 227, + 0, + 229, + 0, + "MESHWITHVOXEL" + ], + [ + 474, + 225, + 0, + 231, + 0, + "MESHWITHVOXEL" + ], + [ + 475, + 203, + 0, + 229, + 1, + "INT" + ], + [ + 476, + 228, + 0, + 232, + 0, + "MESHWITHVOXEL" + ], + [ + 477, + 225, + 1, + 232, + 1, + "BVH" + ], + [ + 478, + 232, + 0, + 233, + 0, + "TRIMESH" + ], + [ + 479, + 204, + 0, + 233, + 1, + "STRING" + ], + [ + 480, + 233, + 0, + 202, + 2, + "STRING" + ] + ], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -414.5659226642583, + -213.34811694305725, + 764.6855244821979, + 1466.2928336195214 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 374.5341225783577, + -726.2357215927766, + 1820.8266085195564, + 1004.7525102622308 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "workflowRendererVersion": "LG", + "ue_links": [], + "ds": { + "scale": 0.6830134553650716, + "offset": [ + 111.46261093344825, + 859.8232012529436 + ] + }, + "links_added_by_ue": [], + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/RefineMesh_MeshOnly.json b/example_workflows/RefineMesh_MeshOnly.json new file mode 100644 index 0000000..e79f714 --- /dev/null +++ b/example_workflows/RefineMesh_MeshOnly.json @@ -0,0 +1,776 @@ +{ + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", + "revision": 0, + "last_node_id": 234, + "last_link_id": 482, + "nodes": [ + { + "id": 226, + "type": "Trellis2LoadMesh", + "pos": [ + -385.0735510204828, + 158.88137472879367 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 460 + ] + } + ], + "title": "Original Mesh", + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2LoadMesh" + }, + "widgets_values": [ + "" + ] + }, + { + "id": 119, + "type": "Trellis2PreProcessImage", + "pos": [ + 432.4981118440155, + -182.40504275448276 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 461 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, + { + "id": 39, + "type": "Trellis2LoadModel", + "pos": [ + 418.8462464575081, + -492.4195525023097 + ], + "size": [ + 301.71875, + 202 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "links": [ + 469 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "c5d66aa46bf1bbf903fd4fe9ff086ac774bb7e03", + "Node name for S&R": "Trellis2LoadModel", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "microsoft/TRELLIS.2-4B", + "flash_attn", + "cuda", + true, + false, + "flex_gemm", + "flash_attn" + ] + }, + { + "id": 227, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1190.2002753188488, + -496.8180011267565 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 470 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 472 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1, + 1024, + true, + true + ] + }, + { + "id": 231, + "type": "Trellis2FillHolesWithCuMesh", + "pos": [ + 1203.103220151927, + -621.2330460050034 + ], + "size": [ + 312.4361328125, + 58 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 474 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 470 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 229, + "type": "Trellis2SimplifyMesh", + "pos": [ + 1194.8431366936325, + -292.14794928709534 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 472 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 475 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 471 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" + }, + "widgets_values": [ + 500000, + "Cumesh" + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 369.6388214873717, + 332.92876516342716 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "camera_info", + "shape": 7, + "type": "LOAD3D_CAMERA", + "link": null + }, + { + "name": "bg_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "model_file", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", + "widget": { + "name": "model_file" + }, + "link": 480 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "Preview3D" + }, + "widgets_values": [ + "", + "" + ] + }, + { + "id": 69, + "type": "Trellis2LoadImageWithTransparency", + "pos": [ + -383.4145219550909, + 310.9862093076388 + ], + "size": [ + 717.521556382955, + 915.5313594325766 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [] + }, + { + "name": "mask", + "type": "MASK", + "links": [] + }, + { + "name": "image_with_alpha", + "type": "IMAGE", + "links": [ + 241 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", + "Node name for S&R": "Trellis2LoadImageWithTransparency", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "Image_1024_00101_.png", + "image" + ] + }, + { + "id": 203, + "type": "PrimitiveInt", + "pos": [ + -379.85391723715725, + 7.869022281792564 + ], + "size": [ + 270, + 82 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 475 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -388.041436029903, + -125.57562894305721 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 479 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 225, + "type": "Trellis2MeshRefiner", + "pos": [ + 753.899038005419, + -389.13958558695583 + ], + "size": [ + 340.390625, + 578 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 469 + }, + { + "name": "trimesh", + "type": "TRIMESH", + "link": 460 + }, + { + "name": "image", + "type": "IMAGE", + "link": 461 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 474 + ] + }, + { + "name": "bvh", + "type": "BVH", + "links": [] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshRefiner" + }, + "widgets_values": [ + 12345, + "fixed", + 1024, + 12, + 7.5, + 0.01, + 3, + 12, + 3, + 0.01, + 3, + 999999, + false, + 16, + 0.1, + 1, + 0, + 0.9, + true, + 1, + "heun" + ] + }, + { + "id": 228, + "type": "Trellis2FillHolesWithMeshlib", + "pos": [ + 1198.1583617137858, + -151.8665797237854 + ], + "size": [ + 249.284765625, + 46 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 471 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 481 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 234, + "type": "Trellis2MeshWithVoxelToTrimesh", + "pos": [ + 1183.7879296915492, + 4.235191654770113 + ], + "size": [ + 349.41171875, + 58 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 481 + } + ], + "outputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "links": [ + 482 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelToTrimesh" + }, + "widgets_values": [ + "90 degrees" + ] + }, + { + "id": 233, + "type": "Trellis2ExportMesh", + "pos": [ + 1205.1587971929546, + 133.46673644186396 + ], + "size": [ + 270, + 102 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "trimesh", + "type": "TRIMESH", + "link": 482 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 479 + } + ], + "outputs": [ + { + "name": "glb_path", + "type": "STRING", + "links": [ + 480 + ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" + }, + "widgets_values": [ + "3D/Trellis2", + "glb" + ] + } + ], + "links": [ + [ + 241, + 69, + 2, + 119, + 0, + "IMAGE" + ], + [ + 460, + 226, + 0, + 225, + 1, + "TRIMESH" + ], + [ + 461, + 119, + 0, + 225, + 2, + "IMAGE" + ], + [ + 469, + 39, + 0, + 225, + 0, + "TRELLIS2PIPELINE" + ], + [ + 470, + 231, + 0, + 227, + 0, + "MESHWITHVOXEL" + ], + [ + 471, + 229, + 0, + 228, + 0, + "MESHWITHVOXEL" + ], + [ + 472, + 227, + 0, + 229, + 0, + "MESHWITHVOXEL" + ], + [ + 474, + 225, + 0, + 231, + 0, + "MESHWITHVOXEL" + ], + [ + 475, + 203, + 0, + 229, + 1, + "INT" + ], + [ + 479, + 204, + 0, + 233, + 1, + "STRING" + ], + [ + 480, + 233, + 0, + 202, + 2, + "STRING" + ], + [ + 481, + 228, + 0, + 234, + 0, + "MESHWITHVOXEL" + ], + [ + 482, + 234, + 0, + 233, + 0, + "TRIMESH" + ] + ], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -414.5659226642583, + -213.34811694305725, + 764.6855244821979, + 1466.2928336195214 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 374.5341225783577, + -726.2357215927766, + 1227.3780156118423, + 1005.7286067160883 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "workflowRendererVersion": "LG", + "ue_links": [], + "ds": { + "scale": 0.6830134553650716, + "offset": [ + 200.2846180257333, + 818.8284012529438 + ] + }, + "links_added_by_ue": [], + "frontendVersion": "1.42.8", + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/example_workflows/Simple.json b/example_workflows/Simple.json index 7877040..18ecd1a 100644 --- a/example_workflows/Simple.json +++ b/example_workflows/Simple.json @@ -1,19 +1,19 @@ { - "id": "440427ce-0c6a-4462-af51-639ed2f16dec", + "id": "cd6e2e00-83cc-4795-abf1-09b46428c270", "revision": 0, - "last_node_id": 44, - "last_link_id": 90, + "last_node_id": 205, + "last_link_id": 413, "nodes": [ { - "id": 6, + "id": 69, "type": "Trellis2LoadImageWithTransparency", "pos": [ - 160.23097908593746, - 1093.4781077920234 + -377.72410571388025, + 328.4870830026766 ], "size": [ - 454.3125, - 496.03125 + 717.521556382955, + 915.5313594325766 ], "flags": {}, "order": 0, @@ -23,18 +23,18 @@ { "name": "image", "type": "IMAGE", - "links": null + "links": [] }, { "name": "mask", "type": "MASK", - "links": null + "links": [] }, { "name": "image_with_alpha", "type": "IMAGE", "links": [ - 89 + 241 ] } ], @@ -45,20 +45,62 @@ "widget_ue_connectable": {} }, "widgets_values": [ - "Image_04291_.png", + "Image_1024_00101_.png", "image" ] }, + { + "id": 119, + "type": "Trellis2PreProcessImage", + "pos": [ + 398.0246741688026, + 149.83708875416676 + ], + "size": [ + 281.8229166666667, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 241 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 391 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "15854282a73cf231b81d52ada22e652f414a078b", + "Node name for S&R": "Trellis2PreProcessImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 10, + false, + 1024 + ] + }, { "id": 39, "type": "Trellis2LoadModel", "pos": [ - 143.14495071454974, - 622.610212818676 + 388.892170207283, + -228.80616406738957 ], "size": [ - 362.0625, - 207.328125 + 301.71875, + 202 ], "flags": {}, "order": 1, @@ -69,7 +111,7 @@ "name": "pipeline", "type": "TRELLIS2PIPELINE", "links": [ - 79 + 390 ] } ], @@ -80,37 +122,34 @@ "widget_ue_connectable": {} }, "widgets_values": [ - "TRELLIS.2-4B", + "microsoft/TRELLIS.2-4B", "flash_attn", "cuda", true, - false + false, + "flex_gemm", + "flash_attn" ] }, { - "id": 41, - "type": "Trellis2MeshWithVoxelGenerator", + "id": 196, + "type": "Trellis2FillHolesWithCuMesh", "pos": [ - 673.7142651175388, - 836.004782154444 + 1117.157533597802, + -362.86738879994687 ], "size": [ - 410.71875, - 386 + 312.4361328125, + 58 ], "flags": {}, - "order": 3, + "order": 6, "mode": 0, "inputs": [ { - "name": "pipeline", - "type": "TRELLIS2PIPELINE", - "link": 79 - }, - { - "name": "image", - "type": "IMAGE", - "link": 90 + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 396 } ], "outputs": [ @@ -118,92 +157,192 @@ "name": "mesh", "type": "MESHWITHVOXEL", "links": [ - 86 - ] - }, - { - "name": "bvh", - "type": "BVH", - "links": [ - 87 + 392 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a28d6cf93b0cf24aa9e7d3dccfd14f406227fbd0", - "Node name for S&R": "Trellis2MeshWithVoxelGenerator", + "ver": "dc66f263e832b103a81a021bf6cd53eec8ff28af", + "Node name for S&R": "Trellis2FillHolesWithCuMesh" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 197, + "type": "Trellis2ReconstructMeshWithQuad", + "pos": [ + 1124.8153675662377, + -239.75157745064854 + ], + "size": [ + 331.5878996659427, + 142.11360851901668 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 392 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 393 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "d4d66cfc9f1ce5b1c3b9a2e504ee4b24ffad48a5", + "Node name for S&R": "Trellis2ReconstructMeshWithQuad", "widget_ue_connectable": {} }, "widgets_values": [ - 12345, - "randomize", - "1024_cascade", - 12, - 12, - 12, - 999999, - 4, - 32, + 1, + 1024, true, true ] }, { - "id": 19, - "type": "Trellis2ExportMesh", + "id": 198, + "type": "Trellis2SimplifyMesh", "pos": [ - 1480.0390859207857, - 603.8997701786345 + 1134.132576655357, + -43.60066335397867 ], "size": [ - 324, - 146 + 270, + 82 ], "flags": {}, - "order": 5, + "order": 8, "mode": 0, "inputs": [ { - "name": "trimesh", - "type": "TRIMESH", - "link": 88 + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 393 + }, + { + "name": "target_face_num", + "type": "INT", + "widget": { + "name": "target_face_num" + }, + "link": 409 } ], "outputs": [ { - "name": "glb_path", - "type": "STRING", + "name": "mesh", + "type": "MESHWITHVOXEL", "links": [ - 22 + 394 ] } ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "e7f9b30df7a09bcedf1c955e176754c73f983254", - "Node name for S&R": "Trellis2ExportMesh", - "widget_ue_connectable": {} + "ver": "5d4c1474ceb477f40bbd6d437f7da90f3d6d9229", + "Node name for S&R": "Trellis2SimplifyMesh" }, "widgets_values": [ - "Trellis2Mesh", - "glb", - true + 500000, + "Cumesh" ] }, { - "id": 10, - "type": "Preview3D", + "id": 203, + "type": "PrimitiveInt", "pos": [ - 1595.8257224427462, - 828.6814144924502 + -373.0898096413628, + 163.12227474974895 ], "size": [ - 752.25, - 729.90625 + 270, + 82 ], "flags": {}, - "order": 6, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 409 + ] + } + ], + "title": "Target Face Number", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveInt" + }, + "widgets_values": [ + 300000, + "fixed" + ] + }, + { + "id": 204, + "type": "PrimitiveString", + "pos": [ + -368.2858630795249, + 34.401714106564214 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "STRING", + "type": "STRING", + "links": [ + 410 + ] + } + ], + "title": "Name", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.18.1", + "Node name for S&R": "PrimitiveString" + }, + "widgets_values": [ + "Tank" + ] + }, + { + "id": 202, + "type": "Preview3D", + "pos": [ + 376.48890096546415, + 331.2515281089069 + ], + "size": [ + 868.8731050199159, + 920.1382602693357 + ], + "flags": {}, + "order": 12, "mode": 0, "inputs": [ { @@ -220,77 +359,152 @@ }, { "name": "model_file", - "type": "STRING", + "type": "STRING,FILE_3D_GLB,FILE_3D_GLTF,FILE_3D_FBX,FILE_3D_OBJ,FILE_3D_STL,FILE_3D_USDZ,FILE_3D", "widget": { "name": "model_file" }, - "link": 22 + "link": 398 } ], "outputs": [], "properties": { "cnr_id": "comfy-core", - "ver": "0.4.0", - "Node name for S&R": "Preview3D", - "widget_ue_connectable": {}, - "Last Time Model File": "Home_2K_00001_.glb", - "Scene Config": { - "showGrid": true, - "backgroundColor": "#282828", - "backgroundImage": "", - "backgroundRenderMode": "tiled" - }, - "Camera Config": { - "cameraType": "perspective", - "fov": 35, - "state": { - "position": { - "x": 12.514315832403488, - "y": 3.895614510886169, - "z": 9.139010516199487 - }, - "target": { - "x": 0, - "y": 2.1870527424637647, - "z": 0 - }, - "zoom": 1, - "cameraType": "perspective" - } - }, - "Light Config": { - "intensity": 3 - } + "ver": "0.18.1", + "Node name for S&R": "Preview3D" }, "widgets_values": [ - "Home_2K_00001_.glb", + "", "" ] }, { - "id": 43, - "type": "Trellis2PostProcessAndUnWrapAndRasterizer", + "id": 199, + "type": "Trellis2FillHolesWithMeshlib", "pos": [ - 1115.7105853634694, - 836.2591783750445 + 1132.506716841996, + 89.90828212755481 ], "size": [ - 419.125, - 651.328125 + 249.284765625, + 46 ], "flags": {}, - "order": 4, + "order": 9, "mode": 0, "inputs": [ { "name": "mesh", "type": "MESHWITHVOXEL", - "link": 86 + "link": 394 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 411 + ] + }, + { + "name": "holes_filled", + "type": "INT", + "links": null + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "e8098f5641702e74133c2dea0b45a13a3ab19985", + "Node name for S&R": "Trellis2FillHolesWithMeshlib" + }, + "widgets_values": [] + }, + { + "id": 195, + "type": "Trellis2MeshWithVoxelGenerator", + "pos": [ + 746.136770520365, + -107.27192057879526 + ], + "size": [ + 342.2818359375, + 342 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "pipeline", + "type": "TRELLIS2PIPELINE", + "link": 390 + }, + { + "name": "image", + "type": "IMAGE", + "link": 391 + } + ], + "outputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "links": [ + 396 + ] }, { "name": "bvh", "type": "BVH", - "link": 87 + "links": [ + 412 + ] + } + ], + "properties": { + "aux_id": "visualbruno/ComfyUI-Trellis2", + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2MeshWithVoxelGenerator" + }, + "widgets_values": [ + 12345, + "fixed", + "1024_cascade", + 12, + 12, + 12, + 999999, + 1, + 32, + true, + true, + "euler" + ] + }, + { + "id": 205, + "type": "Trellis2UnWrapAndRasterizer", + "pos": [ + 1509.48883255403, + -231.18851954550428 + ], + "size": [ + 419.15234375, + 314 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "mesh", + "type": "MESHWITHVOXEL", + "link": 411 + }, + { + "name": "bvh", + "type": "BVH", + "link": 412 } ], "outputs": [ @@ -298,7 +512,7 @@ "name": "trimesh", "type": "TRIMESH", "links": [ - 88 + 413 ] }, { @@ -314,144 +528,223 @@ ], "properties": { "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72", - "Node name for S&R": "Trellis2PostProcessAndUnWrapAndRasterizer", - "widget_ue_connectable": {} + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2UnWrapAndRasterizer" }, "widgets_values": [ 60, 0, 1, 1, - 4096, - true, - 1, - 0, - 2000000, - "Cumesh", - true, + 2048, "OPAQUE", - "1024", - false, - true, false, false, - true + false, + "telea" ] }, { - "id": 44, - "type": "Trellis2PreProcessImage", + "id": 201, + "type": "Trellis2ExportMesh", "pos": [ - 246.66298457453752, - 916.7884835448142 + 1517.3851156356834, + 145.96275340737535 ], "size": [ - 281.78125, - 79.328125 + 270, + 102 ], "flags": {}, - "order": 2, + "order": 11, "mode": 0, "inputs": [ { - "name": "image", - "type": "IMAGE", - "link": 89 + "name": "trimesh", + "type": "TRIMESH", + "link": 413 + }, + { + "name": "filename_prefix", + "type": "STRING", + "widget": { + "name": "filename_prefix" + }, + "link": 410 } ], "outputs": [ { - "name": "image", - "type": "IMAGE", + "name": "glb_path", + "type": "STRING", "links": [ - 90 + 398 ] + }, + { + "name": "relative_path", + "type": "STRING", + "links": null } ], "properties": { - "widget_ue_connectable": {}, "aux_id": "visualbruno/ComfyUI-Trellis2", - "ver": "a354e8c4fead5c152d58b50eae2935cb2923cd72", - "Node name for S&R": "Trellis2PreProcessImage" + "ver": "a5242473f68fe91d5ef0b188f8658ae0da540237", + "Node name for S&R": "Trellis2ExportMesh" }, "widgets_values": [ - 0 + "3D/Trellis2", + "glb" ] } ], "links": [ [ - 22, - 19, - 0, - 10, + 241, + 69, 2, - "STRING" + 119, + 0, + "IMAGE" ], [ - 79, + 390, 39, 0, - 41, + 195, 0, "TRELLIS2PIPELINE" ], [ - 86, - 41, + 391, + 119, 0, - 43, + 195, + 1, + "IMAGE" + ], + [ + 392, + 196, + 0, + 197, 0, "MESHWITHVOXEL" ], [ - 87, - 41, + 393, + 197, + 0, + 198, + 0, + "MESHWITHVOXEL" + ], + [ + 394, + 198, + 0, + 199, + 0, + "MESHWITHVOXEL" + ], + [ + 396, + 195, + 0, + 196, + 0, + "MESHWITHVOXEL" + ], + [ + 398, + 201, + 0, + 202, + 2, + "STRING" + ], + [ + 409, + 203, + 0, + 198, 1, - 43, + "INT" + ], + [ + 410, + 204, + 0, + 201, + 1, + "STRING" + ], + [ + 411, + 199, + 0, + 205, + 0, + "MESHWITHVOXEL" + ], + [ + 412, + 195, + 1, + 205, 1, "BVH" ], [ - 88, - 43, + 413, + 205, 0, - 19, + 201, 0, "TRIMESH" - ], - [ - 89, - 6, - 2, - 44, - 0, - "IMAGE" - ], - [ - 90, - 44, - 0, - 41, - 1, - "IMAGE" ] ], - "groups": [], + "groups": [ + { + "id": 1, + "title": "Configuration", + "bounding": [ + -387.72410571388025, + -53.37077389343578, + 737.521556382955, + 1307.389216328689 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Generation", + "bounding": [ + 378.892170207283, + -436.46738879994683, + 1706.6665797550602, + 727.1063315541132 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], "config": {}, "extra": { - "workflowRendererVersion": "Vue", + "workflowRendererVersion": "LG", "ue_links": [], "ds": { - "scale": 0.6303940863128564, + "scale": 0.5644739300537773, "offset": [ - -52.07567851462492, - -327.7388708088908 + 683.341395587175, + 735.5254552429528 ] }, "links_added_by_ue": [], - "frontendVersion": "1.37.11", + "frontendVersion": "1.42.8", "VHS_latentpreview": false, "VHS_latentpreviewrate": 0, "VHS_MetadataImage": true, diff --git a/nodes.py b/nodes.py index 7a462c0..3813f39 100644 --- a/nodes.py +++ b/nodes.py @@ -331,6 +331,7 @@ class Trellis2LoadModel: "keep_models_loaded": ("BOOLEAN", {"default":True}), "conv_backend": (["spconv","torchsparse","flex_gemm"],{"default":"flex_gemm"}), "sparse_backend": (["xformers","flash_attn"],{"default":"flash_attn"}), + "use_reconviagen": ("BOOLEAN",{"default":False}), }, } @@ -340,7 +341,9 @@ class Trellis2LoadModel: CATEGORY = "Trellis2Wrapper" OUTPUT_NODE = True - def process(self, modelname, backend, device, low_vram, keep_models_loaded, conv_backend, sparse_backend): + def process(self, modelname, backend, device, low_vram, keep_models_loaded, conv_backend, sparse_backend, use_reconviagen): + import requests + os.environ['OPENCV_IO_ENABLE_OPENEXR'] = '1' os.environ["PYTORCH_CUDA_ALLOC_CONF"] = "expandable_segments:True" # Can save GPU memory #os.environ["FLEX_GEMM_AUTOTUNE_CACHE_PATH"] = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'autotune_cache.json') @@ -365,15 +368,19 @@ class Trellis2LoadModel: local_dir=model_path, local_dir_use_symlinks=False, ) - + + reconviagen_pipeline_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','reconviagen_pipeline.json') + if not os.path.exists(reconviagen_pipeline_file): + source_reconviagen_pipeline_file = os.path.join(script_directory,'reconviagen_pipeline.json') + shutil.copyfile(source_reconviagen_pipeline_file,reconviagen_pipeline_file) + dinov3_model_path = os.path.join(folder_paths.models_dir,"facebook","dinov3-vitl16-pretrain-lvd1689m","model.safetensors") if not os.path.exists(dinov3_model_path): raise Exception("Facebook Dinov3 model not found in models/facebook/dinov3-vitl16-pretrain-lvd1689m folder") trellis_image_large_path = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS-image-large","ckpts","ss_dec_conv3d_16l8_fp16.safetensors") if not os.path.exists(trellis_image_large_path): - print('Trellis-Image-Large ss_dec_conv3d_16l8_fp16 files not found. Trying to download the files from huggingface ...') - import requests + print('Trellis-Image-Large ss_dec_conv3d_16l8_fp16 files not found. Trying to download the files from huggingface ...') url = "https://huggingface.co/microsoft/TRELLIS-image-large/resolve/main/ckpts/ss_dec_conv3d_16l8_fp16.json?download=true" filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS-image-large","ckpts","ss_dec_conv3d_16l8_fp16.json") path = Path(filename) @@ -398,12 +405,79 @@ class Trellis2LoadModel: else: raise Exception("Cannot download Trellis-Image-Large file ss_dec_conv3d_16l8_fp16.safetensors") + if use_reconviagen: + reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_vggt_cond.safetensors') + if not os.path.exists(reconviagen_file): + print('ReconViaGen file ss_vggt_cond.safetensors not found. Trying to download the files from huggingface ...') + url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_vggt_cond.safetensors?download=true" + filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_vggt_cond.safetensors") + path = Path(filename) + path.parent.mkdir(parents=True, exist_ok=True) + + response = requests.get(url) + if response.status_code == 200: + with open(filename, "wb") as f: + f.write(response.content) + print("Download ss_vggt_cond.safetensors complete!") + else: + raise Exception("Cannot download ReconViaGen file ss_vggt_cond.safetensors") + + reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_vggt_cond.json') + if not os.path.exists(reconviagen_file): + print('ReconViaGen file ss_vggt_cond.json not found. Trying to download the files from huggingface ...') + url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_vggt_cond.json?download=true" + filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_vggt_cond.json") + path = Path(filename) + path.parent.mkdir(parents=True, exist_ok=True) + + response = requests.get(url) + if response.status_code == 200: + with open(filename, "wb") as f: + f.write(response.content) + print("Download ss_vggt_cond.json complete!") + else: + raise Exception("Cannot download ReconViaGen file ss_vggt_cond.json") + + reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_flow_img_dit_L_16l8_fp16.safetensors') + if not os.path.exists(reconviagen_file): + print('ReconViaGen file ss_flow_img_dit_L_16l8_fp16.safetensors not found. Trying to download the files from huggingface ...') + url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_flow_img_dit_L_16l8_fp16.safetensors?download=true" + filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_flow_img_dit_L_16l8_fp16.safetensors") + path = Path(filename) + path.parent.mkdir(parents=True, exist_ok=True) + + response = requests.get(url) + if response.status_code == 200: + with open(filename, "wb") as f: + f.write(response.content) + print("Download ss_flow_img_dit_L_16l8_fp16.safetensors complete!") + else: + raise Exception("Cannot download ReconViaGen file ss_flow_img_dit_L_16l8_fp16.safetensors") + + reconviagen_file = os.path.join(folder_paths.models_dir,'microsoft','TRELLIS.2-4B','ckpts','ss_flow_img_dit_L_16l8_fp16.json') + if not os.path.exists(reconviagen_file): + print('ReconViaGen file ss_flow_img_dit_L_16l8_fp16.json not found. Trying to download the files from huggingface ...') + url = "https://huggingface.co/Stable-X/trellis-vggt-v0-2/resolve/main/ckpts/ss_flow_img_dit_L_16l8_fp16.json?download=true" + filename = os.path.join(folder_paths.models_dir,"microsoft","TRELLIS.2-4B","ckpts","ss_flow_img_dit_L_16l8_fp16.json") + path = Path(filename) + path.parent.mkdir(parents=True, exist_ok=True) + + response = requests.get(url) + if response.status_code == 200: + with open(filename, "wb") as f: + f.write(response.content) + print("Download ss_flow_img_dit_L_16l8_fp16.json complete!") + else: + raise Exception("Cannot download ReconViaGen file ss_flow_img_dit_L_16l8_fp16.json") + if modelname == "visualbruno/TRELLIS.2-4B-FP8": use_fp8 = True + if use_reconviagen: + raise Exception("ReconViaGen cannot be used with TRELLIS.2-4B-FP8. Select microsoft/TRELLIS.2-4B") else: use_fp8 = False - pipeline = Trellis2ImageTo3DPipeline.from_pretrained(model_path, keep_models_loaded = keep_models_loaded, use_fp8=use_fp8) + pipeline = Trellis2ImageTo3DPipeline.from_pretrained(model_path, keep_models_loaded = keep_models_loaded, use_fp8=use_fp8, use_reconviagen=use_reconviagen) pipeline.low_vram = low_vram if device=="cuda": @@ -4613,6 +4687,380 @@ class Trellis2VoxelToMesh: return (mesh_copy,) +class Trellis2UnloadAllModels: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "input_1": (any,) + }, + } + + RETURN_TYPES = (any,) + RETURN_NAMES = ("output_1",) + FUNCTION = "process" + CATEGORY = "Trellis2Wrapper" + OUTPUT_NODE = True + + def process(self, input_1): + print('Unloading all models ...') + if hasattr(mm, 'current_loaded_models'): + # Iterate backwards to safely remove items + for i in range(len(mm.current_loaded_models) - 1, -1, -1): + loaded_model = mm.current_loaded_models[i] + + print(f"[AbsoluteUnload] Force-killing: {loaded_model.model.model.__class__.__name__}") + + # Force VRAM unload + loaded_model.model_unload(1e32) + + # Force System RAM unpinning (This is what the standard loop skipped) + if hasattr(loaded_model.model, 'partially_unload_ram'): + loaded_model.model.partially_unload_ram(1e32) + + # Clear ComfyUI's intermediate cross-attention and tensor caches + if hasattr(mm, 'current_loaded_models'): + mm.current_loaded_models.clear() + + import comfy.controlnet + if hasattr(comfy.controlnet, 'controlnet_loaded_models'): + comfy.controlnet.controlnet_loaded_models.clear() + + mm.free_memory(memory_required = 1e30, + device = mm.get_torch_device(), + ram_required = 1e30) + + print('Clearing cache ...') + mm.soft_empty_cache() + + gc.collect() + gc.collect() + + if torch.cuda.is_available(): + torch.cuda.empty_cache() + torch.cuda.ipc_collect() + + print('Memory cleared') + + return (input_1,) + +class Trellis2SparseGeneratorWithReconViaGen: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "pipeline": ("TRELLIS2PIPELINE",), + "images": ("IMAGE",), + "seed": ("INT", {"default": 0, "min": 0, "max": 0x7fffffff}), + "sparse_structure_steps": ("INT",{"default":12, "min":1, "max":100},), + "sparse_structure_guidance_strength": ("FLOAT",{"default":6.50,"min":0.00,"max":99.99,"step":0.01}), + "sparse_structure_guidance_rescale": ("FLOAT",{"default":0.05,"min":0.00,"max":1.00,"step":0.01}), + "sparse_structure_rescale_t": ("FLOAT",{"default":4.00,"min":0.00,"max":9.99,"step":0.01}), + "sparse_structure_sampler": (["euler", "heun", "rk4", "rk5"], {"default": "euler"}), + "sparse_structure_resolution": ("INT", {"default":32,"min":32,"max":128,"step":4}), + "sparse_structure_guidance_interval_start": ("FLOAT",{"default":0.10,"min":0.00,"max":1.00,"step":0.01}), + "sparse_structure_guidance_interval_end": ("FLOAT",{"default":1.00,"min":0.00,"max":1.00,"step":0.01}), + }, + } + + RETURN_TYPES = ("COORDS", "INT", "TRELLIS2PIPELINE",) + RETURN_NAMES = ("coords", "sparse_structure_resolution", "pipeline",) + FUNCTION = "process" + CATEGORY = "Trellis2Wrapper" + OUTPUT_NODE = True + + def process(self, pipeline, images, seed, + # sparse + sparse_structure_steps, + sparse_structure_guidance_strength, + sparse_structure_guidance_rescale, + sparse_structure_rescale_t, + sparse_structure_sampler, + sparse_structure_resolution, + sparse_structure_guidance_interval_start, + sparse_structure_guidance_interval_end, + ): + + self.seed_all(seed) + + self.load_vggt_model(pipeline) + + sparse_structure_guidance_interval = [sparse_structure_guidance_interval_start,sparse_structure_guidance_interval_end] + sparse_structure_sampler_params = {"steps":sparse_structure_steps,"guidance_strength":sparse_structure_guidance_strength,"guidance_rescale":sparse_structure_guidance_rescale,"guidance_interval":sparse_structure_guidance_interval,"rescale_t":sparse_structure_rescale_t} + + args = pipeline._pretrained_args + sparse_sampler_prefix = pipeline.GetSamplerName(sparse_structure_sampler) + pipeline.sparse_structure_sampler = getattr(samplers, f"Flow{sparse_sampler_prefix}GuidanceIntervalSampler")(**args['sparse_structure_sampler']['args']) + pipeline.load_sparse_structure_vggt_model() + pipeline.load_sparse_structure_vggt_cond() + + if images.ndim == 3: + images = images.unsqueeze(0) + + coords = self._run_ss_stage_direct(pipeline, images, sparse_structure_resolution, sparse_structure_sampler_params) + + if not pipeline.keep_models_loaded: + pipeline.unload_sparse_structure_vggt_model() + pipeline.unload_sparse_structure_vggt_cond() + self.unload_vggt_model(pipeline) + + return (coords, sparse_structure_resolution, pipeline,) + + def load_vggt_model(self, pipeline): + if pipeline.VGGT_model is None: + from .vggt.vggt.models.vggt import VGGT + pipeline.VGGT_dtype = torch.bfloat16 if torch.cuda.get_device_capability()[0] >= 8 else torch.float16 + model_path = os.path.join(folder_paths.models_dir,'recongenvia') + pipeline.VGGT_model = VGGT.from_pretrained(model_path) + pipeline.VGGT_model.to('cuda') + del pipeline.VGGT_model.depth_head + del pipeline.VGGT_model.track_head + pipeline.VGGT_model.eval() + + self._init_image_cond_model(pipeline) + + + def unload_vggt_model(self, pipeline): + del pipeline.VGGT_model + pipeline.VGGT_model = None + + del pipeline.models['image_cond_model_vggt'] + pipeline.models['image_cond_model_vggt'] = None + pipeline.image_cond_model_transform = None + + gc.collect() + + if torch.cuda.is_available(): + torch.cuda.synchronize() + torch.cuda.empty_cache() + + + def seed_all(self, seed: int = 0): + import random + """ + Set random seeds of all components. + """ + random.seed(seed) + np.random.seed(seed) + torch.manual_seed(seed) + torch.cuda.manual_seed_all(seed) + + @torch.no_grad() + def _run_ss_stage_direct( + self, + pipeline, + images, + target_ss_res: int, + ss_sampler_params: dict, + ) -> torch.Tensor: + """ + Run only ReconViaGen's sparse structure diffusion stage to obtain coords + directly, without proceeding to the SLAT/mesh stage. + + Returns: + coords : (N, 4) int tensor [batch_idx, x, y, z] in [0, target_ss_res) + """ + + cuda_device = torch.device('cuda') + + if pipeline.low_vram: + pipeline.VGGT_model.to(cuda_device) + + with torch.no_grad(): + with torch.cuda.amp.autocast(dtype=pipeline.VGGT_dtype): + aggregated_tokens_list, _ = self.vggt_feat(pipeline, images) + b, n, _, _ = aggregated_tokens_list[0].shape + image_cond = self.encode_image(pipeline, images).reshape(b, n, -1, 1024) + ss_cond = self.get_ss_cond(pipeline, image_cond[:, :, 5:], aggregated_tokens_list, 1) + + ss_flow_model = pipeline.models['sparse_structure_flow_vggt_model'] + sampler_params = {**pipeline.sparse_structure_sampler_params, **ss_sampler_params} + reso = ss_flow_model.resolution + ss_noise = torch.randn(1, ss_flow_model.in_channels, reso, reso, reso).to(cuda_device) + + with torch.autocast('cuda', dtype=torch.float16): + ss_latent = pipeline.sparse_structure_sampler.sample( + ss_flow_model, + ss_noise, + **ss_cond, + **sampler_params, + verbose=True, + ).samples + + decoder = pipeline.models['sparse_structure_decoder'] + decoded = decoder(ss_latent) > 0 + if target_ss_res != decoded.shape[2]: + ratio = decoded.shape[2] // target_ss_res + decoded = torch.nn.functional.max_pool3d(decoded.float(), ratio, ratio, 0) > 0.5 + coords = torch.argwhere(decoded)[:, [0, 2, 3, 4]].int() + + if pipeline.low_vram: + pipeline.VGGT_model.to('cpu') + decoder.to('cpu') + ss_cond = pipeline._cond_cpu(ss_cond) + torch.cuda.empty_cache() + + return coords + + @torch.no_grad() + def _run_ss_stage( + self, + pipeline, + images, + target_ss_res: int, + ss_sampler_params: dict, + slat_sampler_params: dict, + ) -> torch.Tensor: + """ + Generate a rough mesh via vggt_pipeline, then voxelise it into + surface-only coords at target_ss_res^3 for the downstream shape/tex stages. + + Returns: + coords : (N, 4) int tensor [batch_idx, x, y, z] in [0, target_ss_res) + """ + vp = self.vggt_pipeline + # vp.device is dynamic (inferred from model params), so when models are on + # CPU it returns 'cpu'. Hardcode the target cuda device instead. + cuda_device = torch.device('cuda') + + if self.low_vram: + self._vggt_models_to(cuda_device) + + outputs, _, _ = vp.run( + image=images, + formats=["mesh"], + preprocess_image=False, + sparse_structure_sampler_params=ss_sampler_params, + slat_sampler_params=slat_sampler_params, + ) + mesh_result = outputs["mesh"][0] + coords = self._mesh_to_surface_coords(mesh_result, target_ss_res, cuda_device) + + if self.low_vram: + self._vggt_models_to('cpu') + torch.cuda.empty_cache() + + return coords + + @torch.no_grad() + def vggt_feat(self, pipeline, image): + """ + Encode the image. + + Args: + image (Union[torch.Tensor, list[Image.Image]]): The image to encode + + Returns: + torch.Tensor: The encoded features. + """ + if isinstance(image, torch.Tensor): + assert image.ndim == 4, "Image tensor should be batched (B, H, W, C) or (B, C, H, W)" + # ComfyUI IMAGE tensors are (B, H, W, C); convert to (B, C, H, W) + if image.shape[-1] in (3, 4): + image = image.permute(0, 3, 1, 2) + image = F.interpolate(image, 518, mode='bilinear', align_corners=False) + image = image.to(pipeline.device) + elif isinstance(image, list): + assert all(isinstance(i, Image.Image) for i in image), "Image list should be list of PIL images" + image = [i.resize((518, 518), Image.LANCZOS) for i in image] + image = [np.array(i.convert('RGB')).astype(np.float32) / 255 for i in image] + image = [torch.from_numpy(i).permute(2, 0, 1).float() for i in image] + image = torch.stack(image).to(pipeline.device) + else: + raise ValueError(f"Unsupported type of image: {type(image)}") + + with torch.no_grad(): + with torch.cuda.amp.autocast(dtype=pipeline.VGGT_dtype): + # Predict attributes including cameras, depth maps, and point maps. + aggregated_tokens_list, _ = pipeline.VGGT_model.aggregator(image[None]) + + return aggregated_tokens_list, image + + def get_ss_cond(self, pipeline, image_cond: torch.Tensor, aggregated_tokens_list: list, num_samples: int) -> dict: + """ + Get the conditioning information for the model. + + Args: + image (Union[torch.Tensor, list[Image.Image]]): The image prompts. + + Returns: + dict: The conditioning information + """ + cond = pipeline.models['sparse_structure_vggt_cond'](aggregated_tokens_list, image_cond) + neg_cond = torch.zeros_like(cond) + return { + 'cond': cond, + 'neg_cond': neg_cond, + } + + def get_slat_cond(self, pipeline, image_cond: torch.Tensor, aggregated_tokens_list: list, num_samples: int) -> dict: + """ + Get the conditioning information for the model. + + Args: + image (Union[torch.Tensor, list[Image.Image]]): The image prompts. + + Returns: + dict: The conditioning information + """ + b, n, _, _ = aggregated_tokens_list[0].shape + cond = pipeline.models['slat_vggt_cond'](aggregated_tokens_list, image_cond).reshape(b, n, -1, 1024) + cond = [c.squeeze(1) for c in cond.split(1, dim=1)] + neg_cond = [torch.zeros_like(c) for c in cond] + return { + 'cond': cond, + 'neg_cond': neg_cond, + } + + @torch.no_grad() + def encode_image(self, pipeline, image, w_layernorm=True) -> torch.Tensor: + """ + Encode the image. + + Args: + image (Union[torch.Tensor, list[Image.Image]]): The image to encode + + Returns: + torch.Tensor: The encoded features. + """ + if isinstance(image, torch.Tensor): + assert image.ndim == 4, "Image tensor should be batched (B, H, W, C) or (B, C, H, W)" + # ComfyUI IMAGE tensors are (B, H, W, C); convert to (B, C, H, W) + if image.shape[-1] in (3, 4): + image = image.permute(0, 3, 1, 2) + image = F.interpolate(image, 518, mode='bilinear', align_corners=False) + image = image.to(pipeline.device) + elif isinstance(image, list): + assert all(isinstance(i, Image.Image) for i in image), "Image list should be list of PIL images" + image = [i.resize((518, 518), Image.LANCZOS) for i in image] + image = [np.array(i.convert('RGB')).astype(np.float32) / 255 for i in image] + image = [torch.from_numpy(i).permute(2, 0, 1).float() for i in image] + image = torch.stack(image).to(pipeline.device) + else: + raise ValueError(f"Unsupported type of image: {type(image)}") + + image = pipeline.image_cond_model_transform(image).to(pipeline.device) + pipeline.models['image_cond_model_vggt'].to(pipeline.device) + features = pipeline.models['image_cond_model_vggt'](image, is_training=True)['x_prenorm'] + if w_layernorm: + features = F.layer_norm(features, features.shape[-1:]) + return features + + def _init_image_cond_model(self, pipeline, name: str = "dinov2_vitl14_reg"): + """ + Initialize the image conditioning model. + """ + try: + dinov2_model = torch.hub.load(os.path.join(torch.hub.get_dir(), 'facebookresearch_dinov2_main'), name, source='local',pretrained=True) + except: + dinov2_model = torch.hub.load('facebookresearch/dinov2', name, pretrained=True) + dinov2_model.eval() + pipeline.models['image_cond_model_vggt'] = dinov2_model + transform = transforms.Compose([ + transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]), + ]) + pipeline.image_cond_model_transform = transform + NODE_CLASS_MAPPINGS = { "Trellis2LoadModel": Trellis2LoadModel, "Trellis2MeshWithVoxelGenerator": Trellis2MeshWithVoxelGenerator, @@ -4668,6 +5116,8 @@ NODE_CLASS_MAPPINGS = { "Trellis2RenderMultiView": Trellis2RenderMultiView, "Trellis2SaveImage": Trellis2SaveImage, "Trellis2VoxelToMesh": Trellis2VoxelToMesh, + "Trellis2UnloadAllModels": Trellis2UnloadAllModels, + "Trellis2SparseGeneratorWithReconViaGen": Trellis2SparseGeneratorWithReconViaGen, } @@ -4726,4 +5176,6 @@ NODE_DISPLAY_NAME_MAPPINGS = { "Trellis2RenderMultiView": "Trellis2 - Render MultiView", "Trellis2SaveImage": "Trellis2 - Save Image", "Trellis2VoxelToMesh": "Trellis2 - Voxel to Mesh", + "Trellis2UnloadAllModels": "Trellis2 - Unload All ComfyUI Models", + "Trellis2SparseGeneratorWithReconViaGen": "Trellis2 - Sparse Generator with ReconViaGen", } diff --git a/pyproject.toml b/pyproject.toml index d940e8d..570f346 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "trellis2" description = "ComfyUI Wrapper for Microsoft Trellis.2 - Native and Compact Structured Latents for 3D Generation" -version = "1.0.19" +version = "1.0.20" license = {file = "LICENSE"} # classifiers = [ # # For OS-independent nodes (works on all operating systems) diff --git a/reconviagen_pipeline.json b/reconviagen_pipeline.json new file mode 100644 index 0000000..d5a6aaa --- /dev/null +++ b/reconviagen_pipeline.json @@ -0,0 +1,98 @@ +{ + "name": "Trellis2ImageTo3DPipeline", + "args": { + "models": { + "sparse_structure_decoder": "microsoft/TRELLIS-image-large/ckpts/ss_dec_conv3d_16l8_fp16", + "sparse_structure_flow_model": "ckpts/ss_flow_img_dit_1_3B_64_bf16", + "shape_slat_decoder": "ckpts/shape_dec_next_dc_f16c32_fp16", + "shape_slat_flow_model_512": "ckpts/slat_flow_img2shape_dit_1_3B_512_bf16", + "shape_slat_flow_model_1024": "ckpts/slat_flow_img2shape_dit_1_3B_1024_bf16", + "tex_slat_decoder": "ckpts/tex_dec_next_dc_f16c32_fp16", + "tex_slat_flow_model_512": "ckpts/slat_flow_imgshape2tex_dit_1_3B_512_bf16", + "tex_slat_flow_model_1024": "ckpts/slat_flow_imgshape2tex_dit_1_3B_1024_bf16", + "sparse_structure_vggt_cond": "ckpts/ss_vggt_cond", + "slat_vggt_cond": "ckpts/slat_vggt_cond", + "sparse_structure_flow_vggt_model": "ckpts/ss_flow_img_dit_L_16l8_fp16" + }, + "sparse_structure_sampler": { + "name": "FlowEulerGuidanceIntervalSampler", + "args": { + "sigma_min": 1e-5 + }, + "params": { + "steps": 12, + "guidance_strength": 7.5, + "guidance_rescale": 0.7, + "guidance_interval": [0.3, 1.0], + "rescale_t": 5.0 + } + }, + "shape_slat_sampler": { + "name": "FlowEulerGuidanceIntervalSampler", + "args": { + "sigma_min": 1e-5 + }, + "params": { + "steps": 12, + "guidance_strength": 7.5, + "guidance_rescale": 0.5, + "guidance_interval": [0.3, 1.0], + "rescale_t": 3.0 + } + }, + "shape_slat_normalization": { + "mean": [ + 0.781296, 0.018091, -0.495192, -0.558457, 1.060530, 0.093252, 1.518149, -0.933218, + -0.732996, 2.604095, -0.118341, -2.143904, 0.495076, -2.179512, -2.130751, -0.996944, + 0.261421, -2.217463, 1.260067, -0.150213, 3.790713, 1.481266, -1.046058, -1.523667, + -0.059621, 2.220780, 1.621212, 0.877230, 0.567247, -3.175944, -3.186688, 1.578665 + ], + "std": [ + 5.972266, 4.706852, 5.445010, 5.209927, 5.320220, 4.547237, 5.020802, 5.444004, + 5.226681, 5.683095, 4.831436, 5.286469, 5.652043, 5.367606, 5.525084, 4.730578, + 4.805265, 5.124013, 5.530808, 5.619001, 5.103930, 5.417670, 5.269677, 5.547194, + 5.634698, 5.235274, 6.110351, 5.511298, 6.237273, 4.879207, 5.347008, 5.405691 + ] + }, + "tex_slat_sampler": { + "name": "FlowEulerGuidanceIntervalSampler", + "args": { + "sigma_min": 1e-5 + }, + "params": { + "steps": 12, + "guidance_strength": 1.0, + "guidance_rescale": 0.0, + "guidance_interval": [0.6, 0.9], + "rescale_t": 3.0 + } + }, + "tex_slat_normalization": { + "mean": [ + 3.501659, 2.212398, 2.226094, 0.251093, -0.026248, -0.687364, 0.439898, -0.928075, + 0.029398, -0.339596, -0.869527, 1.038479, -0.972385, 0.126042, -1.129303, 0.455149, + -1.209521, 2.069067, 0.544735, 2.569128, -0.323407, 2.293000, -1.925608, -1.217717, + 1.213905, 0.971588, -0.023631, 0.106750, 2.021786, 0.250524, -0.662387, -0.768862 + ], + "std": [ + 2.665652, 2.743913, 2.765121, 2.595319, 3.037293, 2.291316, 2.144656, 2.911822, + 2.969419, 2.501689, 2.154811, 3.163343, 2.621215, 2.381943, 3.186697, 3.021588, + 2.295916, 3.234985, 3.233086, 2.260140, 2.874801, 2.810596, 3.292720, 2.674999, + 2.680878, 2.372054, 2.451546, 2.353556, 2.995195, 2.379849, 2.786195, 2.775190 + ] + }, + "image_cond_model": { + "name": "DinoV3FeatureExtractor", + "args": { + "model_name": "facebook/dinov3-vitl16-pretrain-lvd1689m" + } + }, + "rembg_model": { + "name": "BiRefNet", + "args": { + "model_name": "briaai/RMBG-2.0" + } + }, + "default_pipeline_type": "1024_cascade" + } +} \ No newline at end of file diff --git a/trellis2/models/__init__.py b/trellis2/models/__init__.py index d4fed03..889b6c7 100644 --- a/trellis2/models/__init__.py +++ b/trellis2/models/__init__.py @@ -14,7 +14,10 @@ __attributes = { 'SparseUnetVaeEncoder': 'sc_vaes.sparse_unet_vae', 'SparseUnetVaeDecoder': 'sc_vaes.sparse_unet_vae', 'FlexiDualGridVaeEncoder': 'sc_vaes.fdg_vae', - 'FlexiDualGridVaeDecoder': 'sc_vaes.fdg_vae' + 'FlexiDualGridVaeDecoder': 'sc_vaes.fdg_vae', + + # vggt + 'ModulatedMultiViewCond': 'sparse_structure_flow', } __submodules = [] diff --git a/trellis2/models/sparse_structure_flow.py b/trellis2/models/sparse_structure_flow.py index 5c38c18..60baf1c 100644 --- a/trellis2/models/sparse_structure_flow.py +++ b/trellis2/models/sparse_structure_flow.py @@ -4,8 +4,8 @@ import torch import torch.nn as nn import torch.nn.functional as F import numpy as np -from ..modules.utils import convert_module_to, manual_cast, str_to_dtype -from ..modules.transformer import AbsolutePositionEmbedder, ModulatedTransformerCrossBlock +from ..modules.utils import convert_module_to, manual_cast, str_to_dtype, convert_module_to_f16 +from ..modules.transformer import AbsolutePositionEmbedder, ModulatedTransformerCrossBlock, ModulatedTransformerCrossBlock_woT from ..modules.attention import RotaryPositionEmbedder @@ -247,3 +247,76 @@ class SparseStructureFlowModel(nn.Module): h = h.permute(0, 2, 1).view(h.shape[0], h.shape[2], *[self.resolution] * 3).contiguous() return h + +class ModulatedMultiViewCond(nn.Module): + """ + Transformer cross-attention block (MSA + MCA + FFN) with adaptive layer norm conditioning. + """ + def __init__( + self, + channels: int, + ctx_channels: int, + num_heads: int, + mlp_ratio: float = 4.0, + attn_mode: Literal["full", "windowed"] = "full", + window_size: Optional[int] = None, + shift_window: Optional[Tuple[int, int, int]] = None, + use_checkpoint: bool = False, + use_rope: bool = False, + qk_rms_norm: bool = False, + qk_rms_norm_cross: bool = False, + qkv_bias: bool = True, + share_mod: bool = False, + num_init_tokens: int = 4096, + dtype: Optional[torch.dtype] = torch.float32, + use_fp16: bool = False, + ): + super().__init__() + self.cond_blocks = nn.ModuleList([ + ModulatedTransformerCrossBlock_woT( + channels, + ctx_channels, + num_heads=num_heads, + mlp_ratio=mlp_ratio, + attn_mode=attn_mode, + use_checkpoint=use_checkpoint, + use_rope=use_rope, + share_mod=share_mod, + qk_rms_norm=qk_rms_norm, + qk_rms_norm_cross=qk_rms_norm_cross, + ) + for _ in range(4) + ]) + self.use_fp16 = use_fp16 + if use_fp16: + self.dtype = torch.float16 + else: + self.dtype = dtype + self.multiview_cond_tokens = nn.Parameter(torch.randn(1, num_init_tokens, channels).to(dtype)) + nn.init.normal_(self.multiview_cond_tokens, std=1e-6) + self.intermediate_layer_idx = [4, 11, 17, 23] + if use_fp16: + self.convert_to_fp16() + + + def convert_to_fp16(self) -> None: + """ + Convert the torso of the model to float16. + """ + self.use_fp16 = True + self.dtype = torch.float16 + self.cond_blocks.apply(convert_module_to_f16) + self.multiview_cond_tokens = nn.Parameter(self.multiview_cond_tokens.data.to(self.dtype)) + def forward(self, aggregated_tokens_list: List, image_cond: torch.Tensor): + + b = aggregated_tokens_list[0].shape[0] + patch_start_idx = 5 + idx = 0 + cond = self.multiview_cond_tokens.repeat(b, 1, 1) + for layer_idx in self.intermediate_layer_idx: + x = aggregated_tokens_list[layer_idx][:, :, patch_start_idx:] + # x = x.reshape(b, -1, 2048) + torch.cat([image_cond.reshape(b, -1, 1024), image_cond.reshape(b, -1, 1024)],dim=-1) + x = torch.cat([x.reshape(b, -1, 2048), image_cond.reshape(b, -1, 1024)],dim=-1).to(self.dtype) + cond = self.cond_blocks[idx](cond, x) + idx = idx + 1 + return cond \ No newline at end of file diff --git a/trellis2/modules/transformer/modulated.py b/trellis2/modules/transformer/modulated.py index 3c56eda..e2f6923 100644 --- a/trellis2/modules/transformer/modulated.py +++ b/trellis2/modules/transformer/modulated.py @@ -5,7 +5,6 @@ from ..attention import MultiHeadAttention from ..norm import LayerNorm32 from .blocks import FeedForwardNet - class ModulatedTransformerBlock(nn.Module): """ Transformer block (MSA + FFN) with adaptive layer norm conditioning. @@ -162,4 +161,73 @@ class ModulatedTransformerCrossBlock(nn.Module): return torch.utils.checkpoint.checkpoint(self._forward, x, mod, context, phases, use_reentrant=False) else: return self._forward(x, mod, context, phases) - \ No newline at end of file + +class ModulatedTransformerCrossBlock_woT(nn.Module): + """ + Transformer cross-attention block (MSA + MCA + FFN) with adaptive layer norm conditioning. + """ + def __init__( + self, + channels: int, + ctx_channels: int, + num_heads: int, + mlp_ratio: float = 4.0, + attn_mode: Literal["full", "windowed"] = "full", + window_size: Optional[int] = None, + shift_window: Optional[Tuple[int, int, int]] = None, + use_checkpoint: bool = False, + use_rope: bool = False, + qk_rms_norm: bool = False, + qk_rms_norm_cross: bool = False, + qkv_bias: bool = True, + share_mod: bool = False, + ): + super().__init__() + self.use_checkpoint = use_checkpoint + self.share_mod = share_mod + self.norm1 = LayerNorm32(channels, elementwise_affine=False, eps=1e-6) + self.norm2 = LayerNorm32(channels, elementwise_affine=True, eps=1e-6) + self.norm3 = LayerNorm32(channels, elementwise_affine=False, eps=1e-6) + self.self_attn = MultiHeadAttention( + channels, + num_heads=num_heads, + type="self", + attn_mode=attn_mode, + window_size=window_size, + shift_window=shift_window, + qkv_bias=qkv_bias, + use_rope=use_rope, + qk_rms_norm=qk_rms_norm, + ) + self.cross_attn = MultiHeadAttention( + channels, + ctx_channels=ctx_channels, + num_heads=num_heads, + type="cross", + attn_mode="full", + qkv_bias=qkv_bias, + qk_rms_norm=qk_rms_norm_cross, + ) + self.mlp = FeedForwardNet( + channels, + mlp_ratio=mlp_ratio, + ) + + def _forward(self, x: torch.Tensor, context: torch.Tensor): + + h = self.norm1(x) + h = self.self_attn(h) + x = x + h + h = self.norm2(x) + h = self.cross_attn(h, context) + x = x + h + h = self.norm3(x) + h = self.mlp(h) + x = x + h + return x + + def forward(self, x: torch.Tensor, context: torch.Tensor): + if self.use_checkpoint: + return torch.utils.checkpoint.checkpoint(self._forward, x, context, use_reentrant=False) + else: + return self._forward(x, context) \ No newline at end of file diff --git a/trellis2/pipelines/trellis2_image_to_3d.py b/trellis2/pipelines/trellis2_image_to_3d.py index 10d02d7..595b914 100644 --- a/trellis2/pipelines/trellis2_image_to_3d.py +++ b/trellis2/pipelines/trellis2_image_to_3d.py @@ -26,6 +26,8 @@ import random from comfy.utils import ProgressBar +script_directory = os.path.dirname(os.path.abspath(__file__)) + def pil2tensor(image): return torch.from_numpy(np.array(image).astype(np.float32) / 255.0)[None,] @@ -98,6 +100,7 @@ class Trellis2ImageTo3DPipeline(Pipeline): self.rembg_model = rembg_model self._low_vram = low_vram self.default_pipeline_type = default_pipeline_type + self.VGGT_model = None self.pbr_attr_layout = { 'base_color': slice(0, 3), 'metallic': slice(3, 4), @@ -148,14 +151,16 @@ class Trellis2ImageTo3DPipeline(Pipeline): torch.cuda.empty_cache() @classmethod - def from_pretrained(cls, path: str, config_file: str = "pipeline.json", keep_models_loaded = True, use_fp8 = False) -> "Trellis2ImageTo3DPipeline": + def from_pretrained(cls, path: str, config_file: str = "pipeline.json", keep_models_loaded = True, use_fp8 = False, use_reconviagen = False) -> "Trellis2ImageTo3DPipeline": """ Load a pretrained model. Args: path (str): The path to the model. Can be either local path or a Hugging Face repository. """ - if use_fp8: + if use_reconviagen: + config_file = "reconviagen_pipeline.json" + elif use_fp8: config_file = "pipeline_fp8.json" pipeline = super().from_pretrained(path, config_file) @@ -212,6 +217,45 @@ class Trellis2ImageTo3DPipeline(Pipeline): self.models['sparse_structure_decoder'].to(self._device) if hasattr(self.models['sparse_structure_decoder'], 'low_vram'): self.models['sparse_structure_decoder'].low_vram = self.low_vram + + def load_sparse_structure_vggt_model(self): + if self.models['sparse_structure_flow_vggt_model'] is None: + print('Loading Sparse Structure VGGT model ...') + self.models['sparse_structure_flow_vggt_model'] = models.from_pretrained(f"{self.path}/{self._pretrained_args['models']['sparse_structure_flow_vggt_model']}") + self.models['sparse_structure_flow_vggt_model'].eval() + self.models['sparse_structure_flow_vggt_model'].to(self._device) + + if self.models['sparse_structure_decoder'] is None: + self.models['sparse_structure_decoder'] = models.from_pretrained(self._pretrained_args['models']['sparse_structure_decoder']) + self.models['sparse_structure_decoder'].eval() + self.models['sparse_structure_decoder'].to(self._device) + if hasattr(self.models['sparse_structure_decoder'], 'low_vram'): + self.models['sparse_structure_decoder'].low_vram = self.low_vram + + def unload_sparse_structure_vggt_model(self): + if self.models['sparse_structure_flow_vggt_model']: + del self.models['sparse_structure_flow_vggt_model'] + self.models['sparse_structure_flow_vggt_model'] = None + + if self.models['sparse_structure_decoder']: + del self.models['sparse_structure_decoder'] + self.models['sparse_structure_decoder'] = None + + self._cleanup_cuda() + + def load_sparse_structure_vggt_cond(self): + if self.models['sparse_structure_vggt_cond'] is None: + print('Loading Sparse Structure VGGT cond ...') + self.models['sparse_structure_vggt_cond'] = models.from_pretrained(f"{self.path}/{self._pretrained_args['models']['sparse_structure_vggt_cond']}") + self.models['sparse_structure_vggt_cond'].eval() + self.models['sparse_structure_vggt_cond'].to(self._device) + + def unload_sparse_structure_vggt_cond(self): + if self.models['sparse_structure_vggt_cond']: + del self.models['sparse_structure_vggt_cond'] + self.models['sparse_structure_vggt_cond'] = None + + self._cleanup_cuda() def unload_sparse_structure_model(self): if self.models['sparse_structure_flow_model']: diff --git a/vggt/vggt/heads/camera_head.py b/vggt/vggt/heads/camera_head.py new file mode 100644 index 0000000..153b020 --- /dev/null +++ b/vggt/vggt/heads/camera_head.py @@ -0,0 +1,162 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import math +import numpy as np + +import torch +import torch.nn as nn +import torch.nn.functional as F + +from ..layers import Mlp +from ..layers.block import Block +from ..heads.head_act import activate_pose + + +class CameraHead(nn.Module): + """ + CameraHead predicts camera parameters from token representations using iterative refinement. + + It applies a series of transformer blocks (the "trunk") to dedicated camera tokens. + """ + + def __init__( + self, + dim_in: int = 2048, + trunk_depth: int = 4, + pose_encoding_type: str = "absT_quaR_FoV", + num_heads: int = 16, + mlp_ratio: int = 4, + init_values: float = 0.01, + trans_act: str = "linear", + quat_act: str = "linear", + fl_act: str = "relu", # Field of view activations: ensures FOV values are positive. + ): + super().__init__() + + if pose_encoding_type == "absT_quaR_FoV": + self.target_dim = 9 + else: + raise ValueError(f"Unsupported camera encoding type: {pose_encoding_type}") + + self.trans_act = trans_act + self.quat_act = quat_act + self.fl_act = fl_act + self.trunk_depth = trunk_depth + + # Build the trunk using a sequence of transformer blocks. + self.trunk = nn.Sequential( + *[ + Block( + dim=dim_in, + num_heads=num_heads, + mlp_ratio=mlp_ratio, + init_values=init_values, + ) + for _ in range(trunk_depth) + ] + ) + + # Normalizations for camera token and trunk output. + self.token_norm = nn.LayerNorm(dim_in) + self.trunk_norm = nn.LayerNorm(dim_in) + + # Learnable empty camera pose token. + self.empty_pose_tokens = nn.Parameter(torch.zeros(1, 1, self.target_dim)) + self.embed_pose = nn.Linear(self.target_dim, dim_in) + + # Module for producing modulation parameters: shift, scale, and a gate. + self.poseLN_modulation = nn.Sequential(nn.SiLU(), nn.Linear(dim_in, 3 * dim_in, bias=True)) + + # Adaptive layer normalization without affine parameters. + self.adaln_norm = nn.LayerNorm(dim_in, elementwise_affine=False, eps=1e-6) + self.pose_branch = Mlp( + in_features=dim_in, + hidden_features=dim_in // 2, + out_features=self.target_dim, + drop=0, + ) + + def forward(self, aggregated_tokens_list: list, num_iterations: int = 4) -> list: + """ + Forward pass to predict camera parameters. + + Args: + aggregated_tokens_list (list): List of token tensors from the network; + the last tensor is used for prediction. + num_iterations (int, optional): Number of iterative refinement steps. Defaults to 4. + + Returns: + list: A list of predicted camera encodings (post-activation) from each iteration. + """ + # Use tokens from the last block for camera prediction. + tokens = aggregated_tokens_list[-1] + + # Extract the camera tokens + pose_tokens = tokens[:, :, 0] + pose_tokens = self.token_norm(pose_tokens) + + pred_pose_enc_list = self.trunk_fn(pose_tokens, num_iterations) + return pred_pose_enc_list + + def trunk_fn(self, pose_tokens: torch.Tensor, num_iterations: int) -> list: + """ + Iteratively refine camera pose predictions. + + Args: + pose_tokens (torch.Tensor): Normalized camera tokens with shape [B, 1, C]. + num_iterations (int): Number of refinement iterations. + + Returns: + list: List of activated camera encodings from each iteration. + """ + B, S, C = pose_tokens.shape # S is expected to be 1. + pred_pose_enc = None + pred_pose_enc_list = [] + + for _ in range(num_iterations): + # Use a learned empty pose for the first iteration. + if pred_pose_enc is None: + module_input = self.embed_pose(self.empty_pose_tokens.expand(B, S, -1)) + else: + # Detach the previous prediction to avoid backprop through time. + pred_pose_enc = pred_pose_enc.detach() + module_input = self.embed_pose(pred_pose_enc) + + # Generate modulation parameters and split them into shift, scale, and gate components. + shift_msa, scale_msa, gate_msa = self.poseLN_modulation(module_input).chunk(3, dim=-1) + + # Adaptive layer normalization and modulation. + pose_tokens_modulated = gate_msa * modulate(self.adaln_norm(pose_tokens), shift_msa, scale_msa) + pose_tokens_modulated = pose_tokens_modulated + pose_tokens + + pose_tokens_modulated = self.trunk(pose_tokens_modulated) + # Compute the delta update for the pose encoding. + pred_pose_enc_delta = self.pose_branch(self.trunk_norm(pose_tokens_modulated)) + + if pred_pose_enc is None: + pred_pose_enc = pred_pose_enc_delta + else: + pred_pose_enc = pred_pose_enc + pred_pose_enc_delta + + # Apply final activation functions for translation, quaternion, and field-of-view. + activated_pose = activate_pose( + pred_pose_enc, + trans_act=self.trans_act, + quat_act=self.quat_act, + fl_act=self.fl_act, + ) + pred_pose_enc_list.append(activated_pose) + + return pred_pose_enc_list + + +def modulate(x: torch.Tensor, shift: torch.Tensor, scale: torch.Tensor) -> torch.Tensor: + """ + Modulate the input tensor using scaling and shifting parameters. + """ + # modified from https://github.com/facebookresearch/DiT/blob/796c29e532f47bba17c5b9c5eb39b9354b8b7c64/models.py#L19 + return x * (1 + scale) + shift diff --git a/vggt/vggt/heads/dpt_head.py b/vggt/vggt/heads/dpt_head.py new file mode 100644 index 0000000..c8c8af9 --- /dev/null +++ b/vggt/vggt/heads/dpt_head.py @@ -0,0 +1,497 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + + +# Inspired by https://github.com/DepthAnything/Depth-Anything-V2 + + +import os +from typing import List, Dict, Tuple, Union + +import torch +import torch.nn as nn +import torch.nn.functional as F +from .head_act import activate_head +from .utils import create_uv_grid, position_grid_to_embed + + +class DPTHead(nn.Module): + """ + DPT Head for dense prediction tasks. + + This implementation follows the architecture described in "Vision Transformers for Dense Prediction" + (https://arxiv.org/abs/2103.13413). The DPT head processes features from a vision transformer + backbone and produces dense predictions by fusing multi-scale features. + + Args: + dim_in (int): Input dimension (channels). + patch_size (int, optional): Patch size. Default is 14. + output_dim (int, optional): Number of output channels. Default is 4. + activation (str, optional): Activation type. Default is "inv_log". + conf_activation (str, optional): Confidence activation type. Default is "expp1". + features (int, optional): Feature channels for intermediate representations. Default is 256. + out_channels (List[int], optional): Output channels for each intermediate layer. + intermediate_layer_idx (List[int], optional): Indices of layers from aggregated tokens used for DPT. + pos_embed (bool, optional): Whether to use positional embedding. Default is True. + feature_only (bool, optional): If True, return features only without the last several layers and activation head. Default is False. + down_ratio (int, optional): Downscaling factor for the output resolution. Default is 1. + """ + + def __init__( + self, + dim_in: int, + patch_size: int = 14, + output_dim: int = 4, + activation: str = "inv_log", + conf_activation: str = "expp1", + features: int = 256, + out_channels: List[int] = [256, 512, 1024, 1024], + intermediate_layer_idx: List[int] = [4, 11, 17, 23], + pos_embed: bool = True, + feature_only: bool = False, + down_ratio: int = 1, + ) -> None: + super(DPTHead, self).__init__() + self.patch_size = patch_size + self.activation = activation + self.conf_activation = conf_activation + self.pos_embed = pos_embed + self.feature_only = feature_only + self.down_ratio = down_ratio + self.intermediate_layer_idx = intermediate_layer_idx + + self.norm = nn.LayerNorm(dim_in) + + # Projection layers for each output channel from tokens. + self.projects = nn.ModuleList( + [ + nn.Conv2d( + in_channels=dim_in, + out_channels=oc, + kernel_size=1, + stride=1, + padding=0, + ) + for oc in out_channels + ] + ) + + # Resize layers for upsampling feature maps. + self.resize_layers = nn.ModuleList( + [ + nn.ConvTranspose2d( + in_channels=out_channels[0], out_channels=out_channels[0], kernel_size=4, stride=4, padding=0 + ), + nn.ConvTranspose2d( + in_channels=out_channels[1], out_channels=out_channels[1], kernel_size=2, stride=2, padding=0 + ), + nn.Identity(), + nn.Conv2d( + in_channels=out_channels[3], out_channels=out_channels[3], kernel_size=3, stride=2, padding=1 + ), + ] + ) + + self.scratch = _make_scratch( + out_channels, + features, + expand=False, + ) + + # Attach additional modules to scratch. + self.scratch.stem_transpose = None + self.scratch.refinenet1 = _make_fusion_block(features) + self.scratch.refinenet2 = _make_fusion_block(features) + self.scratch.refinenet3 = _make_fusion_block(features) + self.scratch.refinenet4 = _make_fusion_block(features, has_residual=False) + + head_features_1 = features + head_features_2 = 32 + + if feature_only: + self.scratch.output_conv1 = nn.Conv2d(head_features_1, head_features_1, kernel_size=3, stride=1, padding=1) + else: + self.scratch.output_conv1 = nn.Conv2d( + head_features_1, head_features_1 // 2, kernel_size=3, stride=1, padding=1 + ) + conv2_in_channels = head_features_1 // 2 + + self.scratch.output_conv2 = nn.Sequential( + nn.Conv2d(conv2_in_channels, head_features_2, kernel_size=3, stride=1, padding=1), + nn.ReLU(inplace=True), + nn.Conv2d(head_features_2, output_dim, kernel_size=1, stride=1, padding=0), + ) + + def forward( + self, + aggregated_tokens_list: List[torch.Tensor], + images: torch.Tensor, + patch_start_idx: int, + frames_chunk_size: int = 8, + ) -> Union[torch.Tensor, Tuple[torch.Tensor, torch.Tensor]]: + """ + Forward pass through the DPT head, supports processing by chunking frames. + Args: + aggregated_tokens_list (List[Tensor]): List of token tensors from different transformer layers. + images (Tensor): Input images with shape [B, S, 3, H, W], in range [0, 1]. + patch_start_idx (int): Starting index for patch tokens in the token sequence. + Used to separate patch tokens from other tokens (e.g., camera or register tokens). + frames_chunk_size (int, optional): Number of frames to process in each chunk. + If None or larger than S, all frames are processed at once. Default: 8. + + Returns: + Tensor or Tuple[Tensor, Tensor]: + - If feature_only=True: Feature maps with shape [B, S, C, H, W] + - Otherwise: Tuple of (predictions, confidence) both with shape [B, S, 1, H, W] + """ + B, S, _, H, W = images.shape + + # If frames_chunk_size is not specified or greater than S, process all frames at once + if frames_chunk_size is None or frames_chunk_size >= S: + return self._forward_impl(aggregated_tokens_list, images, patch_start_idx) + + # Otherwise, process frames in chunks to manage memory usage + assert frames_chunk_size > 0 + + # Process frames in batches + all_preds = [] + all_conf = [] + + for frames_start_idx in range(0, S, frames_chunk_size): + frames_end_idx = min(frames_start_idx + frames_chunk_size, S) + + # Process batch of frames + if self.feature_only: + chunk_output = self._forward_impl( + aggregated_tokens_list, images, patch_start_idx, frames_start_idx, frames_end_idx + ) + all_preds.append(chunk_output) + else: + chunk_preds, chunk_conf = self._forward_impl( + aggregated_tokens_list, images, patch_start_idx, frames_start_idx, frames_end_idx + ) + all_preds.append(chunk_preds) + all_conf.append(chunk_conf) + + # Concatenate results along the sequence dimension + if self.feature_only: + return torch.cat(all_preds, dim=1) + else: + return torch.cat(all_preds, dim=1), torch.cat(all_conf, dim=1) + + def _forward_impl( + self, + aggregated_tokens_list: List[torch.Tensor], + images: torch.Tensor, + patch_start_idx: int, + frames_start_idx: int = None, + frames_end_idx: int = None, + ) -> Union[torch.Tensor, Tuple[torch.Tensor, torch.Tensor]]: + """ + Implementation of the forward pass through the DPT head. + + This method processes a specific chunk of frames from the sequence. + + Args: + aggregated_tokens_list (List[Tensor]): List of token tensors from different transformer layers. + images (Tensor): Input images with shape [B, S, 3, H, W]. + patch_start_idx (int): Starting index for patch tokens. + frames_start_idx (int, optional): Starting index for frames to process. + frames_end_idx (int, optional): Ending index for frames to process. + + Returns: + Tensor or Tuple[Tensor, Tensor]: Feature maps or (predictions, confidence). + """ + if frames_start_idx is not None and frames_end_idx is not None: + images = images[:, frames_start_idx:frames_end_idx] + + B, S, _, H, W = images.shape + + patch_h, patch_w = H // self.patch_size, W // self.patch_size + + out = [] + dpt_idx = 0 + + for layer_idx in self.intermediate_layer_idx: + x = aggregated_tokens_list[layer_idx][:, :, patch_start_idx:] + + # Select frames if processing a chunk + if frames_start_idx is not None and frames_end_idx is not None: + x = x[:, frames_start_idx:frames_end_idx] + + x = x.view(B * S, -1, x.shape[-1]) + + x = self.norm(x) + + x = x.permute(0, 2, 1).reshape((x.shape[0], x.shape[-1], patch_h, patch_w)) + + x = self.projects[dpt_idx](x) + if self.pos_embed: + x = self._apply_pos_embed(x, W, H) + x = self.resize_layers[dpt_idx](x) + + out.append(x) + dpt_idx += 1 + + # Fuse features from multiple layers. + out = self.scratch_forward(out) + # Interpolate fused output to match target image resolution. + out = custom_interpolate( + out, + (int(patch_h * self.patch_size / self.down_ratio), int(patch_w * self.patch_size / self.down_ratio)), + mode="bilinear", + align_corners=True, + ) + + if self.pos_embed: + out = self._apply_pos_embed(out, W, H) + + if self.feature_only: + return out.view(B, S, *out.shape[1:]) + + out = self.scratch.output_conv2(out) + preds, conf = activate_head(out, activation=self.activation, conf_activation=self.conf_activation) + + preds = preds.view(B, S, *preds.shape[1:]) + conf = conf.view(B, S, *conf.shape[1:]) + return preds, conf + + def _apply_pos_embed(self, x: torch.Tensor, W: int, H: int, ratio: float = 0.1) -> torch.Tensor: + """ + Apply positional embedding to tensor x. + """ + patch_w = x.shape[-1] + patch_h = x.shape[-2] + pos_embed = create_uv_grid(patch_w, patch_h, aspect_ratio=W / H, dtype=x.dtype, device=x.device) + pos_embed = position_grid_to_embed(pos_embed, x.shape[1]) + pos_embed = pos_embed * ratio + pos_embed = pos_embed.permute(2, 0, 1)[None].expand(x.shape[0], -1, -1, -1) + return x + pos_embed + + def scratch_forward(self, features: List[torch.Tensor]) -> torch.Tensor: + """ + Forward pass through the fusion blocks. + + Args: + features (List[Tensor]): List of feature maps from different layers. + + Returns: + Tensor: Fused feature map. + """ + layer_1, layer_2, layer_3, layer_4 = features + + layer_1_rn = self.scratch.layer1_rn(layer_1) + layer_2_rn = self.scratch.layer2_rn(layer_2) + layer_3_rn = self.scratch.layer3_rn(layer_3) + layer_4_rn = self.scratch.layer4_rn(layer_4) + + out = self.scratch.refinenet4(layer_4_rn, size=layer_3_rn.shape[2:]) + del layer_4_rn, layer_4 + + out = self.scratch.refinenet3(out, layer_3_rn, size=layer_2_rn.shape[2:]) + del layer_3_rn, layer_3 + + out = self.scratch.refinenet2(out, layer_2_rn, size=layer_1_rn.shape[2:]) + del layer_2_rn, layer_2 + + out = self.scratch.refinenet1(out, layer_1_rn) + del layer_1_rn, layer_1 + + out = self.scratch.output_conv1(out) + return out + + +################################################################################ +# Modules +################################################################################ + + +def _make_fusion_block(features: int, size: int = None, has_residual: bool = True, groups: int = 1) -> nn.Module: + return FeatureFusionBlock( + features, + nn.ReLU(inplace=True), + deconv=False, + bn=False, + expand=False, + align_corners=True, + size=size, + has_residual=has_residual, + groups=groups, + ) + + +def _make_scratch(in_shape: List[int], out_shape: int, groups: int = 1, expand: bool = False) -> nn.Module: + scratch = nn.Module() + out_shape1 = out_shape + out_shape2 = out_shape + out_shape3 = out_shape + if len(in_shape) >= 4: + out_shape4 = out_shape + + if expand: + out_shape1 = out_shape + out_shape2 = out_shape * 2 + out_shape3 = out_shape * 4 + if len(in_shape) >= 4: + out_shape4 = out_shape * 8 + + scratch.layer1_rn = nn.Conv2d( + in_shape[0], out_shape1, kernel_size=3, stride=1, padding=1, bias=False, groups=groups + ) + scratch.layer2_rn = nn.Conv2d( + in_shape[1], out_shape2, kernel_size=3, stride=1, padding=1, bias=False, groups=groups + ) + scratch.layer3_rn = nn.Conv2d( + in_shape[2], out_shape3, kernel_size=3, stride=1, padding=1, bias=False, groups=groups + ) + if len(in_shape) >= 4: + scratch.layer4_rn = nn.Conv2d( + in_shape[3], out_shape4, kernel_size=3, stride=1, padding=1, bias=False, groups=groups + ) + return scratch + + +class ResidualConvUnit(nn.Module): + """Residual convolution module.""" + + def __init__(self, features, activation, bn, groups=1): + """Init. + + Args: + features (int): number of features + """ + super().__init__() + + self.bn = bn + self.groups = groups + self.conv1 = nn.Conv2d(features, features, kernel_size=3, stride=1, padding=1, bias=True, groups=self.groups) + self.conv2 = nn.Conv2d(features, features, kernel_size=3, stride=1, padding=1, bias=True, groups=self.groups) + + self.norm1 = None + self.norm2 = None + + self.activation = activation + self.skip_add = nn.quantized.FloatFunctional() + + def forward(self, x): + """Forward pass. + + Args: + x (tensor): input + + Returns: + tensor: output + """ + + out = self.activation(x) + out = self.conv1(out) + if self.norm1 is not None: + out = self.norm1(out) + + out = self.activation(out) + out = self.conv2(out) + if self.norm2 is not None: + out = self.norm2(out) + + return self.skip_add.add(out, x) + + +class FeatureFusionBlock(nn.Module): + """Feature fusion block.""" + + def __init__( + self, + features, + activation, + deconv=False, + bn=False, + expand=False, + align_corners=True, + size=None, + has_residual=True, + groups=1, + ): + """Init. + + Args: + features (int): number of features + """ + super(FeatureFusionBlock, self).__init__() + + self.deconv = deconv + self.align_corners = align_corners + self.groups = groups + self.expand = expand + out_features = features + if self.expand == True: + out_features = features // 2 + + self.out_conv = nn.Conv2d( + features, out_features, kernel_size=1, stride=1, padding=0, bias=True, groups=self.groups + ) + + if has_residual: + self.resConfUnit1 = ResidualConvUnit(features, activation, bn, groups=self.groups) + + self.has_residual = has_residual + self.resConfUnit2 = ResidualConvUnit(features, activation, bn, groups=self.groups) + + self.skip_add = nn.quantized.FloatFunctional() + self.size = size + + def forward(self, *xs, size=None): + """Forward pass. + + Returns: + tensor: output + """ + output = xs[0] + + if self.has_residual: + res = self.resConfUnit1(xs[1]) + output = self.skip_add.add(output, res) + + output = self.resConfUnit2(output) + + if (size is None) and (self.size is None): + modifier = {"scale_factor": 2} + elif size is None: + modifier = {"size": self.size} + else: + modifier = {"size": size} + + output = custom_interpolate(output, **modifier, mode="bilinear", align_corners=self.align_corners) + output = self.out_conv(output) + + return output + + +def custom_interpolate( + x: torch.Tensor, + size: Tuple[int, int] = None, + scale_factor: float = None, + mode: str = "bilinear", + align_corners: bool = True, +) -> torch.Tensor: + """ + Custom interpolate to avoid INT_MAX issues in nn.functional.interpolate. + """ + if size is None: + size = (int(x.shape[-2] * scale_factor), int(x.shape[-1] * scale_factor)) + + INT_MAX = 1610612736 + + input_elements = size[0] * size[1] * x.shape[0] * x.shape[1] + + if input_elements > INT_MAX: + chunks = torch.chunk(x, chunks=(input_elements // INT_MAX) + 1, dim=0) + interpolated_chunks = [ + nn.functional.interpolate(chunk, size=size, mode=mode, align_corners=align_corners) for chunk in chunks + ] + x = torch.cat(interpolated_chunks, dim=0) + return x.contiguous() + else: + return nn.functional.interpolate(x, size=size, mode=mode, align_corners=align_corners) diff --git a/vggt/vggt/heads/head_act.py b/vggt/vggt/heads/head_act.py new file mode 100644 index 0000000..2dedfcf --- /dev/null +++ b/vggt/vggt/heads/head_act.py @@ -0,0 +1,125 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + + +import torch +import torch.nn.functional as F + + +def activate_pose(pred_pose_enc, trans_act="linear", quat_act="linear", fl_act="linear"): + """ + Activate pose parameters with specified activation functions. + + Args: + pred_pose_enc: Tensor containing encoded pose parameters [translation, quaternion, focal length] + trans_act: Activation type for translation component + quat_act: Activation type for quaternion component + fl_act: Activation type for focal length component + + Returns: + Activated pose parameters tensor + """ + T = pred_pose_enc[..., :3] + quat = pred_pose_enc[..., 3:7] + fl = pred_pose_enc[..., 7:] # or fov + + T = base_pose_act(T, trans_act) + quat = base_pose_act(quat, quat_act) + fl = base_pose_act(fl, fl_act) # or fov + + pred_pose_enc = torch.cat([T, quat, fl], dim=-1) + + return pred_pose_enc + + +def base_pose_act(pose_enc, act_type="linear"): + """ + Apply basic activation function to pose parameters. + + Args: + pose_enc: Tensor containing encoded pose parameters + act_type: Activation type ("linear", "inv_log", "exp", "relu") + + Returns: + Activated pose parameters + """ + if act_type == "linear": + return pose_enc + elif act_type == "inv_log": + return inverse_log_transform(pose_enc) + elif act_type == "exp": + return torch.exp(pose_enc) + elif act_type == "relu": + return F.relu(pose_enc) + else: + raise ValueError(f"Unknown act_type: {act_type}") + + +def activate_head(out, activation="norm_exp", conf_activation="expp1"): + """ + Process network output to extract 3D points and confidence values. + + Args: + out: Network output tensor (B, C, H, W) + activation: Activation type for 3D points + conf_activation: Activation type for confidence values + + Returns: + Tuple of (3D points tensor, confidence tensor) + """ + # Move channels from last dim to the 4th dimension => (B, H, W, C) + fmap = out.permute(0, 2, 3, 1) # B,H,W,C expected + + # Split into xyz (first C-1 channels) and confidence (last channel) + xyz = fmap[:, :, :, :-1] + conf = fmap[:, :, :, -1] + + if activation == "norm_exp": + d = xyz.norm(dim=-1, keepdim=True).clamp(min=1e-8) + xyz_normed = xyz / d + pts3d = xyz_normed * torch.expm1(d) + elif activation == "norm": + pts3d = xyz / xyz.norm(dim=-1, keepdim=True) + elif activation == "exp": + pts3d = torch.exp(xyz) + elif activation == "relu": + pts3d = F.relu(xyz) + elif activation == "inv_log": + pts3d = inverse_log_transform(xyz) + elif activation == "xy_inv_log": + xy, z = xyz.split([2, 1], dim=-1) + z = inverse_log_transform(z) + pts3d = torch.cat([xy * z, z], dim=-1) + elif activation == "sigmoid": + pts3d = torch.sigmoid(xyz) + elif activation == "linear": + pts3d = xyz + else: + raise ValueError(f"Unknown activation: {activation}") + + if conf_activation == "expp1": + conf_out = 1 + conf.exp() + elif conf_activation == "expp0": + conf_out = conf.exp() + elif conf_activation == "sigmoid": + conf_out = torch.sigmoid(conf) + else: + raise ValueError(f"Unknown conf_activation: {conf_activation}") + + return pts3d, conf_out + + +def inverse_log_transform(y): + """ + Apply inverse log transform: sign(y) * (exp(|y|) - 1) + + Args: + y: Input tensor + + Returns: + Transformed tensor + """ + return torch.sign(y) * (torch.expm1(torch.abs(y))) diff --git a/vggt/vggt/heads/track_head.py b/vggt/vggt/heads/track_head.py new file mode 100644 index 0000000..9ec7199 --- /dev/null +++ b/vggt/vggt/heads/track_head.py @@ -0,0 +1,108 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch.nn as nn +from .dpt_head import DPTHead +from .track_modules.base_track_predictor import BaseTrackerPredictor + + +class TrackHead(nn.Module): + """ + Track head that uses DPT head to process tokens and BaseTrackerPredictor for tracking. + The tracking is performed iteratively, refining predictions over multiple iterations. + """ + + def __init__( + self, + dim_in, + patch_size=14, + features=128, + iters=4, + predict_conf=True, + stride=2, + corr_levels=7, + corr_radius=4, + hidden_size=384, + ): + """ + Initialize the TrackHead module. + + Args: + dim_in (int): Input dimension of tokens from the backbone. + patch_size (int): Size of image patches used in the vision transformer. + features (int): Number of feature channels in the feature extractor output. + iters (int): Number of refinement iterations for tracking predictions. + predict_conf (bool): Whether to predict confidence scores for tracked points. + stride (int): Stride value for the tracker predictor. + corr_levels (int): Number of correlation pyramid levels + corr_radius (int): Radius for correlation computation, controlling the search area. + hidden_size (int): Size of hidden layers in the tracker network. + """ + super().__init__() + + self.patch_size = patch_size + + # Feature extractor based on DPT architecture + # Processes tokens into feature maps for tracking + self.feature_extractor = DPTHead( + dim_in=dim_in, + patch_size=patch_size, + features=features, + feature_only=True, # Only output features, no activation + down_ratio=2, # Reduces spatial dimensions by factor of 2 + pos_embed=False, + ) + + # Tracker module that predicts point trajectories + # Takes feature maps and predicts coordinates and visibility + self.tracker = BaseTrackerPredictor( + latent_dim=features, # Match the output_dim of feature extractor + predict_conf=predict_conf, + stride=stride, + corr_levels=corr_levels, + corr_radius=corr_radius, + hidden_size=hidden_size, + ) + + self.iters = iters + + def forward(self, aggregated_tokens_list, images, patch_start_idx, query_points=None, iters=None): + """ + Forward pass of the TrackHead. + + Args: + aggregated_tokens_list (list): List of aggregated tokens from the backbone. + images (torch.Tensor): Input images of shape (B, S, C, H, W) where: + B = batch size, S = sequence length. + patch_start_idx (int): Starting index for patch tokens. + query_points (torch.Tensor, optional): Initial query points to track. + If None, points are initialized by the tracker. + iters (int, optional): Number of refinement iterations. If None, uses self.iters. + + Returns: + tuple: + - coord_preds (torch.Tensor): Predicted coordinates for tracked points. + - vis_scores (torch.Tensor): Visibility scores for tracked points. + - conf_scores (torch.Tensor): Confidence scores for tracked points (if predict_conf=True). + """ + B, S, _, H, W = images.shape + + # Extract features from tokens + # feature_maps has shape (B, S, C, H//2, W//2) due to down_ratio=2 + feature_maps = self.feature_extractor(aggregated_tokens_list, images, patch_start_idx) + + # Use default iterations if not specified + if iters is None: + iters = self.iters + + # Perform tracking using the extracted features + coord_preds, vis_scores, conf_scores = self.tracker( + query_points=query_points, + fmaps=feature_maps, + iters=iters, + ) + + return coord_preds, vis_scores, conf_scores diff --git a/vggt/vggt/heads/track_modules/__init__.py b/vggt/vggt/heads/track_modules/__init__.py new file mode 100644 index 0000000..0952fcc --- /dev/null +++ b/vggt/vggt/heads/track_modules/__init__.py @@ -0,0 +1,5 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. diff --git a/vggt/vggt/heads/track_modules/base_track_predictor.py b/vggt/vggt/heads/track_modules/base_track_predictor.py new file mode 100644 index 0000000..3ce8ec4 --- /dev/null +++ b/vggt/vggt/heads/track_modules/base_track_predictor.py @@ -0,0 +1,209 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch +import torch.nn as nn +from einops import rearrange, repeat + + +from .blocks import EfficientUpdateFormer, CorrBlock +from .utils import sample_features4d, get_2d_embedding, get_2d_sincos_pos_embed +from .modules import Mlp + + +class BaseTrackerPredictor(nn.Module): + def __init__( + self, + stride=1, + corr_levels=5, + corr_radius=4, + latent_dim=128, + hidden_size=384, + use_spaceatt=True, + depth=6, + max_scale=518, + predict_conf=True, + ): + super(BaseTrackerPredictor, self).__init__() + """ + The base template to create a track predictor + + Modified from https://github.com/facebookresearch/co-tracker/ + and https://github.com/facebookresearch/vggsfm + """ + + self.stride = stride + self.latent_dim = latent_dim + self.corr_levels = corr_levels + self.corr_radius = corr_radius + self.hidden_size = hidden_size + self.max_scale = max_scale + self.predict_conf = predict_conf + + self.flows_emb_dim = latent_dim // 2 + + self.corr_mlp = Mlp( + in_features=self.corr_levels * (self.corr_radius * 2 + 1) ** 2, + hidden_features=self.hidden_size, + out_features=self.latent_dim, + ) + + self.transformer_dim = self.latent_dim + self.latent_dim + self.latent_dim + 4 + + self.query_ref_token = nn.Parameter(torch.randn(1, 2, self.transformer_dim)) + + space_depth = depth if use_spaceatt else 0 + time_depth = depth + + self.updateformer = EfficientUpdateFormer( + space_depth=space_depth, + time_depth=time_depth, + input_dim=self.transformer_dim, + hidden_size=self.hidden_size, + output_dim=self.latent_dim + 2, + mlp_ratio=4.0, + add_space_attn=use_spaceatt, + ) + + self.fmap_norm = nn.LayerNorm(self.latent_dim) + self.ffeat_norm = nn.GroupNorm(1, self.latent_dim) + + # A linear layer to update track feats at each iteration + self.ffeat_updater = nn.Sequential(nn.Linear(self.latent_dim, self.latent_dim), nn.GELU()) + + self.vis_predictor = nn.Sequential(nn.Linear(self.latent_dim, 1)) + + if predict_conf: + self.conf_predictor = nn.Sequential(nn.Linear(self.latent_dim, 1)) + + def forward(self, query_points, fmaps=None, iters=6, return_feat=False, down_ratio=1, apply_sigmoid=True): + """ + query_points: B x N x 2, the number of batches, tracks, and xy + fmaps: B x S x C x HH x WW, the number of batches, frames, and feature dimension. + note HH and WW is the size of feature maps instead of original images + """ + B, N, D = query_points.shape + B, S, C, HH, WW = fmaps.shape + + assert D == 2, "Input points must be 2D coordinates" + + # apply a layernorm to fmaps here + fmaps = self.fmap_norm(fmaps.permute(0, 1, 3, 4, 2)) + fmaps = fmaps.permute(0, 1, 4, 2, 3) + + # Scale the input query_points because we may downsample the images + # by down_ratio or self.stride + # e.g., if a 3x1024x1024 image is processed to a 128x256x256 feature map + # its query_points should be query_points/4 + if down_ratio > 1: + query_points = query_points / float(down_ratio) + + query_points = query_points / float(self.stride) + + # Init with coords as the query points + # It means the search will start from the position of query points at the reference frames + coords = query_points.clone().reshape(B, 1, N, 2).repeat(1, S, 1, 1) + + # Sample/extract the features of the query points in the query frame + query_track_feat = sample_features4d(fmaps[:, 0], coords[:, 0]) + + # init track feats by query feats + track_feats = query_track_feat.unsqueeze(1).repeat(1, S, 1, 1) # B, S, N, C + # back up the init coords + coords_backup = coords.clone() + + fcorr_fn = CorrBlock(fmaps, num_levels=self.corr_levels, radius=self.corr_radius) + + coord_preds = [] + + # Iterative Refinement + for _ in range(iters): + # Detach the gradients from the last iteration + # (in my experience, not very important for performance) + coords = coords.detach() + + fcorrs = fcorr_fn.corr_sample(track_feats, coords) + + corr_dim = fcorrs.shape[3] + fcorrs_ = fcorrs.permute(0, 2, 1, 3).reshape(B * N, S, corr_dim) + fcorrs_ = self.corr_mlp(fcorrs_) + + # Movement of current coords relative to query points + flows = (coords - coords[:, 0:1]).permute(0, 2, 1, 3).reshape(B * N, S, 2) + + flows_emb = get_2d_embedding(flows, self.flows_emb_dim, cat_coords=False) + + # (In my trials, it is also okay to just add the flows_emb instead of concat) + flows_emb = torch.cat([flows_emb, flows / self.max_scale, flows / self.max_scale], dim=-1) + + track_feats_ = track_feats.permute(0, 2, 1, 3).reshape(B * N, S, self.latent_dim) + + # Concatenate them as the input for the transformers + transformer_input = torch.cat([flows_emb, fcorrs_, track_feats_], dim=2) + + # 2D positional embed + # TODO: this can be much simplified + pos_embed = get_2d_sincos_pos_embed(self.transformer_dim, grid_size=(HH, WW)).to(query_points.device) + sampled_pos_emb = sample_features4d(pos_embed.expand(B, -1, -1, -1), coords[:, 0]) + + sampled_pos_emb = rearrange(sampled_pos_emb, "b n c -> (b n) c").unsqueeze(1) + + x = transformer_input + sampled_pos_emb + + # Add the query ref token to the track feats + query_ref_token = torch.cat( + [self.query_ref_token[:, 0:1], self.query_ref_token[:, 1:2].expand(-1, S - 1, -1)], dim=1 + ) + x = x + query_ref_token.to(x.device).to(x.dtype) + + # B, N, S, C + x = rearrange(x, "(b n) s d -> b n s d", b=B) + + # Compute the delta coordinates and delta track features + delta, _ = self.updateformer(x) + + # BN, S, C + delta = rearrange(delta, " b n s d -> (b n) s d", b=B) + delta_coords_ = delta[:, :, :2] + delta_feats_ = delta[:, :, 2:] + + track_feats_ = track_feats_.reshape(B * N * S, self.latent_dim) + delta_feats_ = delta_feats_.reshape(B * N * S, self.latent_dim) + + # Update the track features + track_feats_ = self.ffeat_updater(self.ffeat_norm(delta_feats_)) + track_feats_ + + track_feats = track_feats_.reshape(B, N, S, self.latent_dim).permute(0, 2, 1, 3) # BxSxNxC + + # B x S x N x 2 + coords = coords + delta_coords_.reshape(B, N, S, 2).permute(0, 2, 1, 3) + + # Force coord0 as query + # because we assume the query points should not be changed + coords[:, 0] = coords_backup[:, 0] + + # The predicted tracks are in the original image scale + if down_ratio > 1: + coord_preds.append(coords * self.stride * down_ratio) + else: + coord_preds.append(coords * self.stride) + + # B, S, N + vis_e = self.vis_predictor(track_feats.reshape(B * S * N, self.latent_dim)).reshape(B, S, N) + if apply_sigmoid: + vis_e = torch.sigmoid(vis_e) + + if self.predict_conf: + conf_e = self.conf_predictor(track_feats.reshape(B * S * N, self.latent_dim)).reshape(B, S, N) + if apply_sigmoid: + conf_e = torch.sigmoid(conf_e) + else: + conf_e = None + + if return_feat: + return coord_preds, vis_e, track_feats, query_track_feat, conf_e + else: + return coord_preds, vis_e, conf_e diff --git a/vggt/vggt/heads/track_modules/blocks.py b/vggt/vggt/heads/track_modules/blocks.py new file mode 100644 index 0000000..8e7763f --- /dev/null +++ b/vggt/vggt/heads/track_modules/blocks.py @@ -0,0 +1,246 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. + +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + + +# Modified from https://github.com/facebookresearch/co-tracker/ + +import math +import torch +import torch.nn as nn +import torch.nn.functional as F + +from .utils import bilinear_sampler +from .modules import Mlp, AttnBlock, CrossAttnBlock, ResidualBlock + + +class EfficientUpdateFormer(nn.Module): + """ + Transformer model that updates track estimates. + """ + + def __init__( + self, + space_depth=6, + time_depth=6, + input_dim=320, + hidden_size=384, + num_heads=8, + output_dim=130, + mlp_ratio=4.0, + add_space_attn=True, + num_virtual_tracks=64, + ): + super().__init__() + + self.out_channels = 2 + self.num_heads = num_heads + self.hidden_size = hidden_size + self.add_space_attn = add_space_attn + + # Add input LayerNorm before linear projection + self.input_norm = nn.LayerNorm(input_dim) + self.input_transform = torch.nn.Linear(input_dim, hidden_size, bias=True) + + # Add output LayerNorm before final projection + self.output_norm = nn.LayerNorm(hidden_size) + self.flow_head = torch.nn.Linear(hidden_size, output_dim, bias=True) + self.num_virtual_tracks = num_virtual_tracks + + if self.add_space_attn: + self.virual_tracks = nn.Parameter(torch.randn(1, num_virtual_tracks, 1, hidden_size)) + else: + self.virual_tracks = None + + self.time_blocks = nn.ModuleList( + [ + AttnBlock( + hidden_size, + num_heads, + mlp_ratio=mlp_ratio, + attn_class=nn.MultiheadAttention, + ) + for _ in range(time_depth) + ] + ) + + if add_space_attn: + self.space_virtual_blocks = nn.ModuleList( + [ + AttnBlock( + hidden_size, + num_heads, + mlp_ratio=mlp_ratio, + attn_class=nn.MultiheadAttention, + ) + for _ in range(space_depth) + ] + ) + self.space_point2virtual_blocks = nn.ModuleList( + [CrossAttnBlock(hidden_size, hidden_size, num_heads, mlp_ratio=mlp_ratio) for _ in range(space_depth)] + ) + self.space_virtual2point_blocks = nn.ModuleList( + [CrossAttnBlock(hidden_size, hidden_size, num_heads, mlp_ratio=mlp_ratio) for _ in range(space_depth)] + ) + assert len(self.time_blocks) >= len(self.space_virtual2point_blocks) + self.initialize_weights() + + def initialize_weights(self): + def _basic_init(module): + if isinstance(module, nn.Linear): + torch.nn.init.xavier_uniform_(module.weight) + if module.bias is not None: + nn.init.constant_(module.bias, 0) + torch.nn.init.trunc_normal_(self.flow_head.weight, std=0.001) + + self.apply(_basic_init) + + def forward(self, input_tensor, mask=None): + # Apply input LayerNorm + input_tensor = self.input_norm(input_tensor) + tokens = self.input_transform(input_tensor) + + init_tokens = tokens + + B, _, T, _ = tokens.shape + + if self.add_space_attn: + virtual_tokens = self.virual_tracks.repeat(B, 1, T, 1) + tokens = torch.cat([tokens, virtual_tokens], dim=1) + + _, N, _, _ = tokens.shape + + j = 0 + for i in range(len(self.time_blocks)): + time_tokens = tokens.contiguous().view(B * N, T, -1) # B N T C -> (B N) T C + + time_tokens = self.time_blocks[i](time_tokens) + + tokens = time_tokens.view(B, N, T, -1) # (B N) T C -> B N T C + if self.add_space_attn and (i % (len(self.time_blocks) // len(self.space_virtual_blocks)) == 0): + space_tokens = tokens.permute(0, 2, 1, 3).contiguous().view(B * T, N, -1) # B N T C -> (B T) N C + point_tokens = space_tokens[:, : N - self.num_virtual_tracks] + virtual_tokens = space_tokens[:, N - self.num_virtual_tracks :] + + virtual_tokens = self.space_virtual2point_blocks[j](virtual_tokens, point_tokens, mask=mask) + virtual_tokens = self.space_virtual_blocks[j](virtual_tokens) + point_tokens = self.space_point2virtual_blocks[j](point_tokens, virtual_tokens, mask=mask) + + space_tokens = torch.cat([point_tokens, virtual_tokens], dim=1) + tokens = space_tokens.view(B, T, N, -1).permute(0, 2, 1, 3) # (B T) N C -> B N T C + j += 1 + + if self.add_space_attn: + tokens = tokens[:, : N - self.num_virtual_tracks] + + tokens = tokens + init_tokens + + # Apply output LayerNorm before final projection + tokens = self.output_norm(tokens) + flow = self.flow_head(tokens) + + return flow, None + + +class CorrBlock: + def __init__(self, fmaps, num_levels=4, radius=4, multiple_track_feats=False, padding_mode="zeros"): + """ + Build a pyramid of feature maps from the input. + + fmaps: Tensor (B, S, C, H, W) + num_levels: number of pyramid levels (each downsampled by factor 2) + radius: search radius for sampling correlation + multiple_track_feats: if True, split the target features per pyramid level + padding_mode: passed to grid_sample / bilinear_sampler + """ + B, S, C, H, W = fmaps.shape + self.S, self.C, self.H, self.W = S, C, H, W + self.num_levels = num_levels + self.radius = radius + self.padding_mode = padding_mode + self.multiple_track_feats = multiple_track_feats + + # Build pyramid: each level is half the spatial resolution of the previous + self.fmaps_pyramid = [fmaps] # level 0 is full resolution + current_fmaps = fmaps + for i in range(num_levels - 1): + B, S, C, H, W = current_fmaps.shape + # Merge batch & sequence dimensions + current_fmaps = current_fmaps.reshape(B * S, C, H, W) + # Avg pool down by factor 2 + current_fmaps = F.avg_pool2d(current_fmaps, kernel_size=2, stride=2) + _, _, H_new, W_new = current_fmaps.shape + current_fmaps = current_fmaps.reshape(B, S, C, H_new, W_new) + self.fmaps_pyramid.append(current_fmaps) + + # Precompute a delta grid (of shape (2r+1, 2r+1, 2)) for sampling. + # This grid is added to the (scaled) coordinate centroids. + r = self.radius + dx = torch.linspace(-r, r, 2 * r + 1, device=fmaps.device, dtype=fmaps.dtype) + dy = torch.linspace(-r, r, 2 * r + 1, device=fmaps.device, dtype=fmaps.dtype) + # delta: for every (dy,dx) displacement (i.e. Δx, Δy) + self.delta = torch.stack(torch.meshgrid(dy, dx, indexing="ij"), dim=-1) # shape: (2r+1, 2r+1, 2) + + def corr_sample(self, targets, coords): + """ + Instead of storing the entire correlation pyramid, we compute each level's correlation + volume, sample it immediately, then discard it. This saves GPU memory. + + Args: + targets: Tensor (B, S, N, C) — features for the current targets. + coords: Tensor (B, S, N, 2) — coordinates at full resolution. + + Returns: + Tensor (B, S, N, L) where L = num_levels * (2*radius+1)**2 (concatenated sampled correlations) + """ + B, S, N, C = targets.shape + + # If you have multiple track features, split them per level. + if self.multiple_track_feats: + targets_split = torch.split(targets, C // self.num_levels, dim=-1) + + out_pyramid = [] + for i, fmaps in enumerate(self.fmaps_pyramid): + # Get current spatial resolution H, W for this pyramid level. + B, S, C, H, W = fmaps.shape + # Reshape feature maps for correlation computation: + # fmap2s: (B, S, C, H*W) + fmap2s = fmaps.view(B, S, C, H * W) + # Choose appropriate target features. + fmap1 = targets_split[i] if self.multiple_track_feats else targets # shape: (B, S, N, C) + + # Compute correlation directly + corrs = compute_corr_level(fmap1, fmap2s, C) + corrs = corrs.view(B, S, N, H, W) + + # Prepare sampling grid: + # Scale down the coordinates for the current level. + centroid_lvl = coords.reshape(B * S * N, 1, 1, 2) / (2**i) + # Make sure our precomputed delta grid is on the same device/dtype. + delta_lvl = self.delta.to(coords.device).to(coords.dtype) + # Now the grid for grid_sample is: + # coords_lvl = centroid_lvl + delta_lvl (broadcasted over grid) + coords_lvl = centroid_lvl + delta_lvl.view(1, 2 * self.radius + 1, 2 * self.radius + 1, 2) + + # Sample from the correlation volume using bilinear interpolation. + # We reshape corrs to (B * S * N, 1, H, W) so grid_sample acts over each target. + corrs_sampled = bilinear_sampler( + corrs.reshape(B * S * N, 1, H, W), coords_lvl, padding_mode=self.padding_mode + ) + # The sampled output is (B * S * N, 1, 2r+1, 2r+1). Flatten the last two dims. + corrs_sampled = corrs_sampled.view(B, S, N, -1) # Now shape: (B, S, N, (2r+1)^2) + out_pyramid.append(corrs_sampled) + + # Concatenate all levels along the last dimension. + out = torch.cat(out_pyramid, dim=-1).contiguous() + return out + + +def compute_corr_level(fmap1, fmap2s, C): + # fmap1: (B, S, N, C) + # fmap2s: (B, S, C, H*W) + corrs = torch.matmul(fmap1, fmap2s) # (B, S, N, H*W) + corrs = corrs.view(fmap1.shape[0], fmap1.shape[1], fmap1.shape[2], -1) # (B, S, N, H*W) + return corrs / math.sqrt(C) diff --git a/vggt/vggt/heads/track_modules/modules.py b/vggt/vggt/heads/track_modules/modules.py new file mode 100644 index 0000000..4b090dd --- /dev/null +++ b/vggt/vggt/heads/track_modules/modules.py @@ -0,0 +1,218 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + + +import torch +import torch.nn as nn +import torch.nn.functional as F +from functools import partial +from typing import Callable +import collections +from torch import Tensor +from itertools import repeat + + +# From PyTorch internals +def _ntuple(n): + def parse(x): + if isinstance(x, collections.abc.Iterable) and not isinstance(x, str): + return tuple(x) + return tuple(repeat(x, n)) + + return parse + + +def exists(val): + return val is not None + + +def default(val, d): + return val if exists(val) else d + + +to_2tuple = _ntuple(2) + + +class ResidualBlock(nn.Module): + """ + ResidualBlock: construct a block of two conv layers with residual connections + """ + + def __init__(self, in_planes, planes, norm_fn="group", stride=1, kernel_size=3): + super(ResidualBlock, self).__init__() + + self.conv1 = nn.Conv2d( + in_planes, + planes, + kernel_size=kernel_size, + padding=1, + stride=stride, + padding_mode="zeros", + ) + self.conv2 = nn.Conv2d( + planes, + planes, + kernel_size=kernel_size, + padding=1, + padding_mode="zeros", + ) + self.relu = nn.ReLU(inplace=True) + + num_groups = planes // 8 + + if norm_fn == "group": + self.norm1 = nn.GroupNorm(num_groups=num_groups, num_channels=planes) + self.norm2 = nn.GroupNorm(num_groups=num_groups, num_channels=planes) + if not stride == 1: + self.norm3 = nn.GroupNorm(num_groups=num_groups, num_channels=planes) + + elif norm_fn == "batch": + self.norm1 = nn.BatchNorm2d(planes) + self.norm2 = nn.BatchNorm2d(planes) + if not stride == 1: + self.norm3 = nn.BatchNorm2d(planes) + + elif norm_fn == "instance": + self.norm1 = nn.InstanceNorm2d(planes) + self.norm2 = nn.InstanceNorm2d(planes) + if not stride == 1: + self.norm3 = nn.InstanceNorm2d(planes) + + elif norm_fn == "none": + self.norm1 = nn.Sequential() + self.norm2 = nn.Sequential() + if not stride == 1: + self.norm3 = nn.Sequential() + else: + raise NotImplementedError + + if stride == 1: + self.downsample = None + else: + self.downsample = nn.Sequential( + nn.Conv2d(in_planes, planes, kernel_size=1, stride=stride), + self.norm3, + ) + + def forward(self, x): + y = x + y = self.relu(self.norm1(self.conv1(y))) + y = self.relu(self.norm2(self.conv2(y))) + + if self.downsample is not None: + x = self.downsample(x) + + return self.relu(x + y) + + +class Mlp(nn.Module): + """MLP as used in Vision Transformer, MLP-Mixer and related networks""" + + def __init__( + self, + in_features, + hidden_features=None, + out_features=None, + act_layer=nn.GELU, + norm_layer=None, + bias=True, + drop=0.0, + use_conv=False, + ): + super().__init__() + out_features = out_features or in_features + hidden_features = hidden_features or in_features + bias = to_2tuple(bias) + drop_probs = to_2tuple(drop) + linear_layer = partial(nn.Conv2d, kernel_size=1) if use_conv else nn.Linear + + self.fc1 = linear_layer(in_features, hidden_features, bias=bias[0]) + self.act = act_layer() + self.drop1 = nn.Dropout(drop_probs[0]) + self.fc2 = linear_layer(hidden_features, out_features, bias=bias[1]) + self.drop2 = nn.Dropout(drop_probs[1]) + + def forward(self, x): + x = self.fc1(x) + x = self.act(x) + x = self.drop1(x) + x = self.fc2(x) + x = self.drop2(x) + return x + + +class AttnBlock(nn.Module): + def __init__( + self, + hidden_size, + num_heads, + attn_class: Callable[..., nn.Module] = nn.MultiheadAttention, + mlp_ratio=4.0, + **block_kwargs + ): + """ + Self attention block + """ + super().__init__() + + self.norm1 = nn.LayerNorm(hidden_size) + self.norm2 = nn.LayerNorm(hidden_size) + + self.attn = attn_class(embed_dim=hidden_size, num_heads=num_heads, batch_first=True, **block_kwargs) + + mlp_hidden_dim = int(hidden_size * mlp_ratio) + + self.mlp = Mlp(in_features=hidden_size, hidden_features=mlp_hidden_dim, drop=0) + + def forward(self, x, mask=None): + # Prepare the mask for PyTorch's attention (it expects a different format) + # attn_mask = mask if mask is not None else None + # Normalize before attention + x = self.norm1(x) + + # PyTorch's MultiheadAttention returns attn_output, attn_output_weights + # attn_output, _ = self.attn(x, x, x, attn_mask=attn_mask) + + attn_output, _ = self.attn(x, x, x) + + # Add & Norm + x = x + attn_output + x = x + self.mlp(self.norm2(x)) + return x + + +class CrossAttnBlock(nn.Module): + def __init__(self, hidden_size, context_dim, num_heads=1, mlp_ratio=4.0, **block_kwargs): + """ + Cross attention block + """ + super().__init__() + + self.norm1 = nn.LayerNorm(hidden_size) + self.norm_context = nn.LayerNorm(hidden_size) + self.norm2 = nn.LayerNorm(hidden_size) + + self.cross_attn = nn.MultiheadAttention( + embed_dim=hidden_size, num_heads=num_heads, batch_first=True, **block_kwargs + ) + + mlp_hidden_dim = int(hidden_size * mlp_ratio) + + self.mlp = Mlp(in_features=hidden_size, hidden_features=mlp_hidden_dim, drop=0) + + def forward(self, x, context, mask=None): + # Normalize inputs + x = self.norm1(x) + context = self.norm_context(context) + + # Apply cross attention + # Note: nn.MultiheadAttention returns attn_output, attn_output_weights + attn_output, _ = self.cross_attn(x, context, context, attn_mask=mask) + + # Add & Norm + x = x + attn_output + x = x + self.mlp(self.norm2(x)) + return x diff --git a/vggt/vggt/heads/track_modules/utils.py b/vggt/vggt/heads/track_modules/utils.py new file mode 100644 index 0000000..51d01d3 --- /dev/null +++ b/vggt/vggt/heads/track_modules/utils.py @@ -0,0 +1,226 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +# Modified from https://github.com/facebookresearch/vggsfm +# and https://github.com/facebookresearch/co-tracker/tree/main + + +import torch +import torch.nn as nn +import torch.nn.functional as F + +from typing import Optional, Tuple, Union + + +def get_2d_sincos_pos_embed(embed_dim: int, grid_size: Union[int, Tuple[int, int]], return_grid=False) -> torch.Tensor: + """ + This function initializes a grid and generates a 2D positional embedding using sine and cosine functions. + It is a wrapper of get_2d_sincos_pos_embed_from_grid. + Args: + - embed_dim: The embedding dimension. + - grid_size: The grid size. + Returns: + - pos_embed: The generated 2D positional embedding. + """ + if isinstance(grid_size, tuple): + grid_size_h, grid_size_w = grid_size + else: + grid_size_h = grid_size_w = grid_size + grid_h = torch.arange(grid_size_h, dtype=torch.float) + grid_w = torch.arange(grid_size_w, dtype=torch.float) + grid = torch.meshgrid(grid_w, grid_h, indexing="xy") + grid = torch.stack(grid, dim=0) + grid = grid.reshape([2, 1, grid_size_h, grid_size_w]) + pos_embed = get_2d_sincos_pos_embed_from_grid(embed_dim, grid) + if return_grid: + return ( + pos_embed.reshape(1, grid_size_h, grid_size_w, -1).permute(0, 3, 1, 2), + grid, + ) + return pos_embed.reshape(1, grid_size_h, grid_size_w, -1).permute(0, 3, 1, 2) + + +def get_2d_sincos_pos_embed_from_grid(embed_dim: int, grid: torch.Tensor) -> torch.Tensor: + """ + This function generates a 2D positional embedding from a given grid using sine and cosine functions. + + Args: + - embed_dim: The embedding dimension. + - grid: The grid to generate the embedding from. + + Returns: + - emb: The generated 2D positional embedding. + """ + assert embed_dim % 2 == 0 + + # use half of dimensions to encode grid_h + emb_h = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[0]) # (H*W, D/2) + emb_w = get_1d_sincos_pos_embed_from_grid(embed_dim // 2, grid[1]) # (H*W, D/2) + + emb = torch.cat([emb_h, emb_w], dim=2) # (H*W, D) + return emb + + +def get_1d_sincos_pos_embed_from_grid(embed_dim: int, pos: torch.Tensor) -> torch.Tensor: + """ + This function generates a 1D positional embedding from a given grid using sine and cosine functions. + + Args: + - embed_dim: The embedding dimension. + - pos: The position to generate the embedding from. + + Returns: + - emb: The generated 1D positional embedding. + """ + assert embed_dim % 2 == 0 + omega = torch.arange(embed_dim // 2, dtype=torch.double) + omega /= embed_dim / 2.0 + omega = 1.0 / 10000**omega # (D/2,) + + pos = pos.reshape(-1) # (M,) + out = torch.einsum("m,d->md", pos, omega) # (M, D/2), outer product + + emb_sin = torch.sin(out) # (M, D/2) + emb_cos = torch.cos(out) # (M, D/2) + + emb = torch.cat([emb_sin, emb_cos], dim=1) # (M, D) + return emb[None].float() + + +def get_2d_embedding(xy: torch.Tensor, C: int, cat_coords: bool = True) -> torch.Tensor: + """ + This function generates a 2D positional embedding from given coordinates using sine and cosine functions. + + Args: + - xy: The coordinates to generate the embedding from. + - C: The size of the embedding. + - cat_coords: A flag to indicate whether to concatenate the original coordinates to the embedding. + + Returns: + - pe: The generated 2D positional embedding. + """ + B, N, D = xy.shape + assert D == 2 + + x = xy[:, :, 0:1] + y = xy[:, :, 1:2] + div_term = (torch.arange(0, C, 2, device=xy.device, dtype=torch.float32) * (1000.0 / C)).reshape(1, 1, int(C / 2)) + + pe_x = torch.zeros(B, N, C, device=xy.device, dtype=torch.float32) + pe_y = torch.zeros(B, N, C, device=xy.device, dtype=torch.float32) + + pe_x[:, :, 0::2] = torch.sin(x * div_term) + pe_x[:, :, 1::2] = torch.cos(x * div_term) + + pe_y[:, :, 0::2] = torch.sin(y * div_term) + pe_y[:, :, 1::2] = torch.cos(y * div_term) + + pe = torch.cat([pe_x, pe_y], dim=2) # (B, N, C*3) + if cat_coords: + pe = torch.cat([xy, pe], dim=2) # (B, N, C*3+3) + return pe + + +def bilinear_sampler(input, coords, align_corners=True, padding_mode="border"): + r"""Sample a tensor using bilinear interpolation + + `bilinear_sampler(input, coords)` samples a tensor :attr:`input` at + coordinates :attr:`coords` using bilinear interpolation. It is the same + as `torch.nn.functional.grid_sample()` but with a different coordinate + convention. + + The input tensor is assumed to be of shape :math:`(B, C, H, W)`, where + :math:`B` is the batch size, :math:`C` is the number of channels, + :math:`H` is the height of the image, and :math:`W` is the width of the + image. The tensor :attr:`coords` of shape :math:`(B, H_o, W_o, 2)` is + interpreted as an array of 2D point coordinates :math:`(x_i,y_i)`. + + Alternatively, the input tensor can be of size :math:`(B, C, T, H, W)`, + in which case sample points are triplets :math:`(t_i,x_i,y_i)`. Note + that in this case the order of the components is slightly different + from `grid_sample()`, which would expect :math:`(x_i,y_i,t_i)`. + + If `align_corners` is `True`, the coordinate :math:`x` is assumed to be + in the range :math:`[0,W-1]`, with 0 corresponding to the center of the + left-most image pixel :math:`W-1` to the center of the right-most + pixel. + + If `align_corners` is `False`, the coordinate :math:`x` is assumed to + be in the range :math:`[0,W]`, with 0 corresponding to the left edge of + the left-most pixel :math:`W` to the right edge of the right-most + pixel. + + Similar conventions apply to the :math:`y` for the range + :math:`[0,H-1]` and :math:`[0,H]` and to :math:`t` for the range + :math:`[0,T-1]` and :math:`[0,T]`. + + Args: + input (Tensor): batch of input images. + coords (Tensor): batch of coordinates. + align_corners (bool, optional): Coordinate convention. Defaults to `True`. + padding_mode (str, optional): Padding mode. Defaults to `"border"`. + + Returns: + Tensor: sampled points. + """ + coords = coords.detach().clone() + ############################################################ + # IMPORTANT: + coords = coords.to(input.device).to(input.dtype) + ############################################################ + + sizes = input.shape[2:] + + assert len(sizes) in [2, 3] + + if len(sizes) == 3: + # t x y -> x y t to match dimensions T H W in grid_sample + coords = coords[..., [1, 2, 0]] + + if align_corners: + scale = torch.tensor( + [2 / max(size - 1, 1) for size in reversed(sizes)], device=coords.device, dtype=coords.dtype + ) + else: + scale = torch.tensor([2 / size for size in reversed(sizes)], device=coords.device, dtype=coords.dtype) + + coords.mul_(scale) # coords = coords * scale + coords.sub_(1) # coords = coords - 1 + + return F.grid_sample(input, coords, align_corners=align_corners, padding_mode=padding_mode) + + +def sample_features4d(input, coords): + r"""Sample spatial features + + `sample_features4d(input, coords)` samples the spatial features + :attr:`input` represented by a 4D tensor :math:`(B, C, H, W)`. + + The field is sampled at coordinates :attr:`coords` using bilinear + interpolation. :attr:`coords` is assumed to be of shape :math:`(B, R, + 2)`, where each sample has the format :math:`(x_i, y_i)`. This uses the + same convention as :func:`bilinear_sampler` with `align_corners=True`. + + The output tensor has one feature per point, and has shape :math:`(B, + R, C)`. + + Args: + input (Tensor): spatial features. + coords (Tensor): points. + + Returns: + Tensor: sampled features. + """ + + B, _, _, _ = input.shape + + # B R 2 -> B R 1 2 + coords = coords.unsqueeze(2) + + # B C R 1 + feats = bilinear_sampler(input, coords) + + return feats.permute(0, 2, 1, 3).view(B, -1, feats.shape[1] * feats.shape[3]) # B C R 1 -> B R C diff --git a/vggt/vggt/heads/utils.py b/vggt/vggt/heads/utils.py new file mode 100644 index 0000000..d7af1f6 --- /dev/null +++ b/vggt/vggt/heads/utils.py @@ -0,0 +1,108 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch +import torch.nn as nn + + +def position_grid_to_embed(pos_grid: torch.Tensor, embed_dim: int, omega_0: float = 100) -> torch.Tensor: + """ + Convert 2D position grid (HxWx2) to sinusoidal embeddings (HxWxC) + + Args: + pos_grid: Tensor of shape (H, W, 2) containing 2D coordinates + embed_dim: Output channel dimension for embeddings + + Returns: + Tensor of shape (H, W, embed_dim) with positional embeddings + """ + H, W, grid_dim = pos_grid.shape + assert grid_dim == 2 + pos_flat = pos_grid.reshape(-1, grid_dim) # Flatten to (H*W, 2) + + # Process x and y coordinates separately + emb_x = make_sincos_pos_embed(embed_dim // 2, pos_flat[:, 0], omega_0=omega_0) # [1, H*W, D/2] + emb_y = make_sincos_pos_embed(embed_dim // 2, pos_flat[:, 1], omega_0=omega_0) # [1, H*W, D/2] + + # Combine and reshape + emb = torch.cat([emb_x, emb_y], dim=-1) # [1, H*W, D] + + return emb.view(H, W, embed_dim) # [H, W, D] + + +def make_sincos_pos_embed(embed_dim: int, pos: torch.Tensor, omega_0: float = 100) -> torch.Tensor: + """ + This function generates a 1D positional embedding from a given grid using sine and cosine functions. + + Args: + - embed_dim: The embedding dimension. + - pos: The position to generate the embedding from. + + Returns: + - emb: The generated 1D positional embedding. + """ + assert embed_dim % 2 == 0 + omega = torch.arange(embed_dim // 2, dtype=torch.double, device=pos.device) + omega /= embed_dim / 2.0 + omega = 1.0 / omega_0**omega # (D/2,) + + pos = pos.reshape(-1) # (M,) + out = torch.einsum("m,d->md", pos, omega) # (M, D/2), outer product + + emb_sin = torch.sin(out) # (M, D/2) + emb_cos = torch.cos(out) # (M, D/2) + + emb = torch.cat([emb_sin, emb_cos], dim=1) # (M, D) + return emb.float() + + +# Inspired by https://github.com/microsoft/moge + + +def create_uv_grid( + width: int, height: int, aspect_ratio: float = None, dtype: torch.dtype = None, device: torch.device = None +) -> torch.Tensor: + """ + Create a normalized UV grid of shape (width, height, 2). + + The grid spans horizontally and vertically according to an aspect ratio, + ensuring the top-left corner is at (-x_span, -y_span) and the bottom-right + corner is at (x_span, y_span), normalized by the diagonal of the plane. + + Args: + width (int): Number of points horizontally. + height (int): Number of points vertically. + aspect_ratio (float, optional): Width-to-height ratio. Defaults to width/height. + dtype (torch.dtype, optional): Data type of the resulting tensor. + device (torch.device, optional): Device on which the tensor is created. + + Returns: + torch.Tensor: A (width, height, 2) tensor of UV coordinates. + """ + # Derive aspect ratio if not explicitly provided + if aspect_ratio is None: + aspect_ratio = float(width) / float(height) + + # Compute normalized spans for X and Y + diag_factor = (aspect_ratio**2 + 1.0) ** 0.5 + span_x = aspect_ratio / diag_factor + span_y = 1.0 / diag_factor + + # Establish the linspace boundaries + left_x = -span_x * (width - 1) / width + right_x = span_x * (width - 1) / width + top_y = -span_y * (height - 1) / height + bottom_y = span_y * (height - 1) / height + + # Generate 1D coordinates + x_coords = torch.linspace(left_x, right_x, steps=width, dtype=dtype, device=device) + y_coords = torch.linspace(top_y, bottom_y, steps=height, dtype=dtype, device=device) + + # Create 2D meshgrid (width x height) and stack into UV + uu, vv = torch.meshgrid(x_coords, y_coords, indexing="xy") + uv_grid = torch.stack((uu, vv), dim=-1) + + return uv_grid diff --git a/vggt/vggt/layers/__init__.py b/vggt/vggt/layers/__init__.py new file mode 100644 index 0000000..8120f4b --- /dev/null +++ b/vggt/vggt/layers/__init__.py @@ -0,0 +1,11 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +from .mlp import Mlp +from .patch_embed import PatchEmbed +from .swiglu_ffn import SwiGLUFFN, SwiGLUFFNFused +from .block import NestedTensorBlock +from .attention import MemEffAttention diff --git a/vggt/vggt/layers/attention.py b/vggt/vggt/layers/attention.py new file mode 100644 index 0000000..ab3089c --- /dev/null +++ b/vggt/vggt/layers/attention.py @@ -0,0 +1,98 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/models/vision_transformer.py + +import logging +import os +import warnings + +from torch import Tensor +from torch import nn +import torch.nn.functional as F + +XFORMERS_AVAILABLE = False + + +class Attention(nn.Module): + def __init__( + self, + dim: int, + num_heads: int = 8, + qkv_bias: bool = True, + proj_bias: bool = True, + attn_drop: float = 0.0, + proj_drop: float = 0.0, + norm_layer: nn.Module = nn.LayerNorm, + qk_norm: bool = False, + fused_attn: bool = True, # use F.scaled_dot_product_attention or not + rope=None, + ) -> None: + super().__init__() + assert dim % num_heads == 0, "dim should be divisible by num_heads" + self.num_heads = num_heads + self.head_dim = dim // num_heads + self.scale = self.head_dim**-0.5 + self.fused_attn = fused_attn + + self.qkv = nn.Linear(dim, dim * 3, bias=qkv_bias) + self.q_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity() + self.k_norm = norm_layer(self.head_dim) if qk_norm else nn.Identity() + self.attn_drop = nn.Dropout(attn_drop) + self.proj = nn.Linear(dim, dim, bias=proj_bias) + self.proj_drop = nn.Dropout(proj_drop) + self.rope = rope + + def forward(self, x: Tensor, pos=None) -> Tensor: + B, N, C = x.shape + qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, self.head_dim).permute(2, 0, 3, 1, 4) + q, k, v = qkv.unbind(0) + q, k = self.q_norm(q), self.k_norm(k) + + if self.rope is not None: + q = self.rope(q, pos) + k = self.rope(k, pos) + + if self.fused_attn: + x = F.scaled_dot_product_attention( + q, + k, + v, + dropout_p=self.attn_drop.p if self.training else 0.0, + ) + else: + q = q * self.scale + attn = q @ k.transpose(-2, -1) + attn = attn.softmax(dim=-1) + attn = self.attn_drop(attn) + x = attn @ v + + x = x.transpose(1, 2).reshape(B, N, C) + x = self.proj(x) + x = self.proj_drop(x) + return x + + +class MemEffAttention(Attention): + def forward(self, x: Tensor, attn_bias=None, pos=None) -> Tensor: + assert pos is None + if not XFORMERS_AVAILABLE: + if attn_bias is not None: + raise AssertionError("xFormers is required for using nested tensors") + return super().forward(x) + + B, N, C = x.shape + qkv = self.qkv(x).reshape(B, N, 3, self.num_heads, C // self.num_heads) + + q, k, v = unbind(qkv, 2) + + x = memory_efficient_attention(q, k, v, attn_bias=attn_bias) + x = x.reshape([B, N, C]) + + x = self.proj(x) + x = self.proj_drop(x) + return x diff --git a/vggt/vggt/layers/block.py b/vggt/vggt/layers/block.py new file mode 100644 index 0000000..5f89e4d --- /dev/null +++ b/vggt/vggt/layers/block.py @@ -0,0 +1,259 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/patch_embed.py + +import logging +import os +from typing import Callable, List, Any, Tuple, Dict +import warnings + +import torch +from torch import nn, Tensor + +from .attention import Attention +from .drop_path import DropPath +from .layer_scale import LayerScale +from .mlp import Mlp + + +XFORMERS_AVAILABLE = False + + +class Block(nn.Module): + def __init__( + self, + dim: int, + num_heads: int, + mlp_ratio: float = 4.0, + qkv_bias: bool = True, + proj_bias: bool = True, + ffn_bias: bool = True, + drop: float = 0.0, + attn_drop: float = 0.0, + init_values=None, + drop_path: float = 0.0, + act_layer: Callable[..., nn.Module] = nn.GELU, + norm_layer: Callable[..., nn.Module] = nn.LayerNorm, + attn_class: Callable[..., nn.Module] = Attention, + ffn_layer: Callable[..., nn.Module] = Mlp, + qk_norm: bool = False, + fused_attn: bool = True, # use F.scaled_dot_product_attention or not + rope=None, + ) -> None: + super().__init__() + + self.norm1 = norm_layer(dim) + + self.attn = attn_class( + dim, + num_heads=num_heads, + qkv_bias=qkv_bias, + proj_bias=proj_bias, + attn_drop=attn_drop, + proj_drop=drop, + qk_norm=qk_norm, + fused_attn=fused_attn, + rope=rope, + ) + + self.ls1 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity() + self.drop_path1 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity() + + self.norm2 = norm_layer(dim) + mlp_hidden_dim = int(dim * mlp_ratio) + self.mlp = ffn_layer( + in_features=dim, + hidden_features=mlp_hidden_dim, + act_layer=act_layer, + drop=drop, + bias=ffn_bias, + ) + self.ls2 = LayerScale(dim, init_values=init_values) if init_values else nn.Identity() + self.drop_path2 = DropPath(drop_path) if drop_path > 0.0 else nn.Identity() + + self.sample_drop_ratio = drop_path + + def forward(self, x: Tensor, pos=None) -> Tensor: + def attn_residual_func(x: Tensor, pos=None) -> Tensor: + return self.ls1(self.attn(self.norm1(x), pos=pos)) + + def ffn_residual_func(x: Tensor) -> Tensor: + return self.ls2(self.mlp(self.norm2(x))) + + if self.training and self.sample_drop_ratio > 0.1: + # the overhead is compensated only for a drop path rate larger than 0.1 + x = drop_add_residual_stochastic_depth( + x, + pos=pos, + residual_func=attn_residual_func, + sample_drop_ratio=self.sample_drop_ratio, + ) + x = drop_add_residual_stochastic_depth( + x, + residual_func=ffn_residual_func, + sample_drop_ratio=self.sample_drop_ratio, + ) + elif self.training and self.sample_drop_ratio > 0.0: + x = x + self.drop_path1(attn_residual_func(x, pos=pos)) + x = x + self.drop_path1(ffn_residual_func(x)) # FIXME: drop_path2 + else: + x = x + attn_residual_func(x, pos=pos) + x = x + ffn_residual_func(x) + return x + + +def drop_add_residual_stochastic_depth( + x: Tensor, + residual_func: Callable[[Tensor], Tensor], + sample_drop_ratio: float = 0.0, + pos=None, +) -> Tensor: + # 1) extract subset using permutation + b, n, d = x.shape + sample_subset_size = max(int(b * (1 - sample_drop_ratio)), 1) + brange = (torch.randperm(b, device=x.device))[:sample_subset_size] + x_subset = x[brange] + + # 2) apply residual_func to get residual + if pos is not None: + # if necessary, apply rope to the subset + pos = pos[brange] + residual = residual_func(x_subset, pos=pos) + else: + residual = residual_func(x_subset) + + x_flat = x.flatten(1) + residual = residual.flatten(1) + + residual_scale_factor = b / sample_subset_size + + # 3) add the residual + x_plus_residual = torch.index_add(x_flat, 0, brange, residual.to(dtype=x.dtype), alpha=residual_scale_factor) + return x_plus_residual.view_as(x) + + +def get_branges_scales(x, sample_drop_ratio=0.0): + b, n, d = x.shape + sample_subset_size = max(int(b * (1 - sample_drop_ratio)), 1) + brange = (torch.randperm(b, device=x.device))[:sample_subset_size] + residual_scale_factor = b / sample_subset_size + return brange, residual_scale_factor + + +def add_residual(x, brange, residual, residual_scale_factor, scaling_vector=None): + if scaling_vector is None: + x_flat = x.flatten(1) + residual = residual.flatten(1) + x_plus_residual = torch.index_add(x_flat, 0, brange, residual.to(dtype=x.dtype), alpha=residual_scale_factor) + else: + x_plus_residual = scaled_index_add( + x, brange, residual.to(dtype=x.dtype), scaling=scaling_vector, alpha=residual_scale_factor + ) + return x_plus_residual + + +attn_bias_cache: Dict[Tuple, Any] = {} + + +def get_attn_bias_and_cat(x_list, branges=None): + """ + this will perform the index select, cat the tensors, and provide the attn_bias from cache + """ + batch_sizes = [b.shape[0] for b in branges] if branges is not None else [x.shape[0] for x in x_list] + all_shapes = tuple((b, x.shape[1]) for b, x in zip(batch_sizes, x_list)) + if all_shapes not in attn_bias_cache.keys(): + seqlens = [] + for b, x in zip(batch_sizes, x_list): + for _ in range(b): + seqlens.append(x.shape[1]) + attn_bias = fmha.BlockDiagonalMask.from_seqlens(seqlens) + attn_bias._batch_sizes = batch_sizes + attn_bias_cache[all_shapes] = attn_bias + + if branges is not None: + cat_tensors = index_select_cat([x.flatten(1) for x in x_list], branges).view(1, -1, x_list[0].shape[-1]) + else: + tensors_bs1 = tuple(x.reshape([1, -1, *x.shape[2:]]) for x in x_list) + cat_tensors = torch.cat(tensors_bs1, dim=1) + + return attn_bias_cache[all_shapes], cat_tensors + + +def drop_add_residual_stochastic_depth_list( + x_list: List[Tensor], + residual_func: Callable[[Tensor, Any], Tensor], + sample_drop_ratio: float = 0.0, + scaling_vector=None, +) -> Tensor: + # 1) generate random set of indices for dropping samples in the batch + branges_scales = [get_branges_scales(x, sample_drop_ratio=sample_drop_ratio) for x in x_list] + branges = [s[0] for s in branges_scales] + residual_scale_factors = [s[1] for s in branges_scales] + + # 2) get attention bias and index+concat the tensors + attn_bias, x_cat = get_attn_bias_and_cat(x_list, branges) + + # 3) apply residual_func to get residual, and split the result + residual_list = attn_bias.split(residual_func(x_cat, attn_bias=attn_bias)) # type: ignore + + outputs = [] + for x, brange, residual, residual_scale_factor in zip(x_list, branges, residual_list, residual_scale_factors): + outputs.append(add_residual(x, brange, residual, residual_scale_factor, scaling_vector).view_as(x)) + return outputs + + +class NestedTensorBlock(Block): + def forward_nested(self, x_list: List[Tensor]) -> List[Tensor]: + """ + x_list contains a list of tensors to nest together and run + """ + assert isinstance(self.attn, MemEffAttention) + + if self.training and self.sample_drop_ratio > 0.0: + + def attn_residual_func(x: Tensor, attn_bias=None) -> Tensor: + return self.attn(self.norm1(x), attn_bias=attn_bias) + + def ffn_residual_func(x: Tensor, attn_bias=None) -> Tensor: + return self.mlp(self.norm2(x)) + + x_list = drop_add_residual_stochastic_depth_list( + x_list, + residual_func=attn_residual_func, + sample_drop_ratio=self.sample_drop_ratio, + scaling_vector=self.ls1.gamma if isinstance(self.ls1, LayerScale) else None, + ) + x_list = drop_add_residual_stochastic_depth_list( + x_list, + residual_func=ffn_residual_func, + sample_drop_ratio=self.sample_drop_ratio, + scaling_vector=self.ls2.gamma if isinstance(self.ls1, LayerScale) else None, + ) + return x_list + else: + + def attn_residual_func(x: Tensor, attn_bias=None) -> Tensor: + return self.ls1(self.attn(self.norm1(x), attn_bias=attn_bias)) + + def ffn_residual_func(x: Tensor, attn_bias=None) -> Tensor: + return self.ls2(self.mlp(self.norm2(x))) + + attn_bias, x = get_attn_bias_and_cat(x_list) + x = x + attn_residual_func(x, attn_bias=attn_bias) + x = x + ffn_residual_func(x) + return attn_bias.split(x) + + def forward(self, x_or_x_list): + if isinstance(x_or_x_list, Tensor): + return super().forward(x_or_x_list) + elif isinstance(x_or_x_list, list): + if not XFORMERS_AVAILABLE: + raise AssertionError("xFormers is required for using nested tensors") + return self.forward_nested(x_or_x_list) + else: + raise AssertionError diff --git a/vggt/vggt/layers/drop_path.py b/vggt/vggt/layers/drop_path.py new file mode 100644 index 0000000..1d640e0 --- /dev/null +++ b/vggt/vggt/layers/drop_path.py @@ -0,0 +1,34 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/drop.py + + +from torch import nn + + +def drop_path(x, drop_prob: float = 0.0, training: bool = False): + if drop_prob == 0.0 or not training: + return x + keep_prob = 1 - drop_prob + shape = (x.shape[0],) + (1,) * (x.ndim - 1) # work with diff dim tensors, not just 2D ConvNets + random_tensor = x.new_empty(shape).bernoulli_(keep_prob) + if keep_prob > 0.0: + random_tensor.div_(keep_prob) + output = x * random_tensor + return output + + +class DropPath(nn.Module): + """Drop paths (Stochastic Depth) per sample (when applied in main path of residual blocks).""" + + def __init__(self, drop_prob=None): + super(DropPath, self).__init__() + self.drop_prob = drop_prob + + def forward(self, x): + return drop_path(x, self.drop_prob, self.training) diff --git a/vggt/vggt/layers/layer_scale.py b/vggt/vggt/layers/layer_scale.py new file mode 100644 index 0000000..51df0d7 --- /dev/null +++ b/vggt/vggt/layers/layer_scale.py @@ -0,0 +1,27 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# Modified from: https://github.com/huggingface/pytorch-image-models/blob/main/timm/models/vision_transformer.py#L103-L110 + +from typing import Union + +import torch +from torch import Tensor +from torch import nn + + +class LayerScale(nn.Module): + def __init__( + self, + dim: int, + init_values: Union[float, Tensor] = 1e-5, + inplace: bool = False, + ) -> None: + super().__init__() + self.inplace = inplace + self.gamma = nn.Parameter(init_values * torch.ones(dim)) + + def forward(self, x: Tensor) -> Tensor: + return x.mul_(self.gamma) if self.inplace else x * self.gamma diff --git a/vggt/vggt/layers/mlp.py b/vggt/vggt/layers/mlp.py new file mode 100644 index 0000000..bbf9432 --- /dev/null +++ b/vggt/vggt/layers/mlp.py @@ -0,0 +1,40 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/mlp.py + + +from typing import Callable, Optional + +from torch import Tensor, nn + + +class Mlp(nn.Module): + def __init__( + self, + in_features: int, + hidden_features: Optional[int] = None, + out_features: Optional[int] = None, + act_layer: Callable[..., nn.Module] = nn.GELU, + drop: float = 0.0, + bias: bool = True, + ) -> None: + super().__init__() + out_features = out_features or in_features + hidden_features = hidden_features or in_features + self.fc1 = nn.Linear(in_features, hidden_features, bias=bias) + self.act = act_layer() + self.fc2 = nn.Linear(hidden_features, out_features, bias=bias) + self.drop = nn.Dropout(drop) + + def forward(self, x: Tensor) -> Tensor: + x = self.fc1(x) + x = self.act(x) + x = self.drop(x) + x = self.fc2(x) + x = self.drop(x) + return x diff --git a/vggt/vggt/layers/patch_embed.py b/vggt/vggt/layers/patch_embed.py new file mode 100644 index 0000000..8b7c080 --- /dev/null +++ b/vggt/vggt/layers/patch_embed.py @@ -0,0 +1,88 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/master/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/layers/patch_embed.py + +from typing import Callable, Optional, Tuple, Union + +from torch import Tensor +import torch.nn as nn + + +def make_2tuple(x): + if isinstance(x, tuple): + assert len(x) == 2 + return x + + assert isinstance(x, int) + return (x, x) + + +class PatchEmbed(nn.Module): + """ + 2D image to patch embedding: (B,C,H,W) -> (B,N,D) + + Args: + img_size: Image size. + patch_size: Patch token size. + in_chans: Number of input image channels. + embed_dim: Number of linear projection output channels. + norm_layer: Normalization layer. + """ + + def __init__( + self, + img_size: Union[int, Tuple[int, int]] = 224, + patch_size: Union[int, Tuple[int, int]] = 16, + in_chans: int = 3, + embed_dim: int = 768, + norm_layer: Optional[Callable] = None, + flatten_embedding: bool = True, + ) -> None: + super().__init__() + + image_HW = make_2tuple(img_size) + patch_HW = make_2tuple(patch_size) + patch_grid_size = ( + image_HW[0] // patch_HW[0], + image_HW[1] // patch_HW[1], + ) + + self.img_size = image_HW + self.patch_size = patch_HW + self.patches_resolution = patch_grid_size + self.num_patches = patch_grid_size[0] * patch_grid_size[1] + + self.in_chans = in_chans + self.embed_dim = embed_dim + + self.flatten_embedding = flatten_embedding + + self.proj = nn.Conv2d(in_chans, embed_dim, kernel_size=patch_HW, stride=patch_HW) + self.norm = norm_layer(embed_dim) if norm_layer else nn.Identity() + + def forward(self, x: Tensor) -> Tensor: + _, _, H, W = x.shape + patch_H, patch_W = self.patch_size + + assert H % patch_H == 0, f"Input image height {H} is not a multiple of patch height {patch_H}" + assert W % patch_W == 0, f"Input image width {W} is not a multiple of patch width: {patch_W}" + + x = self.proj(x) # B C H W + H, W = x.size(2), x.size(3) + x = x.flatten(2).transpose(1, 2) # B HW C + x = self.norm(x) + if not self.flatten_embedding: + x = x.reshape(-1, H, W, self.embed_dim) # B H W C + return x + + def flops(self) -> float: + Ho, Wo = self.patches_resolution + flops = Ho * Wo * self.embed_dim * self.in_chans * (self.patch_size[0] * self.patch_size[1]) + if self.norm is not None: + flops += Ho * Wo * self.embed_dim + return flops diff --git a/vggt/vggt/layers/rope.py b/vggt/vggt/layers/rope.py new file mode 100644 index 0000000..4d5d333 --- /dev/null +++ b/vggt/vggt/layers/rope.py @@ -0,0 +1,188 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + + +# Implementation of 2D Rotary Position Embeddings (RoPE). + +# This module provides a clean implementation of 2D Rotary Position Embeddings, +# which extends the original RoPE concept to handle 2D spatial positions. + +# Inspired by: +# https://github.com/meta-llama/codellama/blob/main/llama/model.py +# https://github.com/naver-ai/rope-vit + + +import numpy as np +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Dict, Tuple + + +class PositionGetter: + """Generates and caches 2D spatial positions for patches in a grid. + + This class efficiently manages the generation of spatial coordinates for patches + in a 2D grid, caching results to avoid redundant computations. + + Attributes: + position_cache: Dictionary storing precomputed position tensors for different + grid dimensions. + """ + + def __init__(self): + """Initializes the position generator with an empty cache.""" + self.position_cache: Dict[Tuple[int, int], torch.Tensor] = {} + + def __call__(self, batch_size: int, height: int, width: int, device: torch.device) -> torch.Tensor: + """Generates spatial positions for a batch of patches. + + Args: + batch_size: Number of samples in the batch. + height: Height of the grid in patches. + width: Width of the grid in patches. + device: Target device for the position tensor. + + Returns: + Tensor of shape (batch_size, height*width, 2) containing y,x coordinates + for each position in the grid, repeated for each batch item. + """ + if (height, width) not in self.position_cache: + y_coords = torch.arange(height, device=device) + x_coords = torch.arange(width, device=device) + positions = torch.cartesian_prod(y_coords, x_coords) + self.position_cache[height, width] = positions + + cached_positions = self.position_cache[height, width] + return cached_positions.view(1, height * width, 2).expand(batch_size, -1, -1).clone() + + +class RotaryPositionEmbedding2D(nn.Module): + """2D Rotary Position Embedding implementation. + + This module applies rotary position embeddings to input tokens based on their + 2D spatial positions. It handles the position-dependent rotation of features + separately for vertical and horizontal dimensions. + + Args: + frequency: Base frequency for the position embeddings. Default: 100.0 + scaling_factor: Scaling factor for frequency computation. Default: 1.0 + + Attributes: + base_frequency: Base frequency for computing position embeddings. + scaling_factor: Factor to scale the computed frequencies. + frequency_cache: Cache for storing precomputed frequency components. + """ + + def __init__(self, frequency: float = 100.0, scaling_factor: float = 1.0): + """Initializes the 2D RoPE module.""" + super().__init__() + self.base_frequency = frequency + self.scaling_factor = scaling_factor + self.frequency_cache: Dict[Tuple, Tuple[torch.Tensor, torch.Tensor]] = {} + + def _compute_frequency_components( + self, dim: int, seq_len: int, device: torch.device, dtype: torch.dtype + ) -> Tuple[torch.Tensor, torch.Tensor]: + """Computes frequency components for rotary embeddings. + + Args: + dim: Feature dimension (must be even). + seq_len: Maximum sequence length. + device: Target device for computations. + dtype: Data type for the computed tensors. + + Returns: + Tuple of (cosine, sine) tensors for frequency components. + """ + cache_key = (dim, seq_len, device, dtype) + if cache_key not in self.frequency_cache: + # Compute frequency bands + exponents = torch.arange(0, dim, 2, device=device).float() / dim + inv_freq = 1.0 / (self.base_frequency**exponents) + + # Generate position-dependent frequencies + positions = torch.arange(seq_len, device=device, dtype=inv_freq.dtype) + angles = torch.einsum("i,j->ij", positions, inv_freq) + + # Compute and cache frequency components + angles = angles.to(dtype) + angles = torch.cat((angles, angles), dim=-1) + cos_components = angles.cos().to(dtype) + sin_components = angles.sin().to(dtype) + self.frequency_cache[cache_key] = (cos_components, sin_components) + + return self.frequency_cache[cache_key] + + @staticmethod + def _rotate_features(x: torch.Tensor) -> torch.Tensor: + """Performs feature rotation by splitting and recombining feature dimensions. + + Args: + x: Input tensor to rotate. + + Returns: + Rotated feature tensor. + """ + feature_dim = x.shape[-1] + x1, x2 = x[..., : feature_dim // 2], x[..., feature_dim // 2 :] + return torch.cat((-x2, x1), dim=-1) + + def _apply_1d_rope( + self, tokens: torch.Tensor, positions: torch.Tensor, cos_comp: torch.Tensor, sin_comp: torch.Tensor + ) -> torch.Tensor: + """Applies 1D rotary position embeddings along one dimension. + + Args: + tokens: Input token features. + positions: Position indices. + cos_comp: Cosine components for rotation. + sin_comp: Sine components for rotation. + + Returns: + Tokens with applied rotary position embeddings. + """ + # Embed positions with frequency components + cos = F.embedding(positions, cos_comp)[:, None, :, :] + sin = F.embedding(positions, sin_comp)[:, None, :, :] + + # Apply rotation + return (tokens * cos) + (self._rotate_features(tokens) * sin) + + def forward(self, tokens: torch.Tensor, positions: torch.Tensor) -> torch.Tensor: + """Applies 2D rotary position embeddings to input tokens. + + Args: + tokens: Input tensor of shape (batch_size, n_heads, n_tokens, dim). + The feature dimension (dim) must be divisible by 4. + positions: Position tensor of shape (batch_size, n_tokens, 2) containing + the y and x coordinates for each token. + + Returns: + Tensor of same shape as input with applied 2D rotary position embeddings. + + Raises: + AssertionError: If input dimensions are invalid or positions are malformed. + """ + # Validate inputs + assert tokens.size(-1) % 2 == 0, "Feature dimension must be even" + assert positions.ndim == 3 and positions.shape[-1] == 2, "Positions must have shape (batch_size, n_tokens, 2)" + + # Compute feature dimension for each spatial direction + feature_dim = tokens.size(-1) // 2 + + # Get frequency components + max_position = int(positions.max()) + 1 + cos_comp, sin_comp = self._compute_frequency_components(feature_dim, max_position, tokens.device, tokens.dtype) + + # Split features for vertical and horizontal processing + vertical_features, horizontal_features = tokens.chunk(2, dim=-1) + + # Apply RoPE separately for each dimension + vertical_features = self._apply_1d_rope(vertical_features, positions[..., 0], cos_comp, sin_comp) + horizontal_features = self._apply_1d_rope(horizontal_features, positions[..., 1], cos_comp, sin_comp) + + # Combine processed features + return torch.cat((vertical_features, horizontal_features), dim=-1) diff --git a/vggt/vggt/layers/swiglu_ffn.py b/vggt/vggt/layers/swiglu_ffn.py new file mode 100644 index 0000000..54fe8e9 --- /dev/null +++ b/vggt/vggt/layers/swiglu_ffn.py @@ -0,0 +1,72 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +import os +from typing import Callable, Optional +import warnings + +from torch import Tensor, nn +import torch.nn.functional as F + + +class SwiGLUFFN(nn.Module): + def __init__( + self, + in_features: int, + hidden_features: Optional[int] = None, + out_features: Optional[int] = None, + act_layer: Callable[..., nn.Module] = None, + drop: float = 0.0, + bias: bool = True, + ) -> None: + super().__init__() + out_features = out_features or in_features + hidden_features = hidden_features or in_features + self.w12 = nn.Linear(in_features, 2 * hidden_features, bias=bias) + self.w3 = nn.Linear(hidden_features, out_features, bias=bias) + + def forward(self, x: Tensor) -> Tensor: + x12 = self.w12(x) + x1, x2 = x12.chunk(2, dim=-1) + hidden = F.silu(x1) * x2 + return self.w3(hidden) + + +XFORMERS_ENABLED = os.environ.get("XFORMERS_DISABLED") is None +# try: +# if XFORMERS_ENABLED: +# from xformers.ops import SwiGLU + +# XFORMERS_AVAILABLE = True +# warnings.warn("xFormers is available (SwiGLU)") +# else: +# warnings.warn("xFormers is disabled (SwiGLU)") +# raise ImportError +# except ImportError: +SwiGLU = SwiGLUFFN +XFORMERS_AVAILABLE = False + +# warnings.warn("xFormers is not available (SwiGLU)") + + +class SwiGLUFFNFused(SwiGLU): + def __init__( + self, + in_features: int, + hidden_features: Optional[int] = None, + out_features: Optional[int] = None, + act_layer: Callable[..., nn.Module] = None, + drop: float = 0.0, + bias: bool = True, + ) -> None: + out_features = out_features or in_features + hidden_features = hidden_features or in_features + hidden_features = (int(hidden_features * 2 / 3) + 7) // 8 * 8 + super().__init__( + in_features=in_features, + hidden_features=hidden_features, + out_features=out_features, + bias=bias, + ) diff --git a/vggt/vggt/layers/vision_transformer.py b/vggt/vggt/layers/vision_transformer.py new file mode 100644 index 0000000..120cbe6 --- /dev/null +++ b/vggt/vggt/layers/vision_transformer.py @@ -0,0 +1,407 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# +# This source code is licensed under the Apache License, Version 2.0 +# found in the LICENSE file in the root directory of this source tree. + +# References: +# https://github.com/facebookresearch/dino/blob/main/vision_transformer.py +# https://github.com/rwightman/pytorch-image-models/tree/master/timm/models/vision_transformer.py + +from functools import partial +import math +import logging +from typing import Sequence, Tuple, Union, Callable + +import torch +import torch.nn as nn +from torch.utils.checkpoint import checkpoint +from torch.nn.init import trunc_normal_ +from . import Mlp, PatchEmbed, SwiGLUFFNFused, MemEffAttention, NestedTensorBlock as Block + +logger = logging.getLogger("dinov2") + + +def named_apply(fn: Callable, module: nn.Module, name="", depth_first=True, include_root=False) -> nn.Module: + if not depth_first and include_root: + fn(module=module, name=name) + for child_name, child_module in module.named_children(): + child_name = ".".join((name, child_name)) if name else child_name + named_apply(fn=fn, module=child_module, name=child_name, depth_first=depth_first, include_root=True) + if depth_first and include_root: + fn(module=module, name=name) + return module + + +class BlockChunk(nn.ModuleList): + def forward(self, x): + for b in self: + x = b(x) + return x + + +class DinoVisionTransformer(nn.Module): + def __init__( + self, + img_size=224, + patch_size=16, + in_chans=3, + embed_dim=768, + depth=12, + num_heads=12, + mlp_ratio=4.0, + qkv_bias=True, + ffn_bias=True, + proj_bias=True, + drop_path_rate=0.0, + drop_path_uniform=False, + init_values=None, # for layerscale: None or 0 => no layerscale + embed_layer=PatchEmbed, + act_layer=nn.GELU, + block_fn=Block, + ffn_layer="mlp", + block_chunks=1, + num_register_tokens=0, + interpolate_antialias=False, + interpolate_offset=0.1, + qk_norm=False, + ): + """ + Args: + img_size (int, tuple): input image size + patch_size (int, tuple): patch size + in_chans (int): number of input channels + embed_dim (int): embedding dimension + depth (int): depth of transformer + num_heads (int): number of attention heads + mlp_ratio (int): ratio of mlp hidden dim to embedding dim + qkv_bias (bool): enable bias for qkv if True + proj_bias (bool): enable bias for proj in attn if True + ffn_bias (bool): enable bias for ffn if True + drop_path_rate (float): stochastic depth rate + drop_path_uniform (bool): apply uniform drop rate across blocks + weight_init (str): weight init scheme + init_values (float): layer-scale init values + embed_layer (nn.Module): patch embedding layer + act_layer (nn.Module): MLP activation layer + block_fn (nn.Module): transformer block class + ffn_layer (str): "mlp", "swiglu", "swiglufused" or "identity" + block_chunks: (int) split block sequence into block_chunks units for FSDP wrap + num_register_tokens: (int) number of extra cls tokens (so-called "registers") + interpolate_antialias: (str) flag to apply anti-aliasing when interpolating positional embeddings + interpolate_offset: (float) work-around offset to apply when interpolating positional embeddings + """ + super().__init__() + norm_layer = partial(nn.LayerNorm, eps=1e-6) + + # tricky but makes it work + self.use_checkpoint = False + # + + self.num_features = self.embed_dim = embed_dim # num_features for consistency with other models + self.num_tokens = 1 + self.n_blocks = depth + self.num_heads = num_heads + self.patch_size = patch_size + self.num_register_tokens = num_register_tokens + self.interpolate_antialias = interpolate_antialias + self.interpolate_offset = interpolate_offset + + self.patch_embed = embed_layer(img_size=img_size, patch_size=patch_size, in_chans=in_chans, embed_dim=embed_dim) + num_patches = self.patch_embed.num_patches + + self.cls_token = nn.Parameter(torch.zeros(1, 1, embed_dim)) + self.pos_embed = nn.Parameter(torch.zeros(1, num_patches + self.num_tokens, embed_dim)) + assert num_register_tokens >= 0 + self.register_tokens = ( + nn.Parameter(torch.zeros(1, num_register_tokens, embed_dim)) if num_register_tokens else None + ) + + if drop_path_uniform is True: + dpr = [drop_path_rate] * depth + else: + dpr = [x.item() for x in torch.linspace(0, drop_path_rate, depth)] # stochastic depth decay rule + + if ffn_layer == "mlp": + logger.info("using MLP layer as FFN") + ffn_layer = Mlp + elif ffn_layer == "swiglufused" or ffn_layer == "swiglu": + logger.info("using SwiGLU layer as FFN") + ffn_layer = SwiGLUFFNFused + elif ffn_layer == "identity": + logger.info("using Identity layer as FFN") + + def f(*args, **kwargs): + return nn.Identity() + + ffn_layer = f + else: + raise NotImplementedError + + blocks_list = [ + block_fn( + dim=embed_dim, + num_heads=num_heads, + mlp_ratio=mlp_ratio, + qkv_bias=qkv_bias, + proj_bias=proj_bias, + ffn_bias=ffn_bias, + drop_path=dpr[i], + norm_layer=norm_layer, + act_layer=act_layer, + ffn_layer=ffn_layer, + init_values=init_values, + qk_norm=qk_norm, + ) + for i in range(depth) + ] + if block_chunks > 0: + self.chunked_blocks = True + chunked_blocks = [] + chunksize = depth // block_chunks + for i in range(0, depth, chunksize): + # this is to keep the block index consistent if we chunk the block list + chunked_blocks.append([nn.Identity()] * i + blocks_list[i : i + chunksize]) + self.blocks = nn.ModuleList([BlockChunk(p) for p in chunked_blocks]) + else: + self.chunked_blocks = False + self.blocks = nn.ModuleList(blocks_list) + + self.norm = norm_layer(embed_dim) + self.head = nn.Identity() + + self.mask_token = nn.Parameter(torch.zeros(1, embed_dim)) + + self.init_weights() + + def init_weights(self): + trunc_normal_(self.pos_embed, std=0.02) + nn.init.normal_(self.cls_token, std=1e-6) + if self.register_tokens is not None: + nn.init.normal_(self.register_tokens, std=1e-6) + named_apply(init_weights_vit_timm, self) + + def interpolate_pos_encoding(self, x, w, h): + previous_dtype = x.dtype + npatch = x.shape[1] - 1 + N = self.pos_embed.shape[1] - 1 + if npatch == N and w == h: + return self.pos_embed + pos_embed = self.pos_embed.float() + class_pos_embed = pos_embed[:, 0] + patch_pos_embed = pos_embed[:, 1:] + dim = x.shape[-1] + w0 = w // self.patch_size + h0 = h // self.patch_size + M = int(math.sqrt(N)) # Recover the number of patches in each dimension + assert N == M * M + kwargs = {} + if self.interpolate_offset: + # Historical kludge: add a small number to avoid floating point error in the interpolation, see https://github.com/facebookresearch/dino/issues/8 + # Note: still needed for backward-compatibility, the underlying operators are using both output size and scale factors + sx = float(w0 + self.interpolate_offset) / M + sy = float(h0 + self.interpolate_offset) / M + kwargs["scale_factor"] = (sx, sy) + else: + # Simply specify an output size instead of a scale factor + kwargs["size"] = (w0, h0) + patch_pos_embed = nn.functional.interpolate( + patch_pos_embed.reshape(1, M, M, dim).permute(0, 3, 1, 2), + mode="bicubic", + antialias=self.interpolate_antialias, + **kwargs, + ) + assert (w0, h0) == patch_pos_embed.shape[-2:] + patch_pos_embed = patch_pos_embed.permute(0, 2, 3, 1).view(1, -1, dim) + return torch.cat((class_pos_embed.unsqueeze(0), patch_pos_embed), dim=1).to(previous_dtype) + + def prepare_tokens_with_masks(self, x, masks=None): + B, nc, w, h = x.shape + x = self.patch_embed(x) + if masks is not None: + x = torch.where(masks.unsqueeze(-1), self.mask_token.to(x.dtype).unsqueeze(0), x) + + x = torch.cat((self.cls_token.expand(x.shape[0], -1, -1), x), dim=1) + x = x + self.interpolate_pos_encoding(x, w, h) + + if self.register_tokens is not None: + x = torch.cat( + ( + x[:, :1], + self.register_tokens.expand(x.shape[0], -1, -1), + x[:, 1:], + ), + dim=1, + ) + + return x + + def forward_features_list(self, x_list, masks_list): + x = [self.prepare_tokens_with_masks(x, masks) for x, masks in zip(x_list, masks_list)] + + for blk in self.blocks: + if self.use_checkpoint: + x = checkpoint(blk, x, use_reentrant=self.use_reentrant) + else: + x = blk(x) + + all_x = x + output = [] + for x, masks in zip(all_x, masks_list): + x_norm = self.norm(x) + output.append( + { + "x_norm_clstoken": x_norm[:, 0], + "x_norm_regtokens": x_norm[:, 1 : self.num_register_tokens + 1], + "x_norm_patchtokens": x_norm[:, self.num_register_tokens + 1 :], + "x_prenorm": x, + "masks": masks, + } + ) + return output + + def forward_features(self, x, masks=None): + if isinstance(x, list): + return self.forward_features_list(x, masks) + + x = self.prepare_tokens_with_masks(x, masks) + + for blk in self.blocks: + if self.use_checkpoint: + x = checkpoint(blk, x, use_reentrant=self.use_reentrant) + else: + x = blk(x) + + x_norm = self.norm(x) + return { + "x_norm_clstoken": x_norm[:, 0], + "x_norm_regtokens": x_norm[:, 1 : self.num_register_tokens + 1], + "x_norm_patchtokens": x_norm[:, self.num_register_tokens + 1 :], + "x_prenorm": x, + "masks": masks, + } + + def _get_intermediate_layers_not_chunked(self, x, n=1): + x = self.prepare_tokens_with_masks(x) + # If n is an int, take the n last blocks. If it's a list, take them + output, total_block_len = [], len(self.blocks) + blocks_to_take = range(total_block_len - n, total_block_len) if isinstance(n, int) else n + for i, blk in enumerate(self.blocks): + x = blk(x) + if i in blocks_to_take: + output.append(x) + assert len(output) == len(blocks_to_take), f"only {len(output)} / {len(blocks_to_take)} blocks found" + return output + + def _get_intermediate_layers_chunked(self, x, n=1): + x = self.prepare_tokens_with_masks(x) + output, i, total_block_len = [], 0, len(self.blocks[-1]) + # If n is an int, take the n last blocks. If it's a list, take them + blocks_to_take = range(total_block_len - n, total_block_len) if isinstance(n, int) else n + for block_chunk in self.blocks: + for blk in block_chunk[i:]: # Passing the nn.Identity() + x = blk(x) + if i in blocks_to_take: + output.append(x) + i += 1 + assert len(output) == len(blocks_to_take), f"only {len(output)} / {len(blocks_to_take)} blocks found" + return output + + def get_intermediate_layers( + self, + x: torch.Tensor, + n: Union[int, Sequence] = 1, # Layers or n last layers to take + reshape: bool = False, + return_class_token: bool = False, + norm=True, + ) -> Tuple[Union[torch.Tensor, Tuple[torch.Tensor]]]: + if self.chunked_blocks: + outputs = self._get_intermediate_layers_chunked(x, n) + else: + outputs = self._get_intermediate_layers_not_chunked(x, n) + if norm: + outputs = [self.norm(out) for out in outputs] + class_tokens = [out[:, 0] for out in outputs] + outputs = [out[:, 1 + self.num_register_tokens :] for out in outputs] + if reshape: + B, _, w, h = x.shape + outputs = [ + out.reshape(B, w // self.patch_size, h // self.patch_size, -1).permute(0, 3, 1, 2).contiguous() + for out in outputs + ] + if return_class_token: + return tuple(zip(outputs, class_tokens)) + return tuple(outputs) + + def forward(self, *args, is_training=True, **kwargs): + ret = self.forward_features(*args, **kwargs) + if is_training: + return ret + else: + return self.head(ret["x_norm_clstoken"]) + + +def init_weights_vit_timm(module: nn.Module, name: str = ""): + """ViT weight initialization, original timm impl (for reproducibility)""" + if isinstance(module, nn.Linear): + trunc_normal_(module.weight, std=0.02) + if module.bias is not None: + nn.init.zeros_(module.bias) + + +def vit_small(patch_size=16, num_register_tokens=0, **kwargs): + model = DinoVisionTransformer( + patch_size=patch_size, + embed_dim=384, + depth=12, + num_heads=6, + mlp_ratio=4, + block_fn=partial(Block, attn_class=MemEffAttention), + num_register_tokens=num_register_tokens, + **kwargs, + ) + return model + + +def vit_base(patch_size=16, num_register_tokens=0, **kwargs): + model = DinoVisionTransformer( + patch_size=patch_size, + embed_dim=768, + depth=12, + num_heads=12, + mlp_ratio=4, + block_fn=partial(Block, attn_class=MemEffAttention), + num_register_tokens=num_register_tokens, + **kwargs, + ) + return model + + +def vit_large(patch_size=16, num_register_tokens=0, **kwargs): + model = DinoVisionTransformer( + patch_size=patch_size, + embed_dim=1024, + depth=24, + num_heads=16, + mlp_ratio=4, + block_fn=partial(Block, attn_class=MemEffAttention), + num_register_tokens=num_register_tokens, + **kwargs, + ) + return model + + +def vit_giant2(patch_size=16, num_register_tokens=0, **kwargs): + """ + Close to ViT-giant, with embed-dim 1536 and 24 heads => embed-dim per head 64 + """ + model = DinoVisionTransformer( + patch_size=patch_size, + embed_dim=1536, + depth=40, + num_heads=24, + mlp_ratio=4, + block_fn=partial(Block, attn_class=MemEffAttention), + num_register_tokens=num_register_tokens, + **kwargs, + ) + return model diff --git a/vggt/vggt/models/aggregator.py b/vggt/vggt/models/aggregator.py new file mode 100644 index 0000000..50db933 --- /dev/null +++ b/vggt/vggt/models/aggregator.py @@ -0,0 +1,331 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import logging +import torch +import torch.nn as nn +import torch.nn.functional as F +from typing import Optional, Tuple, Union, List, Dict, Any + +from ..layers import PatchEmbed +from ..layers.block import Block +from ..layers.rope import RotaryPositionEmbedding2D, PositionGetter +from ..layers.vision_transformer import vit_small, vit_base, vit_large, vit_giant2 + +logger = logging.getLogger(__name__) + +_RESNET_MEAN = [0.485, 0.456, 0.406] +_RESNET_STD = [0.229, 0.224, 0.225] + + +class Aggregator(nn.Module): + """ + The Aggregator applies alternating-attention over input frames, + as described in VGGT: Visual Geometry Grounded Transformer. + + + Args: + img_size (int): Image size in pixels. + patch_size (int): Size of each patch for PatchEmbed. + embed_dim (int): Dimension of the token embeddings. + depth (int): Number of blocks. + num_heads (int): Number of attention heads. + mlp_ratio (float): Ratio of MLP hidden dim to embedding dim. + num_register_tokens (int): Number of register tokens. + block_fn (nn.Module): The block type used for attention (Block by default). + qkv_bias (bool): Whether to include bias in QKV projections. + proj_bias (bool): Whether to include bias in the output projection. + ffn_bias (bool): Whether to include bias in MLP layers. + patch_embed (str): Type of patch embed. e.g., "conv" or "dinov2_vitl14_reg". + aa_order (list[str]): The order of alternating attention, e.g. ["frame", "global"]. + aa_block_size (int): How many blocks to group under each attention type before switching. If not necessary, set to 1. + qk_norm (bool): Whether to apply QK normalization. + rope_freq (int): Base frequency for rotary embedding. -1 to disable. + init_values (float): Init scale for layer scale. + """ + + def __init__( + self, + img_size=518, + patch_size=14, + embed_dim=1024, + depth=24, + num_heads=16, + mlp_ratio=4.0, + num_register_tokens=4, + block_fn=Block, + qkv_bias=True, + proj_bias=True, + ffn_bias=True, + patch_embed="dinov2_vitl14_reg", + aa_order=["frame", "global"], + aa_block_size=1, + qk_norm=True, + rope_freq=100, + init_values=0.01, + ): + super().__init__() + + self.__build_patch_embed__(patch_embed, img_size, patch_size, num_register_tokens, embed_dim=embed_dim) + + # Initialize rotary position embedding if frequency > 0 + self.rope = RotaryPositionEmbedding2D(frequency=rope_freq) if rope_freq > 0 else None + self.position_getter = PositionGetter() if self.rope is not None else None + + self.frame_blocks = nn.ModuleList( + [ + block_fn( + dim=embed_dim, + num_heads=num_heads, + mlp_ratio=mlp_ratio, + qkv_bias=qkv_bias, + proj_bias=proj_bias, + ffn_bias=ffn_bias, + init_values=init_values, + qk_norm=qk_norm, + rope=self.rope, + ) + for _ in range(depth) + ] + ) + + self.global_blocks = nn.ModuleList( + [ + block_fn( + dim=embed_dim, + num_heads=num_heads, + mlp_ratio=mlp_ratio, + qkv_bias=qkv_bias, + proj_bias=proj_bias, + ffn_bias=ffn_bias, + init_values=init_values, + qk_norm=qk_norm, + rope=self.rope, + ) + for _ in range(depth) + ] + ) + + self.depth = depth + self.aa_order = aa_order + self.patch_size = patch_size + self.aa_block_size = aa_block_size + + # Validate that depth is divisible by aa_block_size + if self.depth % self.aa_block_size != 0: + raise ValueError(f"depth ({depth}) must be divisible by aa_block_size ({aa_block_size})") + + self.aa_block_num = self.depth // self.aa_block_size + + # Note: We have two camera tokens, one for the first frame and one for the rest + # The same applies for register tokens + self.camera_token = nn.Parameter(torch.randn(1, 2, 1, embed_dim)) + self.register_token = nn.Parameter(torch.randn(1, 2, num_register_tokens, embed_dim)) + + # The patch tokens start after the camera and register tokens + self.patch_start_idx = 1 + num_register_tokens + + # Initialize parameters with small values + nn.init.normal_(self.camera_token, std=1e-6) + nn.init.normal_(self.register_token, std=1e-6) + + # Register normalization constants as buffers + for name, value in ( + ("_resnet_mean", _RESNET_MEAN), + ("_resnet_std", _RESNET_STD), + ): + self.register_buffer( + name, + torch.FloatTensor(value).view(1, 1, 3, 1, 1), + persistent=False, + ) + + def __build_patch_embed__( + self, + patch_embed, + img_size, + patch_size, + num_register_tokens, + interpolate_antialias=True, + interpolate_offset=0.0, + block_chunks=0, + init_values=1.0, + embed_dim=1024, + ): + """ + Build the patch embed layer. If 'conv', we use a + simple PatchEmbed conv layer. Otherwise, we use a vision transformer. + """ + + if "conv" in patch_embed: + self.patch_embed = PatchEmbed(img_size=img_size, patch_size=patch_size, in_chans=3, embed_dim=embed_dim) + else: + vit_models = { + "dinov2_vitl14_reg": vit_large, + "dinov2_vitb14_reg": vit_base, + "dinov2_vits14_reg": vit_small, + "dinov2_vitg2_reg": vit_giant2, + } + + self.patch_embed = vit_models[patch_embed]( + img_size=img_size, + patch_size=patch_size, + num_register_tokens=num_register_tokens, + interpolate_antialias=interpolate_antialias, + interpolate_offset=interpolate_offset, + block_chunks=block_chunks, + init_values=init_values, + ) + + # Disable gradient updates for mask token + if hasattr(self.patch_embed, "mask_token"): + self.patch_embed.mask_token.requires_grad_(False) + + def forward( + self, + images: torch.Tensor, + ) -> Tuple[List[torch.Tensor], int]: + """ + Args: + images (torch.Tensor): Input images with shape [B, S, 3, H, W], in range [0, 1]. + B: batch size, S: sequence length, 3: RGB channels, H: height, W: width + + Returns: + (list[torch.Tensor], int): + The list of outputs from the attention blocks, + and the patch_start_idx indicating where patch tokens begin. + """ + B, S, C_in, H, W = images.shape + + if C_in != 3: + raise ValueError(f"Expected 3 input channels, got {C_in}") + + # Normalize images and reshape for patch embed + images = (images - self._resnet_mean) / self._resnet_std + + # Reshape to [B*S, C, H, W] for patch embedding + images = images.view(B * S, C_in, H, W) + patch_tokens = self.patch_embed(images) + + if isinstance(patch_tokens, dict): + patch_tokens = patch_tokens["x_norm_patchtokens"] + + _, P, C = patch_tokens.shape + + # Expand camera and register tokens to match batch size and sequence length + camera_token = slice_expand_and_flatten(self.camera_token, B, S) + register_token = slice_expand_and_flatten(self.register_token, B, S) + + # Concatenate special tokens with patch tokens + tokens = torch.cat([camera_token, register_token, patch_tokens], dim=1) + + pos = None + if self.rope is not None: + pos = self.position_getter(B * S, H // self.patch_size, W // self.patch_size, device=images.device) + + if self.patch_start_idx > 0: + # do not use position embedding for special tokens (camera and register tokens) + # so set pos to 0 for the special tokens + pos = pos + 1 + pos_special = torch.zeros(B * S, self.patch_start_idx, 2).to(images.device).to(pos.dtype) + pos = torch.cat([pos_special, pos], dim=1) + + # update P because we added special tokens + _, P, C = tokens.shape + + frame_idx = 0 + global_idx = 0 + output_list = [] + + for _ in range(self.aa_block_num): + for attn_type in self.aa_order: + if attn_type == "frame": + tokens, frame_idx, frame_intermediates = self._process_frame_attention( + tokens, B, S, P, C, frame_idx, pos=pos + ) + elif attn_type == "global": + tokens, global_idx, global_intermediates = self._process_global_attention( + tokens, B, S, P, C, global_idx, pos=pos + ) + else: + raise ValueError(f"Unknown attention type: {attn_type}") + + for i in range(len(frame_intermediates)): + # concat frame and global intermediates, [B x S x P x 2C] + concat_inter = torch.cat([frame_intermediates[i], global_intermediates[i]], dim=-1) + output_list.append(concat_inter) + + del concat_inter + del frame_intermediates + del global_intermediates + return output_list, self.patch_start_idx + + def _process_frame_attention(self, tokens, B, S, P, C, frame_idx, pos=None): + """ + Process frame attention blocks. We keep tokens in shape (B*S, P, C). + """ + # If needed, reshape tokens or positions: + if tokens.shape != (B * S, P, C): + tokens = tokens.view(B, S, P, C).view(B * S, P, C) + + if pos is not None and pos.shape != (B * S, P, 2): + pos = pos.view(B, S, P, 2).view(B * S, P, 2) + + intermediates = [] + + # by default, self.aa_block_size=1, which processes one block at a time + for _ in range(self.aa_block_size): + tokens = self.frame_blocks[frame_idx](tokens, pos=pos) + frame_idx += 1 + intermediates.append(tokens.view(B, S, P, C)) + + return tokens, frame_idx, intermediates + + def _process_global_attention(self, tokens, B, S, P, C, global_idx, pos=None): + """ + Process global attention blocks. We keep tokens in shape (B, S*P, C). + """ + if tokens.shape != (B, S * P, C): + tokens = tokens.view(B, S, P, C).view(B, S * P, C) + + if pos is not None and pos.shape != (B, S * P, 2): + pos = pos.view(B, S, P, 2).view(B, S * P, 2) + + intermediates = [] + + # by default, self.aa_block_size=1, which processes one block at a time + for _ in range(self.aa_block_size): + tokens = self.global_blocks[global_idx](tokens, pos=pos) + global_idx += 1 + intermediates.append(tokens.view(B, S, P, C)) + + return tokens, global_idx, intermediates + + +def slice_expand_and_flatten(token_tensor, B, S): + """ + Processes specialized tokens with shape (1, 2, X, C) for multi-frame processing: + 1) Uses the first position (index=0) for the first frame only + 2) Uses the second position (index=1) for all remaining frames (S-1 frames) + 3) Expands both to match batch size B + 4) Concatenates to form (B, S, X, C) where each sequence has 1 first-position token + followed by (S-1) second-position tokens + 5) Flattens to (B*S, X, C) for processing + + Returns: + torch.Tensor: Processed tokens with shape (B*S, X, C) + """ + + # Slice out the "query" tokens => shape (1, 1, ...) + query = token_tensor[:, 0:1, ...].expand(B, 1, *token_tensor.shape[2:]) + # Slice out the "other" tokens => shape (1, S-1, ...) + others = token_tensor[:, 1:, ...].expand(B, S - 1, *token_tensor.shape[2:]) + # Concatenate => shape (B, S, ...) + combined = torch.cat([query, others], dim=1) + + # Finally flatten => shape (B*S, ...) + combined = combined.view(B * S, *combined.shape[2:]) + return combined diff --git a/vggt/vggt/models/vggt.py b/vggt/vggt/models/vggt.py new file mode 100644 index 0000000..25790f2 --- /dev/null +++ b/vggt/vggt/models/vggt.py @@ -0,0 +1,95 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch +import torch.nn as nn +from huggingface_hub import PyTorchModelHubMixin # used for model hub +from .aggregator import Aggregator +from ..heads.camera_head import CameraHead +from ..heads.dpt_head import DPTHead +from ..heads.track_head import TrackHead + + +class VGGT(nn.Module, PyTorchModelHubMixin): + def __init__(self, img_size=518, patch_size=14, embed_dim=1024): + super().__init__() + + self.aggregator = Aggregator(img_size=img_size, patch_size=patch_size, embed_dim=embed_dim) + self.camera_head = CameraHead(dim_in=2 * embed_dim) + self.point_head = DPTHead(dim_in=2 * embed_dim, output_dim=4, activation="inv_log", conf_activation="expp1") + self.depth_head = DPTHead(dim_in=2 * embed_dim, output_dim=2, activation="exp", conf_activation="expp1") + self.track_head = TrackHead(dim_in=2 * embed_dim, patch_size=patch_size) + + def forward( + self, + images: torch.Tensor, + query_points: torch.Tensor = None, + ): + """ + Forward pass of the VGGT model. + + Args: + images (torch.Tensor): Input images with shape [S, 3, H, W] or [B, S, 3, H, W], in range [0, 1]. + B: batch size, S: sequence length, 3: RGB channels, H: height, W: width + query_points (torch.Tensor, optional): Query points for tracking, in pixel coordinates. + Shape: [N, 2] or [B, N, 2], where N is the number of query points. + Default: None + + Returns: + dict: A dictionary containing the following predictions: + - pose_enc (torch.Tensor): Camera pose encoding with shape [B, S, 9] (from the last iteration) + - depth (torch.Tensor): Predicted depth maps with shape [B, S, H, W, 1] + - depth_conf (torch.Tensor): Confidence scores for depth predictions with shape [B, S, H, W] + - world_points (torch.Tensor): 3D world coordinates for each pixel with shape [B, S, H, W, 3] + - world_points_conf (torch.Tensor): Confidence scores for world points with shape [B, S, H, W] + - images (torch.Tensor): Original input images, preserved for visualization + + If query_points is provided, also includes: + - track (torch.Tensor): Point tracks with shape [B, S, N, 2] (from the last iteration), in pixel coordinates + - vis (torch.Tensor): Visibility scores for tracked points with shape [B, S, N] + - conf (torch.Tensor): Confidence scores for tracked points with shape [B, S, N] + """ + + # If without batch dimension, add it + if len(images.shape) == 4: + images = images.unsqueeze(0) + if query_points is not None and len(query_points.shape) == 2: + query_points = query_points.unsqueeze(0) + + aggregated_tokens_list, patch_start_idx = self.aggregator(images) + + predictions = {} + + with torch.cuda.amp.autocast(enabled=False): + if self.camera_head is not None: + pose_enc_list = self.camera_head(aggregated_tokens_list) + predictions["pose_enc"] = pose_enc_list[-1] # pose encoding of the last iteration + + if self.depth_head is not None: + depth, depth_conf = self.depth_head( + aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx + ) + predictions["depth"] = depth + predictions["depth_conf"] = depth_conf + + if self.point_head is not None: + pts3d, pts3d_conf = self.point_head( + aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx + ) + predictions["world_points"] = pts3d + predictions["world_points_conf"] = pts3d_conf + + if self.track_head is not None and query_points is not None: + track_list, vis, conf = self.track_head( + aggregated_tokens_list, images=images, patch_start_idx=patch_start_idx, query_points=query_points + ) + predictions["track"] = track_list[-1] # track of the last iteration + predictions["vis"] = vis + predictions["conf"] = conf + + predictions["images"] = images + + return predictions diff --git a/vggt/vggt/utils/geometry.py b/vggt/vggt/utils/geometry.py new file mode 100644 index 0000000..e9c3fde --- /dev/null +++ b/vggt/vggt/utils/geometry.py @@ -0,0 +1,236 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import os +import torch +import numpy as np + + +def unproject_depth_map_to_point_map( + depth_map: np.ndarray, extrinsics_cam: np.ndarray, intrinsics_cam: np.ndarray +) -> np.ndarray: + """ + Unproject a batch of depth maps to 3D world coordinates. + + Args: + depth_map (np.ndarray): Batch of depth maps of shape (S, H, W, 1) or (S, H, W) + extrinsics_cam (np.ndarray): Batch of camera extrinsic matrices of shape (S, 3, 4) + intrinsics_cam (np.ndarray): Batch of camera intrinsic matrices of shape (S, 3, 3) + + Returns: + np.ndarray: Batch of 3D world coordinates of shape (S, H, W, 3) + """ + if isinstance(depth_map, torch.Tensor): + depth_map = depth_map.cpu().numpy() + if isinstance(extrinsics_cam, torch.Tensor): + extrinsics_cam = extrinsics_cam.cpu().numpy() + if isinstance(intrinsics_cam, torch.Tensor): + intrinsics_cam = intrinsics_cam.cpu().numpy() + + world_points_list = [] + for frame_idx in range(depth_map.shape[0]): + cur_world_points, _, _ = depth_to_world_coords_points( + depth_map[frame_idx].squeeze(-1), extrinsics_cam[frame_idx], intrinsics_cam[frame_idx] + ) + world_points_list.append(cur_world_points) + world_points_array = np.stack(world_points_list, axis=0) + + return world_points_array + + +def depth_to_world_coords_points( + depth_map: np.ndarray, + extrinsic: np.ndarray, + intrinsic: np.ndarray, + eps=1e-8, +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """ + Convert a depth map to world coordinates. + + Args: + depth_map (np.ndarray): Depth map of shape (H, W). + intrinsic (np.ndarray): Camera intrinsic matrix of shape (3, 3). + extrinsic (np.ndarray): Camera extrinsic matrix of shape (3, 4). OpenCV camera coordinate convention, cam from world. + + Returns: + tuple[np.ndarray, np.ndarray]: World coordinates (H, W, 3) and valid depth mask (H, W). + """ + if depth_map is None: + return None, None, None + + # Valid depth mask + point_mask = depth_map > eps + + # Convert depth map to camera coordinates + cam_coords_points = depth_to_cam_coords_points(depth_map, intrinsic) + + # Multiply with the inverse of extrinsic matrix to transform to world coordinates + # extrinsic_inv is 4x4 (note closed_form_inverse_OpenCV is batched, the output is (N, 4, 4)) + cam_to_world_extrinsic = closed_form_inverse_se3(extrinsic[None])[0] + + R_cam_to_world = cam_to_world_extrinsic[:3, :3] + t_cam_to_world = cam_to_world_extrinsic[:3, 3] + + # Apply the rotation and translation to the camera coordinates + world_coords_points = np.dot(cam_coords_points, R_cam_to_world.T) + t_cam_to_world # HxWx3, 3x3 -> HxWx3 + # world_coords_points = np.einsum("ij,hwj->hwi", R_cam_to_world, cam_coords_points) + t_cam_to_world + + return world_coords_points, cam_coords_points, point_mask + + +def depth_to_cam_coords_points(depth_map: np.ndarray, intrinsic: np.ndarray) -> tuple[np.ndarray, np.ndarray]: + """ + Convert a depth map to camera coordinates. + + Args: + depth_map (np.ndarray): Depth map of shape (H, W). + intrinsic (np.ndarray): Camera intrinsic matrix of shape (3, 3). + + Returns: + tuple[np.ndarray, np.ndarray]: Camera coordinates (H, W, 3) + """ + H, W = depth_map.shape + assert intrinsic.shape == (3, 3), "Intrinsic matrix must be 3x3" + assert intrinsic[0, 1] == 0 and intrinsic[1, 0] == 0, "Intrinsic matrix must have zero skew" + + # Intrinsic parameters + fu, fv = intrinsic[0, 0], intrinsic[1, 1] + cu, cv = intrinsic[0, 2], intrinsic[1, 2] + + # Generate grid of pixel coordinates + u, v = np.meshgrid(np.arange(W), np.arange(H)) + + # Unproject to camera coordinates + x_cam = (u - cu) * depth_map / fu + y_cam = (v - cv) * depth_map / fv + z_cam = depth_map + + # Stack to form camera coordinates + cam_coords = np.stack((x_cam, y_cam, z_cam), axis=-1).astype(np.float32) + + return cam_coords + + +def closed_form_inverse_se3(se3, R=None, T=None): + """ + Compute the inverse of each 4x4 (or 3x4) SE3 matrix in a batch. + + If `R` and `T` are provided, they must correspond to the rotation and translation + components of `se3`. Otherwise, they will be extracted from `se3`. + + Args: + se3: Nx4x4 or Nx3x4 array or tensor of SE3 matrices. + R (optional): Nx3x3 array or tensor of rotation matrices. + T (optional): Nx3x1 array or tensor of translation vectors. + + Returns: + Inverted SE3 matrices with the same type and device as `se3`. + + Shapes: + se3: (N, 4, 4) + R: (N, 3, 3) + T: (N, 3, 1) + """ + # Check if se3 is a numpy array or a torch tensor + is_numpy = isinstance(se3, np.ndarray) + + # Validate shapes + if se3.shape[-2:] != (4, 4) and se3.shape[-2:] != (3, 4): + raise ValueError(f"se3 must be of shape (N,4,4), got {se3.shape}.") + + # Extract R and T if not provided + if R is None: + R = se3[:, :3, :3] # (N,3,3) + if T is None: + T = se3[:, :3, 3:] # (N,3,1) + + # Transpose R + if is_numpy: + # Compute the transpose of the rotation for NumPy + R_transposed = np.transpose(R, (0, 2, 1)) + # -R^T t for NumPy + top_right = -np.matmul(R_transposed, T) + inverted_matrix = np.tile(np.eye(4), (len(R), 1, 1)) + else: + R_transposed = R.permute(0, 2, 1) # (N,3,3) + top_right = -torch.bmm(R_transposed, T) # (N,3,1) + inverted_matrix = torch.eye(4, 4)[None].repeat(len(R), 1, 1) + inverted_matrix = inverted_matrix.to(R.dtype).to(R.device) + + inverted_matrix[:, :3, :3] = R_transposed + inverted_matrix[:, :3, 3:] = top_right + + return inverted_matrix + +def depth_to_cam_coords_points_tensor(depth_map: torch.Tensor, intrinsic: torch.Tensor) -> torch.Tensor: + """ + Convert a depth map to camera coordinates. + + Args: + depth_map (torch.Tensor): Depth map of shape (B, H, W). + intrinsic (torch.Tensor): Camera intrinsic matrix of shape (B, 3, 3). + + Returns: + torch.Tensor: Camera coordinates (B, H, W, 3) + """ + B, H, W = depth_map.shape + + # Intrinsic parameters + fu, fv = intrinsic[:, 0, 0], intrinsic[:, 1, 1] + cu, cv = intrinsic[:, 0, 2], intrinsic[:, 1, 2] + + # Generate grid of pixel coordinates + v, u = torch.meshgrid(torch.arange(W).to(depth_map.device), torch.arange(H).to(depth_map.device)) + + # Unproject to camera coordinates + x_cam = (u[None] - cu[:, None, None]) * depth_map / fu[:, None, None] + y_cam = (v[None] - cv[:, None, None]) * depth_map / fv[:, None, None] + z_cam = depth_map + + # Stack to form camera coordinates + cam_coords = torch.stack((x_cam, y_cam, z_cam), dim=-1).float() + + return cam_coords + +def depth_to_world_coords_points_tensor( + depth_map: torch.Tensor, + extrinsic: torch.Tensor, + intrinsic: torch.Tensor, + eps=1e-8, +) -> torch.Tensor: + """ + Convert a depth map to world coordinates. + + Args: + depth_map (torch.Tensor): Depth map of shape (B, H, W, 1). + intrinsic (torch.Tensor): Camera intrinsic matrix of shape (B, 3, 3). + extrinsic (torch.Tensor): Camera extrinsic matrix of shape (B, 3, 4). OpenCV camera coordinate convention, cam from world. + + Returns: + torch.Tensor: World coordinates (B, H, W, 3). + """ + if depth_map is None: + return None + + # Valid depth mask + point_mask = depth_map > eps + + # Convert depth map to camera coordinates + cam_coords_points = depth_to_cam_coords_points_tensor(depth_map, intrinsic) + + # Multiply with the inverse of extrinsic matrix to transform to world coordinates + # extrinsic_inv is 4x4 (note closed_form_inverse_OpenCV is batched, the output is (N, 4, 4)) + cam_to_world_extrinsic = closed_form_inverse_se3(extrinsic) + + R_cam_to_world = cam_to_world_extrinsic[:, :3, :3] + t_cam_to_world = cam_to_world_extrinsic[:, :3, 3] + + B, H, W, _ = cam_coords_points.shape + # Apply the rotation and translation to the camera coordinates + world_coords_points = torch.matmul(cam_coords_points.reshape(B, -1, 3), R_cam_to_world.float().permute(0,2,1)) + t_cam_to_world[:,None] # BxHWx3, Bx3x3 -> HxWx3 + # world_coords_points = np.einsum("ij,hwj->hwi", R_cam_to_world, cam_coords_points) + t_cam_to_world + + return world_coords_points.reshape(B, H, W, 3) \ No newline at end of file diff --git a/vggt/vggt/utils/load_fn.py b/vggt/vggt/utils/load_fn.py new file mode 100644 index 0000000..1fa765a --- /dev/null +++ b/vggt/vggt/utils/load_fn.py @@ -0,0 +1,118 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch +from PIL import Image +from torchvision import transforms as TF + + +def load_and_preprocess_images(image_path_list): + """ + A quick start function to load and preprocess images for model input. + This assumes the images should have the same shape for easier batching, but our model can also work well with different shapes. + + Args: + image_path_list (list): List of paths to image files + + Returns: + torch.Tensor: Batched tensor of preprocessed images with shape (N, 3, H, W) + + Raises: + ValueError: If the input list is empty + + Notes: + - Images with different dimensions will be padded with white (value=1.0) + - A warning is printed when images have different shapes + - The function ensures width=518px while maintaining aspect ratio + - Height is adjusted to be divisible by 14 for compatibility with model requirements + """ + # Check for empty list + if len(image_path_list) == 0: + raise ValueError("At least 1 image is required") + + images = [] + alphas = [] + shapes = set() + to_tensor = TF.ToTensor() + + # First process all images and collect their shapes + # for image_path in image_path_list: + for img in image_path_list: + + # Open image + # img = Image.open(image_path) + img = img[0] + + # If there's an alpha channel, blend onto white background: + if img.mode == "RGBA": + # Create white background + alphas.append(to_tensor(img)[3:]) + # background = Image.new("RGBA", img.size, (255, 255, 255, 255)) + # Alpha composite onto the white background + # img = Image.alpha_composite(background, img) + + + # Now convert to "RGB" (this step assigns white for transparent areas) + img = img.convert("RGB") + + width, height = img.size + new_width = 518 + + # Calculate height maintaining aspect ratio, divisible by 14 + new_height = round(height * (new_width / width) / 14) * 14 + + # Resize with new dimensions (width, height) + + img = img.resize((new_width, new_height), Image.Resampling.BICUBIC) + img = to_tensor(img) # Convert to tensor (0, 1) + + # Center crop height if it's larger than 518 + + if new_height > 518: + start_y = (new_height - 518) // 2 + img = img[:, start_y : start_y + 518, :] + + shapes.add((img.shape[1], img.shape[2])) + images.append(img) + + # Check if we have different shapes + # In theory our model can also work well with different shapes + + if len(shapes) > 1: + print(f"Warning: Found images with different shapes: {shapes}") + # Find maximum dimensions + max_height = max(shape[0] for shape in shapes) + max_width = max(shape[1] for shape in shapes) + + # Pad images if necessary + padded_images = [] + for img in images: + h_padding = max_height - img.shape[1] + w_padding = max_width - img.shape[2] + + if h_padding > 0 or w_padding > 0: + pad_top = h_padding // 2 + pad_bottom = h_padding - pad_top + pad_left = w_padding // 2 + pad_right = w_padding - pad_left + + img = torch.nn.functional.pad( + img, (pad_left, pad_right, pad_top, pad_bottom), mode="constant", value=1.0 + ) + padded_images.append(img) + images = padded_images + + images = torch.stack(images) # concatenate images + alphas = torch.stack(alphas) # concatenate images + + # Ensure correct shape when single image + if len(image_path_list) == 1: + # Verify shape is (1, C, H, W) + if images.dim() == 3: + images = images.unsqueeze(0) + alphas = alphas.unsqueeze(0) + + return images, alphas diff --git a/vggt/vggt/utils/pose_enc.py b/vggt/vggt/utils/pose_enc.py new file mode 100644 index 0000000..2f98b08 --- /dev/null +++ b/vggt/vggt/utils/pose_enc.py @@ -0,0 +1,130 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import torch +from .rotation import quat_to_mat, mat_to_quat + + +def extri_intri_to_pose_encoding( + extrinsics, + intrinsics, + image_size_hw=None, # e.g., (256, 512) + pose_encoding_type="absT_quaR_FoV", +): + """Convert camera extrinsics and intrinsics to a compact pose encoding. + + This function transforms camera parameters into a unified pose encoding format, + which can be used for various downstream tasks like pose prediction or representation. + + Args: + extrinsics (torch.Tensor): Camera extrinsic parameters with shape BxSx3x4, + where B is batch size and S is sequence length. + In OpenCV coordinate system (x-right, y-down, z-forward), representing camera from world transformation. + The format is [R|t] where R is a 3x3 rotation matrix and t is a 3x1 translation vector. + intrinsics (torch.Tensor): Camera intrinsic parameters with shape BxSx3x3. + Defined in pixels, with format: + [[fx, 0, cx], + [0, fy, cy], + [0, 0, 1]] + where fx, fy are focal lengths and (cx, cy) is the principal point + image_size_hw (tuple): Tuple of (height, width) of the image in pixels. + Required for computing field of view values. For example: (256, 512). + pose_encoding_type (str): Type of pose encoding to use. Currently only + supports "absT_quaR_FoV" (absolute translation, quaternion rotation, field of view). + + Returns: + torch.Tensor: Encoded camera pose parameters with shape BxSx9. + For "absT_quaR_FoV" type, the 9 dimensions are: + - [:3] = absolute translation vector T (3D) + - [3:7] = rotation as quaternion quat (4D) + - [7:] = field of view (2D) + """ + + # extrinsics: BxSx3x4 + # intrinsics: BxSx3x3 + + if pose_encoding_type == "absT_quaR_FoV": + R = extrinsics[:, :, :3, :3] # BxSx3x3 + T = extrinsics[:, :, :3, 3] # BxSx3 + + quat = mat_to_quat(R) + # Note the order of h and w here + H, W = image_size_hw + fov_h = 2 * torch.atan((H / 2) / intrinsics[..., 1, 1]) + fov_w = 2 * torch.atan((W / 2) / intrinsics[..., 0, 0]) + pose_encoding = torch.cat([T, quat, fov_h[..., None], fov_w[..., None]], dim=-1).float() + else: + raise NotImplementedError + + return pose_encoding + + +def pose_encoding_to_extri_intri( + pose_encoding, + image_size_hw=None, # e.g., (256, 512) + pose_encoding_type="absT_quaR_FoV", + build_intrinsics=True, +): + """Convert a pose encoding back to camera extrinsics and intrinsics. + + This function performs the inverse operation of extri_intri_to_pose_encoding, + reconstructing the full camera parameters from the compact encoding. + + Args: + pose_encoding (torch.Tensor): Encoded camera pose parameters with shape BxSx9, + where B is batch size and S is sequence length. + For "absT_quaR_FoV" type, the 9 dimensions are: + - [:3] = absolute translation vector T (3D) + - [3:7] = rotation as quaternion quat (4D) + - [7:] = field of view (2D) + image_size_hw (tuple): Tuple of (height, width) of the image in pixels. + Required for reconstructing intrinsics from field of view values. + For example: (256, 512). + pose_encoding_type (str): Type of pose encoding used. Currently only + supports "absT_quaR_FoV" (absolute translation, quaternion rotation, field of view). + build_intrinsics (bool): Whether to reconstruct the intrinsics matrix. + If False, only extrinsics are returned and intrinsics will be None. + + Returns: + tuple: (extrinsics, intrinsics) + - extrinsics (torch.Tensor): Camera extrinsic parameters with shape BxSx3x4. + In OpenCV coordinate system (x-right, y-down, z-forward), representing camera from world + transformation. The format is [R|t] where R is a 3x3 rotation matrix and t is + a 3x1 translation vector. + - intrinsics (torch.Tensor or None): Camera intrinsic parameters with shape BxSx3x3, + or None if build_intrinsics is False. Defined in pixels, with format: + [[fx, 0, cx], + [0, fy, cy], + [0, 0, 1]] + where fx, fy are focal lengths and (cx, cy) is the principal point, + assumed to be at the center of the image (W/2, H/2). + """ + + intrinsics = None + + if pose_encoding_type == "absT_quaR_FoV": + T = pose_encoding[..., :3] + quat = pose_encoding[..., 3:7] + fov_h = pose_encoding[..., 7] + fov_w = pose_encoding[..., 8] + + R = quat_to_mat(quat) + extrinsics = torch.cat([R, T[..., None]], dim=-1) + + if build_intrinsics: + H, W = image_size_hw + fy = (H / 2.0) / torch.tan(fov_h / 2.0) + fx = (W / 2.0) / torch.tan(fov_w / 2.0) + intrinsics = torch.zeros(pose_encoding.shape[:2] + (3, 3), device=pose_encoding.device) + intrinsics[..., 0, 0] = fx + intrinsics[..., 1, 1] = fy + intrinsics[..., 0, 2] = W / 2 + intrinsics[..., 1, 2] = H / 2 + intrinsics[..., 2, 2] = 1.0 # Set the homogeneous coordinate to 1 + else: + raise NotImplementedError + + return extrinsics, intrinsics diff --git a/vggt/vggt/utils/rotation.py b/vggt/vggt/utils/rotation.py new file mode 100644 index 0000000..657583e --- /dev/null +++ b/vggt/vggt/utils/rotation.py @@ -0,0 +1,138 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +# Modified from PyTorch3D, https://github.com/facebookresearch/pytorch3d + +import torch +import numpy as np +import torch.nn.functional as F + + +def quat_to_mat(quaternions: torch.Tensor) -> torch.Tensor: + """ + Quaternion Order: XYZW or say ijkr, scalar-last + + Convert rotations given as quaternions to rotation matrices. + Args: + quaternions: quaternions with real part last, + as tensor of shape (..., 4). + + Returns: + Rotation matrices as tensor of shape (..., 3, 3). + """ + i, j, k, r = torch.unbind(quaternions, -1) + # pyre-fixme[58]: `/` is not supported for operand types `float` and `Tensor`. + two_s = 2.0 / (quaternions * quaternions).sum(-1) + + o = torch.stack( + ( + 1 - two_s * (j * j + k * k), + two_s * (i * j - k * r), + two_s * (i * k + j * r), + two_s * (i * j + k * r), + 1 - two_s * (i * i + k * k), + two_s * (j * k - i * r), + two_s * (i * k - j * r), + two_s * (j * k + i * r), + 1 - two_s * (i * i + j * j), + ), + -1, + ) + return o.reshape(quaternions.shape[:-1] + (3, 3)) + + +def mat_to_quat(matrix: torch.Tensor) -> torch.Tensor: + """ + Convert rotations given as rotation matrices to quaternions. + + Args: + matrix: Rotation matrices as tensor of shape (..., 3, 3). + + Returns: + quaternions with real part last, as tensor of shape (..., 4). + Quaternion Order: XYZW or say ijkr, scalar-last + """ + if matrix.size(-1) != 3 or matrix.size(-2) != 3: + raise ValueError(f"Invalid rotation matrix shape {matrix.shape}.") + + batch_dim = matrix.shape[:-2] + m00, m01, m02, m10, m11, m12, m20, m21, m22 = torch.unbind(matrix.reshape(batch_dim + (9,)), dim=-1) + + q_abs = _sqrt_positive_part( + torch.stack( + [ + 1.0 + m00 + m11 + m22, + 1.0 + m00 - m11 - m22, + 1.0 - m00 + m11 - m22, + 1.0 - m00 - m11 + m22, + ], + dim=-1, + ) + ) + + # we produce the desired quaternion multiplied by each of r, i, j, k + quat_by_rijk = torch.stack( + [ + # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and + # `int`. + torch.stack([q_abs[..., 0] ** 2, m21 - m12, m02 - m20, m10 - m01], dim=-1), + # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and + # `int`. + torch.stack([m21 - m12, q_abs[..., 1] ** 2, m10 + m01, m02 + m20], dim=-1), + # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and + # `int`. + torch.stack([m02 - m20, m10 + m01, q_abs[..., 2] ** 2, m12 + m21], dim=-1), + # pyre-fixme[58]: `**` is not supported for operand types `Tensor` and + # `int`. + torch.stack([m10 - m01, m20 + m02, m21 + m12, q_abs[..., 3] ** 2], dim=-1), + ], + dim=-2, + ) + + # We floor here at 0.1 but the exact level is not important; if q_abs is small, + # the candidate won't be picked. + flr = torch.tensor(0.1).to(dtype=q_abs.dtype, device=q_abs.device) + quat_candidates = quat_by_rijk / (2.0 * q_abs[..., None].max(flr)) + + # if not for numerical problems, quat_candidates[i] should be same (up to a sign), + # forall i; we pick the best-conditioned one (with the largest denominator) + out = quat_candidates[F.one_hot(q_abs.argmax(dim=-1), num_classes=4) > 0.5, :].reshape(batch_dim + (4,)) + + # Convert from rijk to ijkr + out = out[..., [1, 2, 3, 0]] + + out = standardize_quaternion(out) + + return out + + +def _sqrt_positive_part(x: torch.Tensor) -> torch.Tensor: + """ + Returns torch.sqrt(torch.max(0, x)) + but with a zero subgradient where x is 0. + """ + ret = torch.zeros_like(x) + positive_mask = x > 0 + if torch.is_grad_enabled(): + ret[positive_mask] = torch.sqrt(x[positive_mask]) + else: + ret = torch.where(positive_mask, torch.sqrt(x), ret) + return ret + + +def standardize_quaternion(quaternions: torch.Tensor) -> torch.Tensor: + """ + Convert a unit quaternion to a standard form: one in which the real + part is non negative. + + Args: + quaternions: Quaternions with real part last, + as tensor of shape (..., 4). + + Returns: + Standardized quaternions as tensor of shape (..., 4). + """ + return torch.where(quaternions[..., 3:4] < 0, -quaternions, quaternions) diff --git a/vggt/vggt/utils/visual_track.py b/vggt/vggt/utils/visual_track.py new file mode 100644 index 0000000..796c114 --- /dev/null +++ b/vggt/vggt/utils/visual_track.py @@ -0,0 +1,239 @@ +# Copyright (c) Meta Platforms, Inc. and affiliates. +# All rights reserved. +# +# This source code is licensed under the license found in the +# LICENSE file in the root directory of this source tree. + +import cv2 +import torch +import numpy as np +import os + + +def color_from_xy(x, y, W, H, cmap_name="hsv"): + """ + Map (x, y) -> color in (R, G, B). + 1) Normalize x,y to [0,1]. + 2) Combine them into a single scalar c in [0,1]. + 3) Use matplotlib's colormap to convert c -> (R,G,B). + + You can customize step 2, e.g., c = (x + y)/2, or some function of (x, y). + """ + import matplotlib.cm + import matplotlib.colors + + x_norm = x / max(W - 1, 1) + y_norm = y / max(H - 1, 1) + # Simple combination: + c = (x_norm + y_norm) / 2.0 + + cmap = matplotlib.cm.get_cmap(cmap_name) + # cmap(c) -> (r,g,b,a) in [0,1] + rgba = cmap(c) + r, g, b = rgba[0], rgba[1], rgba[2] + return (r, g, b) # in [0,1], RGB order + + +def get_track_colors_by_position(tracks_b, vis_mask_b=None, image_width=None, image_height=None, cmap_name="hsv"): + """ + Given all tracks in one sample (b), compute a (N,3) array of RGB color values + in [0,255]. The color is determined by the (x,y) position in the first + visible frame for each track. + + Args: + tracks_b: Tensor of shape (S, N, 2). (x,y) for each track in each frame. + vis_mask_b: (S, N) boolean mask; if None, assume all are visible. + image_width, image_height: used for normalizing (x, y). + cmap_name: for matplotlib (e.g., 'hsv', 'rainbow', 'jet'). + + Returns: + track_colors: np.ndarray of shape (N, 3), each row is (R,G,B) in [0,255]. + """ + S, N, _ = tracks_b.shape + track_colors = np.zeros((N, 3), dtype=np.uint8) + + if vis_mask_b is None: + # treat all as visible + vis_mask_b = torch.ones(S, N, dtype=torch.bool, device=tracks_b.device) + + for i in range(N): + # Find first visible frame for track i + visible_frames = torch.where(vis_mask_b[:, i])[0] + if len(visible_frames) == 0: + # track is never visible; just assign black or something + track_colors[i] = (0, 0, 0) + continue + + first_s = int(visible_frames[0].item()) + # use that frame's (x,y) + x, y = tracks_b[first_s, i].tolist() + + # map (x,y) -> (R,G,B) in [0,1] + r, g, b = color_from_xy(x, y, W=image_width, H=image_height, cmap_name=cmap_name) + # scale to [0,255] + r, g, b = int(r * 255), int(g * 255), int(b * 255) + track_colors[i] = (r, g, b) + + return track_colors + + +def visualize_tracks_on_images( + images, + tracks, + track_vis_mask=None, + out_dir="track_visuals_concat_by_xy", + image_format="CHW", # "CHW" or "HWC" + normalize_mode="[0,1]", + cmap_name="hsv", # e.g. "hsv", "rainbow", "jet" + frames_per_row=4, # New parameter for grid layout + save_grid=True, # Flag to control whether to save the grid image +): + """ + Visualizes frames in a grid layout with specified frames per row. + Each track's color is determined by its (x,y) position + in the first visible frame (or frame 0 if always visible). + Finally convert the BGR result to RGB before saving. + Also saves each individual frame as a separate PNG file. + + Args: + images: torch.Tensor (S, 3, H, W) if CHW or (S, H, W, 3) if HWC. + tracks: torch.Tensor (S, N, 2), last dim = (x, y). + track_vis_mask: torch.Tensor (S, N) or None. + out_dir: folder to save visualizations. + image_format: "CHW" or "HWC". + normalize_mode: "[0,1]", "[-1,1]", or None for direct raw -> 0..255 + cmap_name: a matplotlib colormap name for color_from_xy. + frames_per_row: number of frames to display in each row of the grid. + save_grid: whether to save all frames in one grid image. + + Returns: + None (saves images in out_dir). + """ + + if len(tracks.shape) == 4: + tracks = tracks.squeeze(0) + images = images.squeeze(0) + if track_vis_mask is not None: + track_vis_mask = track_vis_mask.squeeze(0) + + import matplotlib + + matplotlib.use("Agg") # for non-interactive (optional) + + os.makedirs(out_dir, exist_ok=True) + + S = images.shape[0] + _, N, _ = tracks.shape # (S, N, 2) + + # Move to CPU + images = images.cpu().clone() + tracks = tracks.cpu().clone() + if track_vis_mask is not None: + track_vis_mask = track_vis_mask.cpu().clone() + + # Infer H, W from images shape + if image_format == "CHW": + # e.g. images[s].shape = (3, H, W) + H, W = images.shape[2], images.shape[3] + else: + # e.g. images[s].shape = (H, W, 3) + H, W = images.shape[1], images.shape[2] + + # Pre-compute the color for each track i based on first visible position + track_colors_rgb = get_track_colors_by_position( + tracks, # shape (S, N, 2) + vis_mask_b=track_vis_mask if track_vis_mask is not None else None, + image_width=W, + image_height=H, + cmap_name=cmap_name, + ) + + # We'll accumulate each frame's drawn image in a list + frame_images = [] + + for s in range(S): + # shape => either (3, H, W) or (H, W, 3) + img = images[s] + + # Convert to (H, W, 3) + if image_format == "CHW": + img = img.permute(1, 2, 0) # (H, W, 3) + # else "HWC", do nothing + + img = img.numpy().astype(np.float32) + + # Scale to [0,255] if needed + if normalize_mode == "[0,1]": + img = np.clip(img, 0, 1) * 255.0 + elif normalize_mode == "[-1,1]": + img = (img + 1.0) * 0.5 * 255.0 + img = np.clip(img, 0, 255.0) + # else no normalization + + # Convert to uint8 + img = img.astype(np.uint8) + + # For drawing in OpenCV, convert to BGR + img_bgr = cv2.cvtColor(img, cv2.COLOR_RGB2BGR) + + # Draw each visible track + cur_tracks = tracks[s] # shape (N, 2) + if track_vis_mask is not None: + valid_indices = torch.where(track_vis_mask[s])[0] + else: + valid_indices = range(N) + + cur_tracks_np = cur_tracks.numpy() + for i in valid_indices: + x, y = cur_tracks_np[i] + pt = (int(round(x)), int(round(y))) + + # track_colors_rgb[i] is (R,G,B). For OpenCV circle, we need BGR + R, G, B = track_colors_rgb[i] + color_bgr = (int(B), int(G), int(R)) + cv2.circle(img_bgr, pt, radius=3, color=color_bgr, thickness=-1) + + # Convert back to RGB for consistent final saving: + img_rgb = cv2.cvtColor(img_bgr, cv2.COLOR_BGR2RGB) + + # Save individual frame + frame_path = os.path.join(out_dir, f"frame_{s:04d}.png") + # Convert to BGR for OpenCV imwrite + frame_bgr = cv2.cvtColor(img_rgb, cv2.COLOR_RGB2BGR) + cv2.imwrite(frame_path, frame_bgr) + + frame_images.append(img_rgb) + + # Only create and save the grid image if save_grid is True + if save_grid: + # Calculate grid dimensions + num_rows = (S + frames_per_row - 1) // frames_per_row # Ceiling division + + # Create a grid of images + grid_img = None + for row in range(num_rows): + start_idx = row * frames_per_row + end_idx = min(start_idx + frames_per_row, S) + + # Concatenate this row horizontally + row_img = np.concatenate(frame_images[start_idx:end_idx], axis=1) + + # If this row has fewer than frames_per_row images, pad with black + if end_idx - start_idx < frames_per_row: + padding_width = (frames_per_row - (end_idx - start_idx)) * W + padding = np.zeros((H, padding_width, 3), dtype=np.uint8) + row_img = np.concatenate([row_img, padding], axis=1) + + # Add this row to the grid + if grid_img is None: + grid_img = row_img + else: + grid_img = np.concatenate([grid_img, row_img], axis=0) + + out_path = os.path.join(out_dir, "tracks_grid.png") + # Convert back to BGR for OpenCV imwrite + grid_img_bgr = cv2.cvtColor(grid_img, cv2.COLOR_RGB2BGR) + cv2.imwrite(out_path, grid_img_bgr) + print(f"[INFO] Saved color-by-XY track visualization grid -> {out_path}") + + print(f"[INFO] Saved {S} individual frames to {out_dir}/frame_*.png")