diff --git a/LongCat/LongCatAvatar_testing_wip.json b/LongCat/LongCatAvatar_testing_wip.json
new file mode 100644
index 0000000..3d8f12f
--- /dev/null
+++ b/LongCat/LongCatAvatar_testing_wip.json
@@ -0,0 +1,5507 @@
+{
+ "id": "8b7a9a57-2303-4ef5-9fc2-bf41713bd1fc",
+ "revision": 0,
+ "last_node_id": 469,
+ "last_link_id": 848,
+ "nodes": [
+ {
+ "id": 239,
+ "type": "MarkdownNote",
+ "pos": [
+ 623.767822265625,
+ -2378.682861328125
+ ],
+ "size": [
+ 452.8328857421875,
+ 214.95587158203125
+ ],
+ "flags": {},
+ "order": 0,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "title": "Model links",
+ "properties": {},
+ "widgets_values": [
+ "You can't mix GGUF MultiTalk with non-GGUF main model, but you can mix different GGUF Qtypes with eachother.\n\nGGUF:\n\n[https://huggingface.co/city96/Wan2.1-I2V-14B-480P-gguf/tree/main](https://huggingface.co/city96/Wan2.1-I2V-14B-480P-gguf/tree/main)\n\n[https://huggingface.co/Kijai/WanVideo_comfy_GGUF/tree/main/InfiniteTalk](https://huggingface.co/Kijai/WanVideo_comfy_GGUF/tree/main/InfiniteTalk)\n\nFp8:\n\n[https://huggingface.co/Kijai/WanVideo_comfy_fp8_scaled](https://huggingface.co/Kijai/WanVideo_comfy_fp8_scaled/tree/main/InfiniteTalk)\n\nfp16:\n\n[https://huggingface.co/Kijai/WanVideo_comfy/tree/main/InfiniteTalk](https://huggingface.co/Kijai/WanVideo_comfy/tree/main/InfiniteTalk)\n"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 260,
+ "type": "SetNode",
+ "pos": [
+ 1843.7393798828125,
+ -2738.34326171875
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 71,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "WANVIDEOMODEL",
+ "type": "WANVIDEOMODEL",
+ "link": 463
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_wanmodel",
+ "properties": {
+ "previousName": "wanmodel"
+ },
+ "widgets_values": [
+ "wanmodel"
+ ]
+ },
+ {
+ "id": 238,
+ "type": "CLIPVisionLoader",
+ "pos": [
+ 1406.407470703125,
+ -2358.7119140625
+ ],
+ "size": [
+ 270,
+ 58
+ ],
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "CLIP_VISION",
+ "type": "CLIP_VISION",
+ "links": [
+ 466
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "CLIPVisionLoader",
+ "cnr_id": "comfy-core",
+ "ver": "0.3.50"
+ },
+ "widgets_values": [
+ "clip_vision_h.safetensors"
+ ],
+ "color": "#233",
+ "bgcolor": "#355"
+ },
+ {
+ "id": 264,
+ "type": "SetNode",
+ "pos": [
+ 1711.6546630859375,
+ -2336.708984375
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 55,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "CLIP_VISION",
+ "type": "CLIP_VISION",
+ "link": 466
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_clip_vision_model",
+ "properties": {
+ "previousName": "clip_vision_model"
+ },
+ "widgets_values": [
+ "clip_vision_model"
+ ]
+ },
+ {
+ "id": 247,
+ "type": "SetNode",
+ "pos": [
+ 2497.053955078125,
+ -2320.50537109375
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 57,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 441
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_width",
+ "properties": {
+ "previousName": "width"
+ },
+ "widgets_values": [
+ "width"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 248,
+ "type": "SetNode",
+ "pos": [
+ 2495.547607421875,
+ -2143.893310546875
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 58,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 442
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_height",
+ "properties": {
+ "previousName": "height"
+ },
+ "widgets_values": [
+ "height"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 282,
+ "type": "GetNode",
+ "pos": [
+ 1036.8980712890625,
+ -1798.782958984375
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 2,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 495
+ ]
+ }
+ ],
+ "title": "Get_height",
+ "properties": {},
+ "widgets_values": [
+ "height"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 293,
+ "type": "PreviewAny",
+ "pos": [
+ 1696.14794921875,
+ -1108.7257080078125
+ ],
+ "size": [
+ 210,
+ 112
+ ],
+ "flags": {},
+ "order": 78,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "source",
+ "type": "*",
+ "link": 520
+ }
+ ],
+ "outputs": [],
+ "properties": {
+ "Node name for S&R": "PreviewAny",
+ "cnr_id": "comfy-core",
+ "ver": "0.3.50"
+ },
+ "widgets_values": [
+ null,
+ null,
+ null
+ ]
+ },
+ {
+ "id": 271,
+ "type": "SetNode",
+ "pos": [
+ 2660.10888671875,
+ -2152.97998046875
+ ],
+ "size": [
+ 210,
+ 50
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 61,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 472
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_max_frames",
+ "properties": {
+ "previousName": "max_frames"
+ },
+ "widgets_values": [
+ "max_frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 272,
+ "type": "GetNode",
+ "pos": [
+ 1098.8834228515625,
+ -1057.341064453125
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 3,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 529
+ ]
+ }
+ ],
+ "title": "Get_max_frames",
+ "properties": {},
+ "widgets_values": [
+ "max_frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 283,
+ "type": "GetNode",
+ "pos": [
+ 1039.0777587890625,
+ -1849.6192626953125
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 4,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 494
+ ]
+ }
+ ],
+ "title": "Get_width",
+ "properties": {},
+ "widgets_values": [
+ "width"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 299,
+ "type": "Note",
+ "pos": [
+ 1424.5263671875,
+ -2250.27001953125
+ ],
+ "size": [
+ 290.9361267089844,
+ 88
+ ],
+ "flags": {},
+ "order": 5,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {},
+ "widgets_values": [
+ "Clip vision is not strictly necessary\n\nAny I2V model should work, MAGREF can be interesting to play with as well."
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 240,
+ "type": "SetNode",
+ "pos": [
+ 1761.82666015625,
+ -3007.7861328125
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 60,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "WANVAE",
+ "type": "WANVAE",
+ "link": 436
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_VAE",
+ "properties": {
+ "previousName": "VAE"
+ },
+ "widgets_values": [
+ "VAE"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 284,
+ "type": "LoadImage",
+ "pos": [
+ 634.8666381835938,
+ -1939.31591796875
+ ],
+ "size": [
+ 274.080078125,
+ 314
+ ],
+ "flags": {},
+ "order": 6,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 496
+ ]
+ },
+ {
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "LoadImage",
+ "cnr_id": "comfy-core",
+ "ver": "0.3.50"
+ },
+ "widgets_values": [
+ "man.png",
+ "image"
+ ]
+ },
+ {
+ "id": 245,
+ "type": "INTConstant",
+ "pos": [
+ 2401.922607421875,
+ -2438.11865234375
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {},
+ "order": 7,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "INT",
+ "links": [
+ 441
+ ]
+ }
+ ],
+ "title": "Width",
+ "properties": {
+ "Node name for S&R": "INTConstant",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324"
+ },
+ "widgets_values": [
+ 832
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 246,
+ "type": "INTConstant",
+ "pos": [
+ 2404.592529296875,
+ -2258.70654296875
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {},
+ "order": 8,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "INT",
+ "links": [
+ 442
+ ]
+ }
+ ],
+ "title": "Height",
+ "properties": {
+ "Node name for S&R": "INTConstant",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324"
+ },
+ "widgets_values": [
+ 480
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 302,
+ "type": "MelBandRoFormerSampler",
+ "pos": [
+ 1180.8266434257762,
+ -1409.5682311530113
+ ],
+ "size": [
+ 273.784147927401,
+ 46
+ ],
+ "flags": {},
+ "order": 69,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "model",
+ "type": "MELROFORMERMODEL",
+ "link": 543
+ },
+ {
+ "name": "audio",
+ "type": "AUDIO",
+ "link": 570
+ }
+ ],
+ "outputs": [
+ {
+ "name": "vocals",
+ "type": "AUDIO",
+ "links": [
+ 564
+ ]
+ },
+ {
+ "name": "instruments",
+ "type": "AUDIO",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "MelBandRoFormerSampler",
+ "cnr_id": "ComfyUI-MelBandRoFormer",
+ "ver": "b68d9077815387b64d596f8c39607052b95b6eba"
+ },
+ "widgets_values": []
+ },
+ {
+ "id": 281,
+ "type": "ImageResizeKJv2",
+ "pos": [
+ 1207.9007568359375,
+ -1937.6312255859375
+ ],
+ "size": [
+ 270,
+ 336
+ ],
+ "flags": {},
+ "order": 56,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 496
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "widget": {
+ "name": "width"
+ },
+ "link": 494
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "widget": {
+ "name": "height"
+ },
+ "link": 495
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 554
+ ]
+ },
+ {
+ "name": "width",
+ "type": "INT",
+ "links": []
+ },
+ {
+ "name": "height",
+ "type": "INT",
+ "links": []
+ },
+ {
+ "name": "mask",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ImageResizeKJv2",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "f7eb33abc80a2aded1b46dff0dd14d07856a7d50"
+ },
+ "widgets_values": [
+ 832,
+ 480,
+ "lanczos",
+ "crop",
+ "0, 0, 0",
+ "center",
+ 16,
+ "cpu",
+ "
| Output: | 1 x 832 x 480 | 4.57MB |
"
+ ]
+ },
+ {
+ "id": 332,
+ "type": "SetNode",
+ "pos": [
+ 3483.584205250587,
+ -2303.3767069540318
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 66,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "WANVIDEOSCHEDULER",
+ "type": "WANVIDEOSCHEDULER",
+ "link": 598
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_scheduler",
+ "properties": {
+ "previousName": "scheduler"
+ },
+ "widgets_values": [
+ "scheduler"
+ ]
+ },
+ {
+ "id": 125,
+ "type": "LoadAudio",
+ "pos": [
+ 453.9859313964844,
+ -1425.9039306640625
+ ],
+ "size": [
+ 404.9698791503906,
+ 136
+ ],
+ "flags": {},
+ "order": 9,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 569
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "LoadAudio",
+ "cnr_id": "comfy-core",
+ "ver": "0.3.41"
+ },
+ "widgets_values": [
+ "man.mp3",
+ null,
+ null
+ ]
+ },
+ {
+ "id": 134,
+ "type": "WanVideoBlockSwap",
+ "pos": [
+ 587.2186279296875,
+ -3026.727294921875
+ ],
+ "size": [
+ 281.404296875,
+ 202
+ ],
+ "flags": {},
+ "order": 10,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "block_swap_args",
+ "type": "BLOCKSWAPARGS",
+ "links": [
+ 362
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoBlockSwap",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0"
+ },
+ "widgets_values": [
+ 20,
+ false,
+ false,
+ true,
+ 0,
+ 1,
+ false
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 122,
+ "type": "WanVideoModelLoader",
+ "pos": [
+ 1121.4686279296875,
+ -2764.419921875
+ ],
+ "size": [
+ 668.1777954101562,
+ 338
+ ],
+ "flags": {},
+ "order": 67,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "compile_args",
+ "shape": 7,
+ "type": "WANCOMPILEARGS",
+ "link": 770
+ },
+ {
+ "name": "block_swap_args",
+ "shape": 7,
+ "type": "BLOCKSWAPARGS",
+ "link": 362
+ },
+ {
+ "name": "lora",
+ "shape": 7,
+ "type": "WANVIDLORA",
+ "link": 848
+ },
+ {
+ "name": "vram_management_args",
+ "shape": 7,
+ "type": "VRAM_MANAGEMENTARGS",
+ "link": null
+ },
+ {
+ "name": "extra_model",
+ "shape": 7,
+ "type": "VACEPATH",
+ "link": null
+ },
+ {
+ "name": "fantasytalking_model",
+ "shape": 7,
+ "type": "FANTASYTALKINGMODEL",
+ "link": null
+ },
+ {
+ "name": "multitalk_model",
+ "shape": 7,
+ "type": "MULTITALKMODEL",
+ "link": null
+ },
+ {
+ "name": "fantasyportrait_model",
+ "shape": 7,
+ "type": "FANTASYPORTRAITMODEL",
+ "link": null
+ },
+ {
+ "name": "vace_model",
+ "shape": 7,
+ "type": "VACEPATH",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "model",
+ "type": "WANVIDEOMODEL",
+ "links": [
+ 463
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoModelLoader",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0"
+ },
+ "widgets_values": [
+ "LongCat/LongCat-Avatar_bf16.safetensors",
+ "bf16",
+ "disabled",
+ "offload_device",
+ "sageattn",
+ "default"
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 129,
+ "type": "WanVideoVAELoader",
+ "pos": [
+ 1379.0877685546875,
+ -3038.031494140625
+ ],
+ "size": [
+ 315,
+ 106
+ ],
+ "flags": {},
+ "order": 11,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "compile_args",
+ "shape": 7,
+ "type": "WANCOMPILEARGS",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "slot_index": 0,
+ "links": [
+ 436
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoVAELoader",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "c3ee35f3ece76e38099dc516182d69b406e16772"
+ },
+ "widgets_values": [
+ "Wan2_1_VAE_bf16.safetensors",
+ "bf16",
+ false
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 317,
+ "type": "TrimAudioDuration",
+ "pos": [
+ 913.1688728460073,
+ -1244.2847184950829
+ ],
+ "size": [
+ 270,
+ 82
+ ],
+ "flags": {},
+ "order": 59,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "audio",
+ "type": "AUDIO",
+ "link": 569
+ }
+ ],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 570,
+ 588
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "TrimAudioDuration",
+ "cnr_id": "comfy-core",
+ "ver": "0.5.0"
+ },
+ "widgets_values": [
+ 5,
+ 60
+ ]
+ },
+ {
+ "id": 270,
+ "type": "INTConstant",
+ "pos": [
+ 2655.538330078125,
+ -2261.7578125
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {},
+ "order": 12,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "INT",
+ "links": [
+ 472
+ ]
+ }
+ ],
+ "title": "Max frames",
+ "properties": {
+ "Node name for S&R": "INTConstant",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324"
+ },
+ "widgets_values": [
+ 1000
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 177,
+ "type": "WanVideoTorchCompileSettings",
+ "pos": [
+ 970.6024169921875,
+ -3048.0654296875
+ ],
+ "size": [
+ 342.74609375,
+ 250
+ ],
+ "flags": {},
+ "order": 13,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "torch_compile_args",
+ "type": "WANCOMPILEARGS",
+ "links": [
+ 770
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoTorchCompileSettings",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "f3614e6720744247f3211d60f7b9333f43572384"
+ },
+ "widgets_values": [
+ "inductor",
+ true,
+ "default",
+ false,
+ 64,
+ true,
+ 128,
+ false,
+ false
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 241,
+ "type": "WanVideoTextEncodeCached",
+ "pos": [
+ 2281.288330078125,
+ -1946.174072265625
+ ],
+ "size": [
+ 465.5179138183594,
+ 386.354248046875
+ ],
+ "flags": {},
+ "order": 14,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "extender_args",
+ "shape": 7,
+ "type": "WANVIDEOPROMPTEXTENDER_ARGS",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "text_embeds",
+ "type": "WANVIDEOTEXTEMBEDS",
+ "links": [
+ 586,
+ 786
+ ]
+ },
+ {
+ "name": "negative_text_embeds",
+ "type": "WANVIDEOTEXTEMBEDS",
+ "links": null
+ },
+ {
+ "name": "positive_prompt",
+ "type": "STRING",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoTextEncodeCached",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ff779c91714d8ee3484cd4119b082c72a1734b72"
+ },
+ "widgets_values": [
+ "umt5-xxl-enc-bf16.safetensors",
+ "bf16",
+ "A western man stands on stage under dramatic lighting, holding a microphone close to their mouth. Wearing a vibrant red jacket with gold embroidery, the singer is speaking while smoke swirls around them, creating a dynamic and atmospheric scene.",
+ "Close-up, bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards",
+ "disabled",
+ true,
+ "gpu"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 421,
+ "type": "SetNode",
+ "pos": [
+ 2577.1444054849508,
+ -2021.7597809680915
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 62,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "WANVIDEOTEXTEMBEDS",
+ "type": "WANVIDEOTEXTEMBEDS",
+ "link": 786
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_text_embeds",
+ "properties": {
+ "previousName": "text_embeds"
+ },
+ "widgets_values": [
+ "text_embeds"
+ ]
+ },
+ {
+ "id": 427,
+ "type": "FloatConstant",
+ "pos": [
+ 2663.1993102045894,
+ -2430.058297029451
+ ],
+ "size": [
+ 200,
+ 58
+ ],
+ "flags": {},
+ "order": 15,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "FLOAT",
+ "links": [
+ 796
+ ]
+ }
+ ],
+ "title": "cfg",
+ "properties": {
+ "Node name for S&R": "FloatConstant"
+ },
+ "widgets_values": [
+ 1
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 429,
+ "type": "GetNode",
+ "pos": [
+ 3547.4961269124406,
+ -1746.947487708598
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 16,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "FLOAT",
+ "type": "FLOAT",
+ "links": [
+ 797
+ ]
+ }
+ ],
+ "title": "Get_cfg",
+ "properties": {},
+ "widgets_values": [
+ "cfg"
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 431,
+ "type": "GetNode",
+ "pos": [
+ 4889.217494995688,
+ -1785.313527456512
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 17,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 799
+ ]
+ }
+ ],
+ "title": "Get_overlap",
+ "properties": {},
+ "widgets_values": [
+ "overlap"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 358,
+ "type": "WanVideoEncode",
+ "pos": [
+ 6033.426552518465,
+ -2106.8177479295186
+ ],
+ "size": [
+ 273.705078125,
+ 242
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 87,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 658
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 657
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 662
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoEncode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ 0,
+ 1
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 331,
+ "type": "GetNode",
+ "pos": [
+ 5851.264114149541,
+ -2158.9826716244315
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 18,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVAE",
+ "type": "WANVAE",
+ "links": [
+ 597,
+ 658
+ ]
+ }
+ ],
+ "title": "Get_VAE",
+ "properties": {},
+ "widgets_values": [
+ "VAE"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 359,
+ "type": "ReplaceVideoLatentFrames",
+ "pos": [
+ 6027.1360600789685,
+ -2171.887335754559
+ ],
+ "size": [
+ 280.355078125,
+ 78
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 89,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "destination",
+ "type": "LATENT",
+ "link": 660
+ },
+ {
+ "name": "source",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 662
+ }
+ ],
+ "outputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 661
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ReplaceVideoLatentFrames",
+ "cnr_id": "comfy-core",
+ "ver": "0.5.0"
+ },
+ "widgets_values": [
+ 0
+ ]
+ },
+ {
+ "id": 430,
+ "type": "GetNode",
+ "pos": [
+ 5500.715753469097,
+ -1769.775882664444
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 19,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "FLOAT",
+ "type": "FLOAT",
+ "links": [
+ 798
+ ]
+ }
+ ],
+ "title": "Get_cfg",
+ "properties": {},
+ "widgets_values": [
+ "cfg"
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 333,
+ "type": "GetNode",
+ "pos": [
+ 5476.389981217721,
+ -1911.7305257347355
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 20,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOSCHEDULER",
+ "type": "WANVIDEOSCHEDULER",
+ "links": [
+ 599
+ ]
+ }
+ ],
+ "title": "Get_scheduler",
+ "properties": {},
+ "widgets_values": [
+ "scheduler"
+ ]
+ },
+ {
+ "id": 334,
+ "type": "GetNode",
+ "pos": [
+ 5477.073656625263,
+ -1964.0034735327247
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 21,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOMODEL",
+ "type": "WANVIDEOMODEL",
+ "links": [
+ 600
+ ]
+ }
+ ],
+ "title": "Get_wanmodel",
+ "properties": {},
+ "widgets_values": [
+ "wanmodel"
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 254,
+ "type": "GetNode",
+ "pos": [
+ 3729.2348190478037,
+ -1412.7409186976204
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 22,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 587
+ ]
+ }
+ ],
+ "title": "Get_input_audio",
+ "properties": {},
+ "widgets_values": [
+ "input_audio"
+ ]
+ },
+ {
+ "id": 324,
+ "type": "WanVideoSamplerv2",
+ "pos": [
+ 3691.387697836247,
+ -1904.3182035540826
+ ],
+ "size": [
+ 295.37109375,
+ 426.8679387019231
+ ],
+ "flags": {},
+ "order": 77,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "model",
+ "type": "WANVIDEOMODEL",
+ "link": 583
+ },
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "link": 629
+ },
+ {
+ "name": "scheduler",
+ "type": "WANVIDEOSCHEDULER",
+ "link": 584
+ },
+ {
+ "name": "text_embeds",
+ "shape": 7,
+ "type": "WANVIDEOTEXTEMBEDS",
+ "link": 586
+ },
+ {
+ "name": "samples",
+ "shape": 7,
+ "type": "LATENT",
+ "link": null
+ },
+ {
+ "name": "extra_args",
+ "shape": 7,
+ "type": "WANVIDSAMPLEREXTRAARGS",
+ "link": null
+ },
+ {
+ "name": "cfg",
+ "type": "FLOAT",
+ "widget": {
+ "name": "cfg"
+ },
+ "link": 797
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 804
+ ]
+ },
+ {
+ "name": "denoised_samples",
+ "type": "LATENT",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoSamplerv2",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 1,
+ 1,
+ "fixed",
+ true,
+ false
+ ]
+ },
+ {
+ "id": 434,
+ "type": "Reroute",
+ "pos": [
+ 4138.790765629495,
+ -1902.7746618128708
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 79,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 804
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "LATENT",
+ "links": [
+ 805,
+ 806
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 432,
+ "type": "GetNode",
+ "pos": [
+ 5684.035848672816,
+ -2280.6895397335575
+ ],
+ "size": [
+ 210,
+ 50
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 23,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 800,
+ 802
+ ]
+ }
+ ],
+ "title": "Get_overlap",
+ "properties": {},
+ "widgets_values": [
+ "overlap"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 437,
+ "type": "Reroute",
+ "pos": [
+ 5518.907286717461,
+ -2376.4027170928425
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 83,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 809
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "IMAGE",
+ "links": [
+ 810,
+ 811
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 440,
+ "type": "GetNode",
+ "pos": [
+ 4892.130729575904,
+ -1836.9559441141416
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 24,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 814
+ ]
+ }
+ ],
+ "title": "Get_frames",
+ "properties": {},
+ "widgets_values": [
+ "frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 261,
+ "type": "GetNode",
+ "pos": [
+ 3694.968011543167,
+ -1961.958040031301
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 25,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOMODEL",
+ "type": "WANVIDEOMODEL",
+ "links": [
+ 583
+ ]
+ }
+ ],
+ "title": "Get_wanmodel",
+ "properties": {},
+ "widgets_values": [
+ "wanmodel"
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 438,
+ "type": "INTConstant",
+ "pos": [
+ 2650.9810689169576,
+ -2664.3064774938034
+ ],
+ "size": [
+ 200,
+ 58
+ ],
+ "flags": {},
+ "order": 26,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "INT",
+ "links": [
+ 813
+ ]
+ }
+ ],
+ "title": "frames_per_window",
+ "properties": {
+ "Node name for S&R": "INTConstant"
+ },
+ "widgets_values": [
+ 93
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 439,
+ "type": "SetNode",
+ "pos": [
+ 2899.513214973296,
+ -2637.0549377782186
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 64,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 813
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_frames",
+ "properties": {
+ "previousName": "frames"
+ },
+ "widgets_values": [
+ "frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 425,
+ "type": "SetNode",
+ "pos": [
+ 2896.8877238392656,
+ -2508.664563650443
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 65,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 794
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_overlap",
+ "properties": {
+ "previousName": "overlap"
+ },
+ "widgets_values": [
+ "overlap"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 428,
+ "type": "SetNode",
+ "pos": [
+ 2904.9087841665228,
+ -2404.003672002015
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 63,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "FLOAT",
+ "type": "FLOAT",
+ "link": 796
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_cfg",
+ "properties": {
+ "previousName": "cfg"
+ },
+ "widgets_values": [
+ "cfg"
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 442,
+ "type": "GetNode",
+ "pos": [
+ 2961.7813895890044,
+ -1740.5456406171209
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 27,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 815
+ ]
+ }
+ ],
+ "title": "Get_frames",
+ "properties": {},
+ "widgets_values": [
+ "frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 413,
+ "type": "GetNode",
+ "pos": [
+ 5071.410986510317,
+ -1685.578721396226
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 28,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 775
+ ]
+ }
+ ],
+ "title": "Get_ref_latent",
+ "properties": {},
+ "widgets_values": [
+ "ref_latent"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 443,
+ "type": "GetNode",
+ "pos": [
+ 2961.7813895890044,
+ -1822.2995362330341
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 29,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 816
+ ]
+ }
+ ],
+ "title": "Get_ref_latent",
+ "properties": {},
+ "widgets_values": [
+ "ref_latent"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 411,
+ "type": "SetNode",
+ "pos": [
+ 1910.4741649115756,
+ -1927.6893724506806
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 72,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "link": 776
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_ref_latent",
+ "properties": {
+ "previousName": "ref_latent"
+ },
+ "widgets_values": [
+ "ref_latent"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 330,
+ "type": "GetNode",
+ "pos": [
+ 4100.336376587513,
+ -2094.413489687845
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 30,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVAE",
+ "type": "WANVAE",
+ "links": [
+ 596
+ ]
+ }
+ ],
+ "title": "Get_VAE",
+ "properties": {},
+ "widgets_values": [
+ "VAE"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 444,
+ "type": "GetNode",
+ "pos": [
+ 1580.6838604750237,
+ -1677.3225175650712
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 31,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVAE",
+ "type": "WANVAE",
+ "links": [
+ 817
+ ]
+ }
+ ],
+ "title": "Get_VAE",
+ "properties": {},
+ "widgets_values": [
+ "VAE"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 312,
+ "type": "WanVideoEncode",
+ "pos": [
+ 1572.3931759352845,
+ -1958.3412486430054
+ ],
+ "size": [
+ 273.705078125,
+ 242
+ ],
+ "flags": {},
+ "order": 68,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 817
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 554
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 776
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoEncode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ 0,
+ 1
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 313,
+ "type": "WanVideoDecode",
+ "pos": [
+ 4250.386298022776,
+ -2127.7849994112507
+ ],
+ "size": [
+ 270,
+ 198
+ ],
+ "flags": {},
+ "order": 80,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 596
+ },
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "link": 805
+ }
+ ],
+ "outputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "links": [
+ 574,
+ 650
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoDecode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ "default"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 345,
+ "type": "WanVideoLongCatAvatarExtendEmbeds",
+ "pos": [
+ 3130.407144659971,
+ -1832.4734688295339
+ ],
+ "size": [
+ 379.476171875,
+ 170
+ ],
+ "flags": {},
+ "order": 74,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "prev_latents",
+ "type": "LATENT",
+ "link": 816
+ },
+ {
+ "name": "audio_embeds",
+ "type": "MULTITALK_EMBEDS",
+ "link": 628
+ },
+ {
+ "name": "ref_latent",
+ "shape": 7,
+ "type": "LATENT",
+ "link": null
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 815
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "links": [
+ 629
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoLongCatAvatarExtendEmbeds",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 93,
+ 1,
+ 0,
+ "pad_with_start"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 341,
+ "type": "ImageBatchExtendWithOverlap",
+ "pos": [
+ 6285.035226507292,
+ -2371.049091818544
+ ],
+ "size": [
+ 321.6236328125,
+ 146
+ ],
+ "flags": {},
+ "order": 91,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "source_images",
+ "type": "IMAGE",
+ "link": 810
+ },
+ {
+ "name": "new_images",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 688
+ },
+ {
+ "name": "overlap",
+ "type": "INT",
+ "widget": {
+ "name": "overlap"
+ },
+ "link": 802
+ }
+ ],
+ "outputs": [
+ {
+ "name": "source_images",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "start_images",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "extended_images",
+ "type": "IMAGE",
+ "links": [
+ 732,
+ 808
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ImageBatchExtendWithOverlap",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "widgets_values": [
+ 13,
+ "new_images",
+ "cut"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 418,
+ "type": "GetNode",
+ "pos": [
+ 5063.620609727339,
+ -1950.8726913782168
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 32,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "MULTITALK_EMBEDS",
+ "type": "MULTITALK_EMBEDS",
+ "links": [
+ 783
+ ]
+ }
+ ],
+ "title": "Get_audio_embeds",
+ "properties": {},
+ "widgets_values": [
+ "audio_embeds"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 344,
+ "type": "GetImageSizeAndCount",
+ "pos": [
+ 4786.60313188824,
+ -2373.1692877632895
+ ],
+ "size": [
+ 197.07500305175782,
+ 86
+ ],
+ "flags": {},
+ "order": 82,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 650
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 809
+ ]
+ },
+ {
+ "label": "832 width",
+ "name": "width",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "label": "480 height",
+ "name": "height",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "label": "93 count",
+ "name": "count",
+ "type": "INT",
+ "links": [
+ 727
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "GetImageSizeAndCount",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "widgets_values": [],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 357,
+ "type": "GetImageRangeFromBatch",
+ "pos": [
+ 5692.4678020173515,
+ -2210.5884604350235
+ ],
+ "size": [
+ 349.7056640625,
+ 102
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 85,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 811
+ },
+ {
+ "name": "masks",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 800
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 657
+ ]
+ },
+ {
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "GetImageRangeFromBatch",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "widgets_values": [
+ -1,
+ 13
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 329,
+ "type": "WanVideoDecode",
+ "pos": [
+ 6031.904860420774,
+ -2234.0664709345306
+ ],
+ "size": [
+ 270,
+ 198
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 90,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 597
+ },
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "link": 661
+ }
+ ],
+ "outputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "links": [
+ 688
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoDecode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ "default"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 416,
+ "type": "GetNode",
+ "pos": [
+ 6312.57992402471,
+ -2051.2548628112504
+ ],
+ "size": [
+ 210,
+ 58
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 33,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 781
+ ]
+ }
+ ],
+ "title": "Get_input_audio",
+ "properties": {},
+ "widgets_values": [
+ "input_audio"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 445,
+ "type": "GetNode",
+ "pos": [
+ 7677.687298483467,
+ -1775.1907567982662
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 34,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 837
+ ]
+ }
+ ],
+ "title": "Get_overlap",
+ "properties": {},
+ "widgets_values": [
+ "overlap"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 446,
+ "type": "WanVideoEncode",
+ "pos": [
+ 8821.896356006253,
+ -2096.6949772712733
+ ],
+ "size": [
+ 273.705078125,
+ 242
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 99,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 818
+ },
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 819
+ },
+ {
+ "name": "mask",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 821
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoEncode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ 0,
+ 1
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 447,
+ "type": "GetNode",
+ "pos": [
+ 8639.73391763733,
+ -2148.859900966186
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 35,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVAE",
+ "type": "WANVAE",
+ "links": [
+ 818,
+ 841
+ ]
+ }
+ ],
+ "title": "Get_VAE",
+ "properties": {},
+ "widgets_values": [
+ "VAE"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 448,
+ "type": "ReplaceVideoLatentFrames",
+ "pos": [
+ 8815.605863566758,
+ -2161.7645650963136
+ ],
+ "size": [
+ 280.355078125,
+ 78
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 100,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "destination",
+ "type": "LATENT",
+ "link": 820
+ },
+ {
+ "name": "source",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 821
+ }
+ ],
+ "outputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 842
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ReplaceVideoLatentFrames",
+ "cnr_id": "comfy-core",
+ "ver": "0.5.0"
+ },
+ "widgets_values": [
+ 0
+ ]
+ },
+ {
+ "id": 450,
+ "type": "GetNode",
+ "pos": [
+ 8289.185556956885,
+ -1759.6531120061982
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 36,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "FLOAT",
+ "type": "FLOAT",
+ "links": [
+ 829
+ ]
+ }
+ ],
+ "title": "Get_cfg",
+ "properties": {},
+ "widgets_values": [
+ "cfg"
+ ],
+ "color": "#232",
+ "bgcolor": "#353"
+ },
+ {
+ "id": 451,
+ "type": "GetNode",
+ "pos": [
+ 8264.85978470551,
+ -1901.6077550764896
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 37,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOSCHEDULER",
+ "type": "WANVIDEOSCHEDULER",
+ "links": [
+ 827
+ ]
+ }
+ ],
+ "title": "Get_scheduler",
+ "properties": {},
+ "widgets_values": [
+ "scheduler"
+ ]
+ },
+ {
+ "id": 452,
+ "type": "GetNode",
+ "pos": [
+ 8265.543460113053,
+ -1953.8807028744789
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 38,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOMODEL",
+ "type": "WANVIDEOMODEL",
+ "links": [
+ 825
+ ]
+ }
+ ],
+ "title": "Get_wanmodel",
+ "properties": {},
+ "widgets_values": [
+ "wanmodel"
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 454,
+ "type": "GetNode",
+ "pos": [
+ 8472.505652160606,
+ -2270.566769075312
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 39,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 833,
+ 840
+ ]
+ }
+ ],
+ "title": "Get_overlap",
+ "properties": {},
+ "widgets_values": [
+ "overlap"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 455,
+ "type": "Reroute",
+ "pos": [
+ 8307.377090205251,
+ -2366.279946434597
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 95,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 824
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "IMAGE",
+ "links": [
+ 831,
+ 839
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 457,
+ "type": "GetNode",
+ "pos": [
+ 7680.6005330636835,
+ -1826.8331734558958
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 40,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 836
+ ]
+ }
+ ],
+ "title": "Get_frames",
+ "properties": {},
+ "widgets_values": [
+ "frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 458,
+ "type": "Reroute",
+ "pos": [
+ 10267.183588819598,
+ -2320.9337486082222
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 104,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 830
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "IMAGE",
+ "links": null
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 459,
+ "type": "GetNode",
+ "pos": [
+ 7859.880789998097,
+ -1675.4559507379802
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 41,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "LATENT",
+ "type": "LATENT",
+ "links": [
+ 835
+ ]
+ }
+ ],
+ "title": "Get_ref_latent",
+ "properties": {},
+ "widgets_values": [
+ "ref_latent"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 460,
+ "type": "ImageBatchExtendWithOverlap",
+ "pos": [
+ 9073.50502999508,
+ -2360.926321160299
+ ],
+ "size": [
+ 321.6236328125,
+ 146
+ ],
+ "flags": {},
+ "order": 102,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "source_images",
+ "type": "IMAGE",
+ "link": 831
+ },
+ {
+ "name": "new_images",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 832
+ },
+ {
+ "name": "overlap",
+ "type": "INT",
+ "widget": {
+ "name": "overlap"
+ },
+ "link": 833
+ }
+ ],
+ "outputs": [
+ {
+ "name": "source_images",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "start_images",
+ "type": "IMAGE",
+ "links": null
+ },
+ {
+ "name": "extended_images",
+ "type": "IMAGE",
+ "links": [
+ 822,
+ 830
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "ImageBatchExtendWithOverlap",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "widgets_values": [
+ 13,
+ "new_images",
+ "cut"
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 461,
+ "type": "WanVideoLongCatAvatarExtendEmbeds",
+ "pos": [
+ 7854.07317149529,
+ -1892.6690512715838
+ ],
+ "size": [
+ 379.476171875,
+ 170
+ ],
+ "flags": {},
+ "order": 96,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "prev_latents",
+ "type": "LATENT",
+ "link": 845
+ },
+ {
+ "name": "audio_embeds",
+ "type": "MULTITALK_EMBEDS",
+ "link": 834
+ },
+ {
+ "name": "ref_latent",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 835
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 836
+ },
+ {
+ "name": "overlap",
+ "type": "INT",
+ "widget": {
+ "name": "overlap"
+ },
+ "link": 837
+ },
+ {
+ "name": "frames_processed",
+ "type": "INT",
+ "widget": {
+ "name": "frames_processed"
+ },
+ "link": 838
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "links": [
+ 826
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoLongCatAvatarExtendEmbeds",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 93,
+ 13,
+ 93,
+ "pad_with_start"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 462,
+ "type": "GetNode",
+ "pos": [
+ 7852.090413215118,
+ -1940.749920719971
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 42,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "MULTITALK_EMBEDS",
+ "type": "MULTITALK_EMBEDS",
+ "links": [
+ 834
+ ]
+ }
+ ],
+ "title": "Get_audio_embeds",
+ "properties": {},
+ "widgets_values": [
+ "audio_embeds"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 463,
+ "type": "GetImageSizeAndCount",
+ "pos": [
+ 7575.072935376019,
+ -2363.046517105044
+ ],
+ "size": [
+ 197.07500305175782,
+ 86
+ ],
+ "flags": {},
+ "order": 94,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "link": 843
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image",
+ "type": "IMAGE",
+ "links": [
+ 824
+ ]
+ },
+ {
+ "label": "832 width",
+ "name": "width",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "label": "480 height",
+ "name": "height",
+ "type": "INT",
+ "links": null
+ },
+ {
+ "label": "173 count",
+ "name": "count",
+ "type": "INT",
+ "links": [
+ 838
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "GetImageSizeAndCount",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 464,
+ "type": "GetImageRangeFromBatch",
+ "pos": [
+ 8480.937605505142,
+ -2200.465689776778
+ ],
+ "size": [
+ 349.7056640625,
+ 102
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 97,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "shape": 7,
+ "type": "IMAGE",
+ "link": 839
+ },
+ {
+ "name": "masks",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 840
+ }
+ ],
+ "outputs": [
+ {
+ "name": "IMAGE",
+ "type": "IMAGE",
+ "links": [
+ 819
+ ]
+ },
+ {
+ "name": "MASK",
+ "type": "MASK",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "GetImageRangeFromBatch",
+ "cnr_id": "comfyui-kjnodes",
+ "ver": "16cbf238a74cac17082d6888bc8934899c850645"
+ },
+ "widgets_values": [
+ -1,
+ 13
+ ],
+ "color": "#2a363b",
+ "bgcolor": "#3f5159"
+ },
+ {
+ "id": 465,
+ "type": "WanVideoDecode",
+ "pos": [
+ 8820.374663908564,
+ -2223.943700276285
+ ],
+ "size": [
+ 270,
+ 198
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 101,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "vae",
+ "type": "WANVAE",
+ "link": 841
+ },
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "link": 842
+ }
+ ],
+ "outputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "links": [
+ 832
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoDecode",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "ae6fe0853e2d1ba0f1f47e086befdb089dc07490"
+ },
+ "widgets_values": [
+ false,
+ 272,
+ 272,
+ 144,
+ 128,
+ "default"
+ ],
+ "color": "#322",
+ "bgcolor": "#533"
+ },
+ {
+ "id": 466,
+ "type": "GetNode",
+ "pos": [
+ 9101.0497275125,
+ -2041.1320921530048
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 43,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 823
+ ]
+ }
+ ],
+ "title": "Get_input_audio",
+ "properties": {},
+ "widgets_values": [
+ "input_audio"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 436,
+ "type": "Reroute",
+ "pos": [
+ 7478.713785331808,
+ -2331.0565192664676
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 93,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 808
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "IMAGE",
+ "links": [
+ 843
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 467,
+ "type": "Reroute",
+ "pos": [
+ 7484.396984040454,
+ -1930.5275855179693
+ ],
+ "size": [
+ 75,
+ 26
+ ],
+ "flags": {},
+ "order": 88,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "",
+ "type": "*",
+ "link": 844
+ }
+ ],
+ "outputs": [
+ {
+ "name": "",
+ "type": "LATENT",
+ "links": [
+ 845
+ ]
+ }
+ ],
+ "properties": {
+ "showOutputText": false,
+ "horizontal": false
+ }
+ },
+ {
+ "id": 300,
+ "type": "Wav2VecModelLoader",
+ "pos": [
+ 430.49593921985445,
+ -761.4912158297939
+ ],
+ "size": [
+ 397.12591274100396,
+ 106
+ ],
+ "flags": {},
+ "order": 44,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "wav2vec_model",
+ "type": "WAV2VECMODEL",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "Wav2VecModelLoader",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "6fce0e2d3bb976b0006bc6d8e37e1f23460938ee"
+ },
+ "widgets_values": [
+ "wav2vec2-chinese-base_fp16.safetensors",
+ "fp16",
+ "main_device"
+ ]
+ },
+ {
+ "id": 137,
+ "type": "DownloadAndLoadWav2VecModel",
+ "pos": [
+ 448.3987129033839,
+ -929.4363680488067
+ ],
+ "size": [
+ 330.96728515625,
+ 106
+ ],
+ "flags": {},
+ "order": 45,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "wav2vec_model",
+ "type": "WAV2VECMODEL",
+ "links": [
+ 334
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "DownloadAndLoadWav2VecModel",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0"
+ },
+ "widgets_values": [
+ "TencentGameMate/chinese-wav2vec2-base",
+ "fp16",
+ "main_device"
+ ]
+ },
+ {
+ "id": 303,
+ "type": "MarkdownNote",
+ "pos": [
+ 856.889396897066,
+ -785.3839495993304
+ ],
+ "size": [
+ 355.3899841308594,
+ 125.51774597167969
+ ],
+ "flags": {},
+ "order": 46,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "title": "Wav2vec2 safetensors",
+ "properties": {},
+ "widgets_values": [
+ "Alternative to the download node is to use single .safetensors file from:\n\n[https://huggingface.co/Kijai/wav2vec2_safetensors/tree/main](https://huggingface.co/Kijai/wav2vec2_safetensors/tree/main)\n\nThe model goes to `ComfyUI/models/wav2vec2`"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 301,
+ "type": "MelBandRoFormerModelLoader",
+ "pos": [
+ 444.8198405539688,
+ -1074.2418481327345
+ ],
+ "size": [
+ 400.5706481933594,
+ 58
+ ],
+ "flags": {},
+ "order": 47,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "model",
+ "type": "MELROFORMERMODEL",
+ "links": [
+ 543
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "MelBandRoFormerModelLoader",
+ "cnr_id": "ComfyUI-MelBandRoFormer",
+ "ver": "b68d9077815387b64d596f8c39607052b95b6eba"
+ },
+ "widgets_values": [
+ "MelBandRoformer_fp32.safetensors"
+ ]
+ },
+ {
+ "id": 304,
+ "type": "MarkdownNote",
+ "pos": [
+ 447.9326709055299,
+ -1227.2145082672455
+ ],
+ "size": [
+ 345.2374267578125,
+ 91.6759033203125
+ ],
+ "flags": {},
+ "order": 48,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "title": "Wav2vec2 safetensors",
+ "properties": {},
+ "widgets_values": [
+ "Vocal separator model\n\n[https://huggingface.co/Kijai/MelBandRoFormer_comfy/tree/main](https://huggingface.co/Kijai/MelBandRoFormer_comfy/tree/main)\n\nThe model goes to `ComfyUI/models/diffusion_models`"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 253,
+ "type": "SetNode",
+ "pos": [
+ 1025.2719859135082,
+ -1115.8318099244855
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 70,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "link": 588
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": []
+ }
+ ],
+ "title": "Set_input_audio",
+ "properties": {
+ "previousName": "input_audio"
+ },
+ "widgets_values": [
+ "input_audio"
+ ]
+ },
+ {
+ "id": 263,
+ "type": "Note",
+ "pos": [
+ 1650.7933620718532,
+ -932.9952147664453
+ ],
+ "size": [
+ 420.0259563580471,
+ 212.22914465030635
+ ],
+ "flags": {},
+ "order": 49,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [],
+ "properties": {},
+ "widgets_values": [
+ "increase audio_scale for stronger effect\n\naudio_cfg adds extra model pass for improved/stronger lipsync, default range is 3-5\n\nnum_frames is the maximum frame count to generate, if the value is higher than your audio length would be, the audio length is used\n\nThe model uses audio stride 2, meaning output fps is 16 while input audio fps is 32"
+ ],
+ "color": "#432",
+ "bgcolor": "#653"
+ },
+ {
+ "id": 423,
+ "type": "INTConstant",
+ "pos": [
+ 2651.6255629540756,
+ -2543.5461851956397
+ ],
+ "size": [
+ 200,
+ 58
+ ],
+ "flags": {},
+ "order": 50,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "value",
+ "type": "INT",
+ "links": [
+ 794
+ ]
+ }
+ ],
+ "title": "Overlap",
+ "properties": {
+ "Node name for S&R": "INTConstant"
+ },
+ "widgets_values": [
+ 13
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 294,
+ "type": "SetNode",
+ "pos": [
+ 1659.8309623476516,
+ -1155.4069912825444
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 76,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "link": 519
+ }
+ ],
+ "outputs": [
+ {
+ "name": "INT",
+ "type": "INT",
+ "links": [
+ 520
+ ]
+ }
+ ],
+ "title": "Set_actual_audio_frames",
+ "properties": {
+ "previousName": "actual_audio_frames"
+ },
+ "widgets_values": [
+ "actual_audio_frames"
+ ],
+ "color": "#1b4669",
+ "bgcolor": "#29699c"
+ },
+ {
+ "id": 417,
+ "type": "SetNode",
+ "pos": [
+ 1905.8380700989962,
+ -1201.1160366370482
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 75,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "MULTITALK_EMBEDS",
+ "type": "MULTITALK_EMBEDS",
+ "link": 782
+ }
+ ],
+ "outputs": [
+ {
+ "name": "*",
+ "type": "*",
+ "links": null
+ }
+ ],
+ "title": "Set_audio_embeds",
+ "properties": {
+ "previousName": "audio_embeds"
+ },
+ "widgets_values": [
+ "audio_embeds"
+ ]
+ },
+ {
+ "id": 327,
+ "type": "WanVideoSamplerv2",
+ "pos": [
+ 5639.164088586091,
+ -1924.6922101794962
+ ],
+ "size": [
+ 526.7895498026237,
+ 560.3785864245906
+ ],
+ "flags": {},
+ "order": 86,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "model",
+ "type": "WANVIDEOMODEL",
+ "link": 600
+ },
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "link": 635
+ },
+ {
+ "name": "scheduler",
+ "type": "WANVIDEOSCHEDULER",
+ "link": 599
+ },
+ {
+ "name": "text_embeds",
+ "shape": 7,
+ "type": "WANVIDEOTEXTEMBEDS",
+ "link": 846
+ },
+ {
+ "name": "samples",
+ "shape": 7,
+ "type": "LATENT",
+ "link": null
+ },
+ {
+ "name": "extra_args",
+ "shape": 7,
+ "type": "WANVIDSAMPLEREXTRAARGS",
+ "link": null
+ },
+ {
+ "name": "cfg",
+ "type": "FLOAT",
+ "widget": {
+ "name": "cfg"
+ },
+ "link": 798
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 660,
+ 844
+ ]
+ },
+ {
+ "name": "denoised_samples",
+ "type": "LATENT",
+ "links": []
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoSamplerv2",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 1,
+ 2,
+ "fixed",
+ true,
+ false
+ ]
+ },
+ {
+ "id": 346,
+ "type": "WanVideoLongCatAvatarExtendEmbeds",
+ "pos": [
+ 5065.603368007511,
+ -1902.7918219298297
+ ],
+ "size": [
+ 379.476171875,
+ 170
+ ],
+ "flags": {},
+ "order": 84,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "prev_latents",
+ "type": "LATENT",
+ "link": 806
+ },
+ {
+ "name": "audio_embeds",
+ "type": "MULTITALK_EMBEDS",
+ "link": 783
+ },
+ {
+ "name": "ref_latent",
+ "shape": 7,
+ "type": "LATENT",
+ "link": 775
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 814
+ },
+ {
+ "name": "overlap",
+ "type": "INT",
+ "widget": {
+ "name": "overlap"
+ },
+ "link": 799
+ },
+ {
+ "name": "frames_processed",
+ "type": "INT",
+ "widget": {
+ "name": "frames_processed"
+ },
+ "link": 727
+ }
+ ],
+ "outputs": [
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "links": [
+ 635
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoLongCatAvatarExtendEmbeds",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 93,
+ 13,
+ 93,
+ "pad_with_start"
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 468,
+ "type": "GetNode",
+ "pos": [
+ 5463.7083954517075,
+ -1827.6393258180478
+ ],
+ "size": [
+ 210,
+ 60
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 51,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOTEXTEMBEDS",
+ "type": "WANVIDEOTEXTEMBEDS",
+ "links": [
+ 846
+ ]
+ }
+ ],
+ "title": "Get_text_embeds",
+ "properties": {},
+ "widgets_values": [
+ "text_embeds"
+ ]
+ },
+ {
+ "id": 469,
+ "type": "GetNode",
+ "pos": [
+ 8244.977684331303,
+ -1820.0896151916613
+ ],
+ "size": [
+ 210,
+ 34
+ ],
+ "flags": {
+ "collapsed": true
+ },
+ "order": 52,
+ "mode": 0,
+ "inputs": [],
+ "outputs": [
+ {
+ "name": "WANVIDEOTEXTEMBEDS",
+ "type": "WANVIDEOTEXTEMBEDS",
+ "links": [
+ 847
+ ]
+ }
+ ],
+ "title": "Get_text_embeds",
+ "properties": {},
+ "widgets_values": [
+ "text_embeds"
+ ]
+ },
+ {
+ "id": 456,
+ "type": "WanVideoSamplerv2",
+ "pos": [
+ 8427.63389207388,
+ -1914.5694395212504
+ ],
+ "size": [
+ 526.7895498026237,
+ 560.3785864245906
+ ],
+ "flags": {},
+ "order": 98,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "model",
+ "type": "WANVIDEOMODEL",
+ "link": 825
+ },
+ {
+ "name": "image_embeds",
+ "type": "WANVIDIMAGE_EMBEDS",
+ "link": 826
+ },
+ {
+ "name": "scheduler",
+ "type": "WANVIDEOSCHEDULER",
+ "link": 827
+ },
+ {
+ "name": "text_embeds",
+ "shape": 7,
+ "type": "WANVIDEOTEXTEMBEDS",
+ "link": 847
+ },
+ {
+ "name": "samples",
+ "shape": 7,
+ "type": "LATENT",
+ "link": null
+ },
+ {
+ "name": "extra_args",
+ "shape": 7,
+ "type": "WANVIDSAMPLEREXTRAARGS",
+ "link": null
+ },
+ {
+ "name": "cfg",
+ "type": "FLOAT",
+ "widget": {
+ "name": "cfg"
+ },
+ "link": 829
+ }
+ ],
+ "outputs": [
+ {
+ "name": "samples",
+ "type": "LATENT",
+ "links": [
+ 820
+ ]
+ },
+ {
+ "name": "denoised_samples",
+ "type": "LATENT",
+ "links": []
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoSamplerv2",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ 1,
+ 1,
+ "fixed",
+ true,
+ false
+ ]
+ },
+ {
+ "id": 325,
+ "type": "WanVideoSchedulerv2",
+ "pos": [
+ 3117.3259066097007,
+ -2340.4385015751564
+ ],
+ "size": [
+ 273.2701598165304,
+ 353.33726502245645
+ ],
+ "flags": {},
+ "order": 53,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "sigmas",
+ "shape": 7,
+ "type": "SIGMAS",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "scheduler",
+ "type": "WANVIDEOSCHEDULER",
+ "links": [
+ 584,
+ 598
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoSchedulerv2",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd"
+ },
+ "widgets_values": [
+ "euler",
+ 10,
+ 12,
+ 0,
+ -1,
+ false,
+ "
"
+ ]
+ },
+ {
+ "id": 138,
+ "type": "WanVideoLoraSelect",
+ "pos": [
+ 564.2077204363451,
+ -2675.4893783617654
+ ],
+ "size": [
+ 503.4073486328125,
+ 200
+ ],
+ "flags": {},
+ "order": 54,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "prev_lora",
+ "shape": 7,
+ "type": "WANVIDLORA",
+ "link": null
+ },
+ {
+ "name": "blocks",
+ "shape": 7,
+ "type": "SELECTEDBLOCKS",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "lora",
+ "type": "WANVIDLORA",
+ "links": [
+ 848
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "WanVideoLoraSelect",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0"
+ },
+ "widgets_values": [
+ "LongCat_distill_lora_rank128_bf16.safetensors",
+ 1,
+ false,
+ false,
+ "Metadata
| Metadata |
| model_type | LongCat_distill_lora |
| format | pt |
"
+ ],
+ "color": "#223",
+ "bgcolor": "#335"
+ },
+ {
+ "id": 194,
+ "type": "MultiTalkWav2VecEmbeds",
+ "pos": [
+ 1324.7750244140625,
+ -1235.9822998046875
+ ],
+ "size": [
+ 291.08203125,
+ 326
+ ],
+ "flags": {},
+ "order": 73,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "wav2vec_model",
+ "type": "WAV2VECMODEL",
+ "link": 334
+ },
+ {
+ "name": "audio_1",
+ "type": "AUDIO",
+ "link": 564
+ },
+ {
+ "name": "audio_2",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": null
+ },
+ {
+ "name": "audio_3",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": null
+ },
+ {
+ "name": "audio_4",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": null
+ },
+ {
+ "name": "ref_target_masks",
+ "shape": 7,
+ "type": "MASK",
+ "link": null
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "widget": {
+ "name": "num_frames"
+ },
+ "link": 529
+ }
+ ],
+ "outputs": [
+ {
+ "name": "multitalk_embeds",
+ "type": "MULTITALK_EMBEDS",
+ "links": [
+ 628,
+ 782
+ ]
+ },
+ {
+ "name": "audio",
+ "type": "AUDIO",
+ "links": []
+ },
+ {
+ "name": "num_frames",
+ "type": "INT",
+ "links": [
+ 519
+ ]
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "MultiTalkWav2VecEmbeds",
+ "cnr_id": "ComfyUI-WanVideoWrapper",
+ "ver": "3d7801cee4c8e3106078dd9b9f146caee95069ba"
+ },
+ "widgets_values": [
+ true,
+ 400,
+ 32,
+ 1,
+ 2,
+ "para",
+ true,
+ true
+ ],
+ "color": "#323",
+ "bgcolor": "#535"
+ },
+ {
+ "id": 453,
+ "type": "VHS_VideoCombine",
+ "pos": [
+ 9094.011012739598,
+ -1979.025295960334
+ ],
+ "size": [
+ 991.5499877929688,
+ 908.5096083420973
+ ],
+ "flags": {},
+ "order": 103,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 822
+ },
+ {
+ "name": "audio",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": 823
+ },
+ {
+ "name": "meta_batch",
+ "shape": 7,
+ "type": "VHS_BatchManager",
+ "link": null
+ },
+ {
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "Filenames",
+ "type": "VHS_FILENAMES",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "VHS_VideoCombine",
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8"
+ },
+ "widgets_values": {
+ "frame_rate": 16,
+ "loop_count": 0,
+ "filename_prefix": "LongCat-Avatar",
+ "format": "video/h264-mp4",
+ "pix_fmt": "yuv420p",
+ "crf": 19,
+ "save_metadata": true,
+ "trim_to_audio": false,
+ "pingpong": false,
+ "save_output": false,
+ "videopreview": {
+ "hidden": false,
+ "paused": false,
+ "params": {
+ "filename": "WanVideo2_1_InfiniteTalk_00009-audio.mp4",
+ "subfolder": "",
+ "type": "temp",
+ "format": "video/h264-mp4",
+ "frame_rate": 16,
+ "workflow": "WanVideo2_1_InfiniteTalk_00009.png",
+ "fullpath": "/home/kijai/AI/ComfyUI/temp/WanVideo2_1_InfiniteTalk_00009-audio.mp4"
+ }
+ }
+ }
+ },
+ {
+ "id": 386,
+ "type": "VHS_VideoCombine",
+ "pos": [
+ 6305.541209251808,
+ -1989.1480666185798
+ ],
+ "size": [
+ 991.5499877929688,
+ 908.5096083420973
+ ],
+ "flags": {},
+ "order": 92,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 732
+ },
+ {
+ "name": "audio",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": 781
+ },
+ {
+ "name": "meta_batch",
+ "shape": 7,
+ "type": "VHS_BatchManager",
+ "link": null
+ },
+ {
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "Filenames",
+ "type": "VHS_FILENAMES",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "VHS_VideoCombine",
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8"
+ },
+ "widgets_values": {
+ "frame_rate": 16,
+ "loop_count": 0,
+ "filename_prefix": "LongCat-Avatar",
+ "format": "video/h264-mp4",
+ "pix_fmt": "yuv420p",
+ "crf": 19,
+ "save_metadata": true,
+ "trim_to_audio": false,
+ "pingpong": false,
+ "save_output": false,
+ "videopreview": {
+ "hidden": false,
+ "paused": false,
+ "params": {
+ "filename": "WanVideo2_1_InfiniteTalk_00008-audio.mp4",
+ "subfolder": "",
+ "type": "temp",
+ "format": "video/h264-mp4",
+ "frame_rate": 16,
+ "workflow": "WanVideo2_1_InfiniteTalk_00008.png",
+ "fullpath": "/home/kijai/AI/ComfyUI/temp/WanVideo2_1_InfiniteTalk_00008-audio.mp4"
+ }
+ }
+ }
+ },
+ {
+ "id": 320,
+ "type": "VHS_VideoCombine",
+ "pos": [
+ 3709.5320822568765,
+ -1344.6071062656185
+ ],
+ "size": [
+ 991.5499877929688,
+ 908.5096083420973
+ ],
+ "flags": {},
+ "order": 81,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "images",
+ "type": "IMAGE",
+ "link": 574
+ },
+ {
+ "name": "audio",
+ "shape": 7,
+ "type": "AUDIO",
+ "link": 587
+ },
+ {
+ "name": "meta_batch",
+ "shape": 7,
+ "type": "VHS_BatchManager",
+ "link": null
+ },
+ {
+ "name": "vae",
+ "shape": 7,
+ "type": "VAE",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "Filenames",
+ "type": "VHS_FILENAMES",
+ "links": null
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "VHS_VideoCombine",
+ "cnr_id": "comfyui-videohelpersuite",
+ "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8"
+ },
+ "widgets_values": {
+ "frame_rate": 16,
+ "loop_count": 0,
+ "filename_prefix": "LongCat-Avatar",
+ "format": "video/h264-mp4",
+ "pix_fmt": "yuv420p",
+ "crf": 19,
+ "save_metadata": true,
+ "trim_to_audio": false,
+ "pingpong": false,
+ "save_output": false,
+ "videopreview": {
+ "hidden": false,
+ "paused": false,
+ "params": {
+ "filename": "WanVideo2_1_InfiniteTalk_00007-audio.mp4",
+ "subfolder": "",
+ "type": "temp",
+ "format": "video/h264-mp4",
+ "frame_rate": 16,
+ "workflow": "WanVideo2_1_InfiniteTalk_00007.png",
+ "fullpath": "/home/kijai/AI/ComfyUI/temp/WanVideo2_1_InfiniteTalk_00007-audio.mp4"
+ }
+ }
+ }
+ }
+ ],
+ "links": [
+ [
+ 334,
+ 137,
+ 0,
+ 194,
+ 0,
+ "WAV2VECMODEL"
+ ],
+ [
+ 362,
+ 134,
+ 0,
+ 122,
+ 1,
+ "BLOCKSWAPARGS"
+ ],
+ [
+ 436,
+ 129,
+ 0,
+ 240,
+ 0,
+ "*"
+ ],
+ [
+ 441,
+ 245,
+ 0,
+ 247,
+ 0,
+ "*"
+ ],
+ [
+ 442,
+ 246,
+ 0,
+ 248,
+ 0,
+ "*"
+ ],
+ [
+ 463,
+ 122,
+ 0,
+ 260,
+ 0,
+ "*"
+ ],
+ [
+ 466,
+ 238,
+ 0,
+ 264,
+ 0,
+ "*"
+ ],
+ [
+ 472,
+ 270,
+ 0,
+ 271,
+ 0,
+ "*"
+ ],
+ [
+ 494,
+ 283,
+ 0,
+ 281,
+ 2,
+ "INT"
+ ],
+ [
+ 495,
+ 282,
+ 0,
+ 281,
+ 3,
+ "INT"
+ ],
+ [
+ 496,
+ 284,
+ 0,
+ 281,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 519,
+ 194,
+ 2,
+ 294,
+ 0,
+ "*"
+ ],
+ [
+ 520,
+ 294,
+ 0,
+ 293,
+ 0,
+ "*"
+ ],
+ [
+ 529,
+ 272,
+ 0,
+ 194,
+ 6,
+ "INT"
+ ],
+ [
+ 543,
+ 301,
+ 0,
+ 302,
+ 0,
+ "MELROFORMERMODEL"
+ ],
+ [
+ 554,
+ 281,
+ 0,
+ 312,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 564,
+ 302,
+ 0,
+ 194,
+ 1,
+ "AUDIO"
+ ],
+ [
+ 569,
+ 125,
+ 0,
+ 317,
+ 0,
+ "AUDIO"
+ ],
+ [
+ 570,
+ 317,
+ 0,
+ 302,
+ 1,
+ "AUDIO"
+ ],
+ [
+ 574,
+ 313,
+ 0,
+ 320,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 583,
+ 261,
+ 0,
+ 324,
+ 0,
+ "WANVIDEOMODEL"
+ ],
+ [
+ 584,
+ 325,
+ 0,
+ 324,
+ 2,
+ "WANVIDEOSCHEDULER"
+ ],
+ [
+ 586,
+ 241,
+ 0,
+ 324,
+ 3,
+ "WANVIDEOTEXTEMBEDS"
+ ],
+ [
+ 587,
+ 254,
+ 0,
+ 320,
+ 1,
+ "AUDIO"
+ ],
+ [
+ 588,
+ 317,
+ 0,
+ 253,
+ 0,
+ "AUDIO"
+ ],
+ [
+ 596,
+ 330,
+ 0,
+ 313,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 597,
+ 331,
+ 0,
+ 329,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 598,
+ 325,
+ 0,
+ 332,
+ 0,
+ "WANVIDEOSCHEDULER"
+ ],
+ [
+ 599,
+ 333,
+ 0,
+ 327,
+ 2,
+ "WANVIDEOSCHEDULER"
+ ],
+ [
+ 600,
+ 334,
+ 0,
+ 327,
+ 0,
+ "WANVIDEOMODEL"
+ ],
+ [
+ 628,
+ 194,
+ 0,
+ 345,
+ 1,
+ "MULTITALK_EMBEDS"
+ ],
+ [
+ 629,
+ 345,
+ 0,
+ 324,
+ 1,
+ "WANVIDIMAGE_EMBEDS"
+ ],
+ [
+ 635,
+ 346,
+ 0,
+ 327,
+ 1,
+ "WANVIDIMAGE_EMBEDS"
+ ],
+ [
+ 650,
+ 313,
+ 0,
+ 344,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 657,
+ 357,
+ 0,
+ 358,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 658,
+ 331,
+ 0,
+ 358,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 660,
+ 327,
+ 0,
+ 359,
+ 0,
+ "LATENT"
+ ],
+ [
+ 661,
+ 359,
+ 0,
+ 329,
+ 1,
+ "LATENT"
+ ],
+ [
+ 662,
+ 358,
+ 0,
+ 359,
+ 1,
+ "LATENT"
+ ],
+ [
+ 688,
+ 329,
+ 0,
+ 341,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 727,
+ 344,
+ 3,
+ 346,
+ 5,
+ "INT"
+ ],
+ [
+ 732,
+ 341,
+ 2,
+ 386,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 770,
+ 177,
+ 0,
+ 122,
+ 0,
+ "WANCOMPILEARGS"
+ ],
+ [
+ 775,
+ 413,
+ 0,
+ 346,
+ 2,
+ "LATENT"
+ ],
+ [
+ 776,
+ 312,
+ 0,
+ 411,
+ 0,
+ "LATENT"
+ ],
+ [
+ 781,
+ 416,
+ 0,
+ 386,
+ 1,
+ "AUDIO"
+ ],
+ [
+ 782,
+ 194,
+ 0,
+ 417,
+ 0,
+ "MULTITALK_EMBEDS"
+ ],
+ [
+ 783,
+ 418,
+ 0,
+ 346,
+ 1,
+ "MULTITALK_EMBEDS"
+ ],
+ [
+ 786,
+ 241,
+ 0,
+ 421,
+ 0,
+ "WANVIDEOTEXTEMBEDS"
+ ],
+ [
+ 794,
+ 423,
+ 0,
+ 425,
+ 0,
+ "INT"
+ ],
+ [
+ 796,
+ 427,
+ 0,
+ 428,
+ 0,
+ "FLOAT"
+ ],
+ [
+ 797,
+ 429,
+ 0,
+ 324,
+ 6,
+ "FLOAT"
+ ],
+ [
+ 798,
+ 430,
+ 0,
+ 327,
+ 6,
+ "FLOAT"
+ ],
+ [
+ 799,
+ 431,
+ 0,
+ 346,
+ 4,
+ "INT"
+ ],
+ [
+ 800,
+ 432,
+ 0,
+ 357,
+ 2,
+ "INT"
+ ],
+ [
+ 802,
+ 432,
+ 0,
+ 341,
+ 2,
+ "INT"
+ ],
+ [
+ 804,
+ 324,
+ 0,
+ 434,
+ 0,
+ "LATENT"
+ ],
+ [
+ 805,
+ 434,
+ 0,
+ 313,
+ 1,
+ "LATENT"
+ ],
+ [
+ 806,
+ 434,
+ 0,
+ 346,
+ 0,
+ "LATENT"
+ ],
+ [
+ 808,
+ 341,
+ 2,
+ 436,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 809,
+ 344,
+ 0,
+ 437,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 810,
+ 437,
+ 0,
+ 341,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 811,
+ 437,
+ 0,
+ 357,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 813,
+ 438,
+ 0,
+ 439,
+ 0,
+ "INT"
+ ],
+ [
+ 814,
+ 440,
+ 0,
+ 346,
+ 3,
+ "INT"
+ ],
+ [
+ 815,
+ 442,
+ 0,
+ 345,
+ 3,
+ "INT"
+ ],
+ [
+ 816,
+ 443,
+ 0,
+ 345,
+ 0,
+ "LATENT"
+ ],
+ [
+ 817,
+ 444,
+ 0,
+ 312,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 818,
+ 447,
+ 0,
+ 446,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 819,
+ 464,
+ 0,
+ 446,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 820,
+ 456,
+ 0,
+ 448,
+ 0,
+ "LATENT"
+ ],
+ [
+ 821,
+ 446,
+ 0,
+ 448,
+ 1,
+ "LATENT"
+ ],
+ [
+ 822,
+ 460,
+ 2,
+ 453,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 823,
+ 466,
+ 0,
+ 453,
+ 1,
+ "AUDIO"
+ ],
+ [
+ 824,
+ 463,
+ 0,
+ 455,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 825,
+ 452,
+ 0,
+ 456,
+ 0,
+ "WANVIDEOMODEL"
+ ],
+ [
+ 826,
+ 461,
+ 0,
+ 456,
+ 1,
+ "WANVIDIMAGE_EMBEDS"
+ ],
+ [
+ 827,
+ 451,
+ 0,
+ 456,
+ 2,
+ "WANVIDEOSCHEDULER"
+ ],
+ [
+ 829,
+ 450,
+ 0,
+ 456,
+ 6,
+ "FLOAT"
+ ],
+ [
+ 830,
+ 460,
+ 2,
+ 458,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 831,
+ 455,
+ 0,
+ 460,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 832,
+ 465,
+ 0,
+ 460,
+ 1,
+ "IMAGE"
+ ],
+ [
+ 833,
+ 454,
+ 0,
+ 460,
+ 2,
+ "INT"
+ ],
+ [
+ 834,
+ 462,
+ 0,
+ 461,
+ 1,
+ "MULTITALK_EMBEDS"
+ ],
+ [
+ 835,
+ 459,
+ 0,
+ 461,
+ 2,
+ "LATENT"
+ ],
+ [
+ 836,
+ 457,
+ 0,
+ 461,
+ 3,
+ "INT"
+ ],
+ [
+ 837,
+ 445,
+ 0,
+ 461,
+ 4,
+ "INT"
+ ],
+ [
+ 838,
+ 463,
+ 3,
+ 461,
+ 5,
+ "INT"
+ ],
+ [
+ 839,
+ 455,
+ 0,
+ 464,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 840,
+ 454,
+ 0,
+ 464,
+ 2,
+ "INT"
+ ],
+ [
+ 841,
+ 447,
+ 0,
+ 465,
+ 0,
+ "WANVAE"
+ ],
+ [
+ 842,
+ 448,
+ 0,
+ 465,
+ 1,
+ "LATENT"
+ ],
+ [
+ 843,
+ 436,
+ 0,
+ 463,
+ 0,
+ "IMAGE"
+ ],
+ [
+ 844,
+ 327,
+ 0,
+ 467,
+ 0,
+ "LATENT"
+ ],
+ [
+ 845,
+ 467,
+ 0,
+ 461,
+ 0,
+ "LATENT"
+ ],
+ [
+ 846,
+ 468,
+ 0,
+ 327,
+ 3,
+ "WANVIDEOTEXTEMBEDS"
+ ],
+ [
+ 847,
+ 469,
+ 0,
+ 456,
+ 3,
+ "WANVIDEOTEXTEMBEDS"
+ ],
+ [
+ 848,
+ 138,
+ 0,
+ 122,
+ 2,
+ "WANVIDLORA"
+ ]
+ ],
+ "groups": [
+ {
+ "id": 1,
+ "title": "Models",
+ "bounding": [
+ 480.1058044433594,
+ -3147.42529296875,
+ 1615.13037109375,
+ 1070.3359375
+ ],
+ "color": "#88A",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 4,
+ "title": "Audio",
+ "bounding": [
+ 386.6732482910156,
+ -1527.863525390625,
+ 1839.7524532251455,
+ 970.1723037978243
+ ],
+ "color": "#a1309b",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 5,
+ "title": "Input image",
+ "bounding": [
+ 462.7972412109375,
+ -2044.41552734375,
+ 1652.696533203125,
+ 495.40948486328125
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 6,
+ "title": "Extend",
+ "bounding": [
+ 4863.6216889635625,
+ -2579.643204578784,
+ 2666.416285962696,
+ 1623.188417420909
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ },
+ {
+ "id": 7,
+ "title": "Extend",
+ "bounding": [
+ 7652.091492451342,
+ -2569.520433920539,
+ 2666.416285962696,
+ 1623.188417420909
+ ],
+ "color": "#3f789e",
+ "font_size": 24,
+ "flags": {}
+ }
+ ],
+ "config": {},
+ "extra": {
+ "ds": {
+ "scale": 0.6115909044842043,
+ "offset": [
+ -159.8550083364596,
+ 3258.921342530495
+ ]
+ },
+ "frontendVersion": "1.36.2",
+ "workflowRendererVersion": "LG",
+ "node_versions": {
+ "comfy-core": "0.5.1",
+ "ComfyUI-KJNodes": "16cbf238a74cac17082d6888bc8934899c850645",
+ "ComfyUI-MelBandRoFormer": "b68d9077815387b64d596f8c39607052b95b6eba",
+ "ComfyUI-WanVideoWrapper": "93f7af6dc8559e2f6815caa67fd0982e4d8940dd",
+ "comfyui-videohelpersuite": "537f3a02269e14ab4b0350d4189a52034e9a3b12"
+ },
+ "VHS_latentpreview": true,
+ "VHS_latentpreviewrate": 0,
+ "VHS_MetadataImage": true,
+ "VHS_KeepIntermediate": true
+ },
+ "version": 0.4
+}
\ No newline at end of file
diff --git a/LongCat/layers.py b/LongCat/layers.py
index df98c01..06b94b1 100644
--- a/LongCat/layers.py
+++ b/LongCat/layers.py
@@ -2,6 +2,10 @@ import torch.nn as nn
import torch.nn.functional as F
import torch
import math
+from einops import rearrange
+
+from ..wanvideo.modules.model import WanRMSNorm, attention
+from ..multitalk.multitalk import RotaryPositionalEmbedding1D, normalize_and_scale
class FeedForwardSwiGLU(nn.Module):
def __init__(
@@ -22,7 +26,7 @@ class FeedForwardSwiGLU(nn.Module):
def forward(self, x):
return self.w2(F.silu(self.w1(x)) * self.w3(x))
-
+
class TimestepEmbedder(nn.Module):
"""
Embeds scalar timesteps into vector representations.
@@ -62,4 +66,147 @@ class TimestepEmbedder(nn.Module):
if t_freq.dtype != dtype:
t_freq = t_freq.to(dtype)
t_emb = self.mlp(t_freq)
- return t_emb
\ No newline at end of file
+ return t_emb
+
+
+class SingleStreamAttention(nn.Module):
+ def __init__(
+ self,
+ dim: int,
+ encoder_hidden_states_dim: int,
+ num_heads: int,
+ qkv_bias: bool,
+ qk_norm: bool,
+ attn_drop: float = 0.0,
+ proj_drop: float = 0.0,
+ eps: float = 1e-6,
+ class_range: int = 24,
+ class_interval: int = 4,
+ attention_mode: str = "sdpa",
+ ) -> None:
+ super().__init__()
+ assert dim % num_heads == 0, "dim should be divisible by num_heads"
+ self.dim = dim
+ self.encoder_hidden_states_dim = encoder_hidden_states_dim
+ self.num_heads = num_heads
+ self.head_dim = dim // num_heads
+ self.scale = self.head_dim**-0.5
+
+ self.q_linear = nn.Linear(dim, dim, bias=qkv_bias)
+ self.q_norm = WanRMSNorm(self.head_dim, eps=eps) if qk_norm else nn.Identity()
+
+ self.attn_drop = nn.Dropout(attn_drop)
+ self.proj = nn.Linear(dim, dim)
+ self.proj_drop = nn.Dropout(proj_drop)
+
+ self.kv_linear = nn.Linear(encoder_hidden_states_dim, dim * 2, bias=qkv_bias)
+ self.k_norm = WanRMSNorm(self.head_dim, eps=eps) if qk_norm else nn.Identity()
+
+ self.attention_mode = attention_mode
+
+ # multitalk related params
+ self.class_interval = class_interval
+ self.class_range = class_range
+ self.rope_h1 = (0, self.class_interval)
+ self.rope_h2 = (self.class_range - self.class_interval, self.class_range)
+ self.rope_bak = int(self.class_range // 2)
+ self.rope_1d = RotaryPositionalEmbedding1D(self.head_dim)
+
+ def _process_cross_attn(self, x, cond, frames_num=None, x_ref_attn_map=None):
+
+ N_t = frames_num
+ out_dtype = x.dtype
+ x = rearrange(x, "B (N_t S) C -> (B N_t) S C", N_t=N_t)
+
+ # get q for hidden_state
+ B, N, C = x.shape
+ q = self.q_linear(x)
+ q_shape = (B, N, self.num_heads, self.head_dim)
+ q = q.view(q_shape).permute((0, 2, 1, 3)) # [B, H, N, D]
+ q = self.q_norm(q.to(self.q_norm.weight.dtype)).to(q.dtype)
+
+ # multitalk with rope1d pe
+ if x_ref_attn_map is not None:
+ max_values = x_ref_attn_map.max(1).values[:, None, None]
+ min_values = x_ref_attn_map.min(1).values[:, None, None]
+ max_min_values = torch.cat([max_values, min_values], dim=2)
+ human1_max_value, human1_min_value = max_min_values[0, :, 0].max(), max_min_values[0, :, 1].min()
+ human2_max_value, human2_min_value = max_min_values[1, :, 0].max(), max_min_values[1, :, 1].min()
+
+ human1 = normalize_and_scale(x_ref_attn_map[0], (human1_min_value, human1_max_value), (self.rope_h1[0], self.rope_h1[1]))
+ human2 = normalize_and_scale(x_ref_attn_map[1], (human2_min_value, human2_max_value), (self.rope_h2[0], self.rope_h2[1]))
+ back = torch.full((x_ref_attn_map.size(1),), self.rope_bak, dtype=human1.dtype).to(human1.device)
+ max_indices = x_ref_attn_map.argmax(dim=0)
+ normalized_map = torch.stack([human1, human2, back], dim=1)
+ normalized_pos = normalized_map[range(x_ref_attn_map.size(1)), max_indices]
+
+ q = rearrange(q, "(B N_t) H S C -> B H (N_t S) C", N_t=N_t)
+ q = self.rope_1d(q, normalized_pos)
+ q = rearrange(q, "B H (N_t S) C -> (B N_t) H S C", N_t=N_t)
+
+ # get kv from encoder_hidden_states
+ _, N_a, _ = cond.shape
+ encoder_kv = self.kv_linear(cond)
+ encoder_kv_shape = (B, N_a, 2, self.num_heads, self.head_dim)
+ encoder_kv = encoder_kv.view(encoder_kv_shape).permute((2, 0, 3, 1, 4))
+
+ encoder_k, encoder_v = encoder_kv.unbind(0)
+ encoder_k = self.k_norm(encoder_k.to(self.k_norm.weight.dtype)).to(encoder_k.dtype)
+
+
+ # multitalk with rope1d pe
+ if x_ref_attn_map is not None:
+ per_frame = torch.zeros(N_a, dtype=encoder_k.dtype).to(encoder_k.device)
+ per_frame[:per_frame.size(0)//2] = (self.rope_h1[0] + self.rope_h1[1]) / 2
+ per_frame[per_frame.size(0)//2:] = (self.rope_h2[0] + self.rope_h2[1]) / 2
+ encoder_pos = torch.concat([per_frame]*N_t, dim=0)
+ encoder_k = rearrange(encoder_k, "(B N_t) H S C -> B H (N_t S) C", N_t=N_t)
+ encoder_k = self.rope_1d(encoder_k, encoder_pos)
+ encoder_k = rearrange(encoder_k, "B H (N_t S) C -> (B N_t) H S C", N_t=N_t)
+
+ # Input tensors must be in format ``[B, M, H, K]``, where B is the batch size, M \
+ # the sequence length, H the number of heads, and K the embeding size per head
+
+ q = rearrange(q, "B H M K -> B M H K")
+ encoder_k = rearrange(encoder_k, "B H M K -> B M H K")
+ encoder_v = rearrange(encoder_v, "B H M K -> B M H K")
+ x = attention(q, encoder_k, encoder_v, attention_mode=self.attention_mode)
+ x = rearrange(x, "B M H K -> B H M K")
+
+ # linear transform
+ x_output_shape = (B, N, C)
+ x = x.transpose(1, 2)
+ x = x.reshape(x_output_shape)
+ x = self.proj(x)
+ x = self.proj_drop(x)
+
+ # reshape x to origin shape
+ x = rearrange(x, "(B N_t) S C -> B (N_t S) C", N_t=N_t)
+
+ return x.type(out_dtype)
+
+ def forward(self, x, cond, num_latent_frames=None, num_cond_latents=None, x_ref_attn_map=None, human_num=None):
+
+ B, N, C = x.shape
+ if (num_cond_latents is None or num_cond_latents == 0):
+ # text to video
+ output = self._process_cross_attn(x, cond, num_latent_frames, x_ref_attn_map)
+ return None, output
+ elif num_cond_latents is not None and num_cond_latents > 0:
+ # image to video or video continuation
+ num_cond_latents_thw = num_cond_latents * (N // num_latent_frames)
+ x_noise = x[:, num_cond_latents_thw:]
+ cond = rearrange(cond, "(B N_t) M C -> B N_t M C", B=B)
+ cond = cond[:, num_cond_latents:]
+ cond = rearrange(cond, "B N_t M C -> (B N_t) M C")
+ frames_num = num_latent_frames - num_cond_latents
+ if human_num is not None and human_num == 2:
+ # multitalk mode
+ output_noise = self._process_cross_attn(x_noise, cond, frames_num, x_ref_attn_map)
+ else:
+ # singletalk mode
+ output_noise = self._process_cross_attn(x_noise, cond, frames_num)
+ output_cond = torch.zeros((B, num_cond_latents_thw, C), dtype=output_noise.dtype, device=output_noise.device)
+ return output_cond, output_noise
+ else:
+ raise NotImplementedError
diff --git a/LongCat/nodes.py b/LongCat/nodes.py
new file mode 100644
index 0000000..941a4fa
--- /dev/null
+++ b/LongCat/nodes.py
@@ -0,0 +1,119 @@
+import torch
+from ..utils import log
+import comfy.model_management as mm
+from comfy_api.latest import io
+
+device = mm.get_torch_device()
+offload_device = mm.unet_offload_device()
+
+
+class WanVideoLongCatAvatarExtendEmbeds(io.ComfyNode):
+ @classmethod
+ def define_schema(cls):
+ return io.Schema(
+ node_id="WanVideoLongCatAvatarExtendEmbeds",
+ category="WanVideoWrapper",
+ inputs=[
+ io.Latent.Input("prev_latents", tooltip="Full previous latents to be used to continue generation, continuation frames are selected based on 'overlap' parameter"),
+ io.Custom("MULTITALK_EMBEDS").Input("audio_embeds", tooltip="Full length audio embeddings"),
+ io.Int.Input("num_frames", default=93, min=1, max=256, step=1, tooltip="Number of new frames to generate"),
+ io.Int.Input("overlap", default=13, min=0, max=16, step=1, tooltip="Number of overlapping frames from previous latents for video continuation, set to 0 for T2V"),
+ io.Int.Input("frames_processed", default=0, min=0, max=10000, step=1, tooltip="Number of frames already processed in the video, used to select audio features"),
+ io.Combo.Input("if_not_enough_audio", ["pad_with_start", "mirror_from_end"], default="pad_with_start", tooltip="What to do if there are not enough frames in pose_images for the window"),
+ io.Int.Input("ref_frame_index", default=10, min=0, max=1000, step=1, tooltip="Values between 0 - 24 ensures better consistency, while selecting other ranges (e.g., -10 or 30) helps reduce repeated actions"),
+ io.Int.Input("ref_mask_frame_range", default=3, min=0, max=20, step=1, tooltip="Larger range can further help mitigate repeated actions, but excessively large values may introduce artifacts"),
+ io.Latent.Input("ref_latent", optional=True, tooltip="Reference latent used for consistency, generally should be either the init image, or first latent from first generation"),
+ io.Latent.Input("samples", optional=True, tooltip="For the sampler 'samples' input, used for slicing samples per window for vid2vid"),
+ ],
+ outputs=[
+ io.Custom("WANVIDIMAGE_EMBEDS").Output(display_name="image_embeds", tooltip="Embeds for WanVideo LongCat Avatar generation"),
+ io.Latent.Output(display_name="samples_slice", tooltip="Sliced latent samples for the new frames"),
+ ],
+ )
+
+ @classmethod
+ def execute(cls, prev_latents, audio_embeds, num_frames, overlap, if_not_enough_audio, frames_processed, ref_frame_index, ref_mask_frame_range, ref_latent=None, samples=None) -> io.NodeOutput:
+
+ new_audio_embed = audio_embeds.copy()
+
+ audio_features = torch.stack(new_audio_embed["audio_features"])
+ if audio_features.shape[1] < frames_processed + num_frames:
+ deficit = frames_processed + num_frames - audio_features.shape[1]
+ if if_not_enough_audio == "pad_with_start":
+ pad = audio_features[:, :1].repeat(1, deficit, 1, 1, 1)
+ audio_features = torch.cat([audio_features, pad], dim=1)
+ elif if_not_enough_audio == "mirror_from_end":
+ to_add = audio_features[:, -deficit:, :].flip(dims=[1])
+ audio_features = torch.cat([audio_features, to_add], dim=1)
+ log.info(f"Not enough audio features, extended from {new_audio_embed['audio_features'].shape[1]} to {audio_features.shape[1]} frames.")
+
+ ref_target_masks = new_audio_embed.get("ref_target_masks", None)
+ if ref_target_masks is not None:
+ new_audio_embed["ref_target_masks"] = ref_target_masks[:, frames_processed:frames_processed+num_frames, :]
+
+ prev_samples = prev_latents["samples"].clone()
+ if overlap != 0:
+ latent_overlap = (overlap - 1) // 4 + 1
+ prev_samples = prev_samples[:, :, -latent_overlap:]
+
+ ref_sample = None
+ if ref_latent is not None:
+ ref_sample = ref_latent["samples"][0, :, :1].clone()
+ log.info(f"Previous latents shape: {prev_samples.shape}, using last {latent_overlap} latent frames for overlap.")
+
+ new_latent_frames = (num_frames - 1) // 4 + 1
+ target_shape = (16, new_latent_frames, prev_samples.shape[-2], prev_samples.shape[-1])
+
+ audio_stride = 2
+ indices = torch.arange(2 * 2 + 1) - 2
+
+ if frames_processed == 0:
+ audio_start_idx = 0
+ else:
+ audio_start_idx = (frames_processed - overlap) * audio_stride
+ audio_end_idx = audio_start_idx + num_frames * audio_stride
+
+ log.info(f"Extracting audio embeddings from index {audio_start_idx} to {audio_end_idx}")
+
+ audio_embs = []
+ for human_idx in range(len(audio_features)):
+ center_indices = torch.arange(audio_start_idx, audio_end_idx, audio_stride).unsqueeze(1) + indices.unsqueeze(0)
+ center_indices = torch.clamp(center_indices, min=0, max=audio_features[human_idx].shape[0] - 1)
+
+ audio_emb = audio_features[human_idx][center_indices].unsqueeze(0).to(device)
+ audio_embs.append(audio_emb)
+ audio_emb = torch.cat(audio_embs, dim=0)
+
+ new_audio_embed["audio_features"] = None
+ new_audio_embed["audio_emb_slice"] = audio_emb
+
+ longcat_avatar_options = {
+ "longcat_ref_latent": ref_sample,
+ "ref_frame_index": ref_frame_index,
+ "ref_mask_frame_range": ref_mask_frame_range,
+ }
+
+ embeds = {
+ "target_shape": target_shape,
+ "num_frames": num_frames,
+ "extra_latents": [{"samples": prev_samples, "index": 0}] if overlap != 0 else None,
+ "multitalk_embeds": new_audio_embed,
+ "longcat_avatar_options": longcat_avatar_options,
+ }
+
+ samples_slice = None
+ if samples is not None:
+ latent_start_index = (frames_processed - 1) // 4 + 1 if frames_processed > 0 else 0
+ latent_end_index = latent_start_index + new_latent_frames
+ samples_slice = samples.copy()
+ samples_slice["samples"] = samples["samples"][:, :, latent_start_index:latent_end_index].clone()
+
+ return io.NodeOutput(embeds, samples_slice)
+
+
+NODE_CLASS_MAPPINGS = {
+ "WanVideoLongCatAvatarExtendEmbeds": WanVideoLongCatAvatarExtendEmbeds,
+ }
+NODE_DISPLAY_NAME_MAPPINGS = {
+ "WanVideoLongCatAvatarExtendEmbeds": "WanVideo LongCat Avatar Extend Embeds",
+ }
\ No newline at end of file
diff --git a/__init__.py b/__init__.py
index fef124e..4efb858 100644
--- a/__init__.py
+++ b/__init__.py
@@ -48,6 +48,7 @@ OPTIONAL_MODULES = [
(".onetoall.nodes", "OneToAll"),
(".WanMove.nodes", "WanMove"),
(".SCAIL.nodes", "SCAIL"),
+ (".LongCat.nodes", "LongCat"),
]
def register_nodes(module_path: str, name: str, optional: bool) -> None:
diff --git a/multitalk/nodes.py b/multitalk/nodes.py
index fed8fb8..837a763 100644
--- a/multitalk/nodes.py
+++ b/multitalk/nodes.py
@@ -7,6 +7,8 @@ from ..utils import log, set_module_tensor_to_device
import os
import json
import datetime
+import scipy.signal as ss
+import numpy as np
script_directory = os.path.dirname(os.path.abspath(__file__))
folder_paths.add_model_folder_path("wav2vec2", os.path.join(folder_paths.models_dir, "wav2vec2"))
@@ -134,6 +136,15 @@ def loudness_norm(audio_array, sr=16000, lufs=-23):
return audio_array
normalized_audio = pyloudnorm.normalize.loudness(audio_array, loudness, lufs)
return normalized_audio
+
+def _add_noise_floor(audio, noise_db=-45):
+ noise_amp = 10 ** (noise_db / 20)
+ noise = np.random.randn(len(audio)) * noise_amp
+ return audio + noise
+
+def _smooth_transients(audio, sr=16000):
+ b, a = ss.butter(3, 3000 / (sr/2))
+ return ss.lfilter(b, a, audio)
class MultiTalkWav2VecEmbeds:
@classmethod
@@ -153,6 +164,8 @@ class MultiTalkWav2VecEmbeds:
"audio_3": ("AUDIO",),
"audio_4": ("AUDIO",),
"ref_target_masks": ("MASK", {"tooltip": "Per-speaker semantic mask(s) in pixel space. Supply one mask per speaker (plus optional background) to guide mouth assignment"}),
+ "add_noise_floor": ("BOOLEAN", {"default": False, "tooltip": "Add a low-level noise floor to the audio to reduce silent gaps"}),
+ "smooth_transients": ("BOOLEAN", {"default": False, "tooltip": "Apply a low-pass filter to the audio to smooth out transients"}),
}
}
@@ -161,7 +174,8 @@ class MultiTalkWav2VecEmbeds:
FUNCTION = "process"
CATEGORY = "WanVideoWrapper"
- def process(self, wav2vec_model, normalize_loudness, fps, num_frames, audio_1, audio_scale, audio_cfg_scale, multi_audio_type, audio_2=None, audio_3=None, audio_4=None, ref_target_masks=None):
+ def process(self, wav2vec_model, normalize_loudness, fps, num_frames, audio_1, audio_scale, audio_cfg_scale, multi_audio_type, audio_2=None, audio_3=None, audio_4=None,
+ ref_target_masks=None, add_noise_floor=False, smooth_transients=False):
model_type = wav2vec_model["model_type"]
if not "tencent" in model_type.lower():
raise ValueError("Only tencent wav2vec2 models supported by MultiTalk")
@@ -207,6 +221,10 @@ class MultiTalkWav2VecEmbeds:
if normalize_loudness:
audio_segment = loudness_norm(audio_segment, sr=sr)
+ if add_noise_floor:
+ audio_segment = _add_noise_floor(audio_segment, noise_db=-45)
+ if smooth_transients:
+ audio_segment = _smooth_transients(audio_segment, sr=sr)
audio_feature = np.squeeze(
wav2vec2_feature_extractor(audio_segment, sampling_rate=sr).input_values
diff --git a/nodes.py b/nodes.py
index c35f5cd..0c78f44 100644
--- a/nodes.py
+++ b/nodes.py
@@ -2046,7 +2046,7 @@ class WanVideoDecode:
video.clamp_(-1.0, 1.0)
video.add_(1.0).div_(2.0)
return video.cpu().float(),
- latents = samples["samples"]
+ latents = samples["samples"].clone()
end_image = samples.get("end_image", None)
has_ref = samples.get("has_ref", False)
drop_last = samples.get("drop_last", False)
diff --git a/nodes_model_loading.py b/nodes_model_loading.py
index e2c24d7..68239da 100644
--- a/nodes_model_loading.py
+++ b/nodes_model_loading.py
@@ -1128,7 +1128,7 @@ class WanVideoModelLoader:
scale_weights = {}
if "fp8" in quantization:
for k, v in sd.items():
- if k.endswith(".scale_weight"):
+ if k.endswith(".scale_weight") or k.endswith(".weight_scale"):
is_scaled_fp8 = True
break
@@ -1153,7 +1153,7 @@ class WanVideoModelLoader:
# currently this can be VACE, MTV-Crafter, Lynx or Ovi-audio weights
if extra_model is not None:
for _model in extra_model:
- print("Loading extra model: ", _model["path"])
+ log.info(f"Loading extra model: {_model['path']}")
if gguf:
if not _model["path"].endswith(".gguf"):
raise ValueError("With GGUF main model the extra model must also be GGUF quantized, if the main model already has VACE included, you can disconnect the extra module loader")
@@ -1479,6 +1479,39 @@ class WanVideoModelLoader:
sd.update(extra_sd)
del extra_sd
+ elif "multitalk_audio_proj.proj1.weight" in sd:
+ log.info("MultiTalk/InfiniteTalk model detected, patching model...")
+ from .multitalk.multitalk import AudioProjModel
+ from .wanvideo.modules.model import WanLayerNorm
+ from .LongCat.layers import SingleStreamAttention
+
+ audio_window = 5
+ vae_scale = 4
+
+ for block in transformer.blocks:
+ with init_empty_weights():
+ if "blocks.0.audio_modulation.1.weight" in sd:
+ block.audio_modulation = nn.Sequential(nn.SiLU(), nn.Linear(512, 3 * dim, bias=True))
+ block.norm_x = WanLayerNorm(dim, transformer.eps, elementwise_affine=True)
+ block.audio_cross_attn = SingleStreamAttention(
+ dim=dim,
+ encoder_hidden_states_dim=768,
+ num_heads=num_heads,
+ qkv_bias=True,
+ qk_norm=True,
+ class_range=24,
+ class_interval=4,
+ attention_mode=attention_mode,
+ )
+ multitalk_proj_model = AudioProjModel(
+ seq_len=audio_window,
+ seq_len_vf=audio_window+vae_scale-1,
+ intermediate_dim=512,
+ output_dim=768,
+ context_tokens=32,
+ norm_output_audio=True,
+ )
+ transformer.multitalk_audio_proj = multitalk_proj_model
sd = {k.replace(".weight_scale", ".scale_weight"): v for k, v in sd.items()}
diff --git a/nodes_sampler.py b/nodes_sampler.py
index 6078a5e..d6afd6b 100644
--- a/nodes_sampler.py
+++ b/nodes_sampler.py
@@ -294,10 +294,11 @@ class WanVideoSampler:
control_latents = control_camera_latents = clip_fea = clip_fea_neg = end_image = recammaster = camera_embed = unianim_data = mocha_embeds = image_cond_neg =None
vace_data = vace_context = vace_scale = None
- fun_or_fl2v_model = has_ref = drop_last = False
+ fun_or_fl2v_model = drop_last = False
phantom_latents = fun_ref_image = ATI_tracks = None
add_cond = attn_cond = attn_cond_neg = noise_pred_flipped = None
humo_audio = humo_audio_neg = None
+ has_ref = image_embeds.get("has_ref", False)
#I2V
image_cond = image_embeds.get("image_embeds", None)
@@ -363,15 +364,11 @@ class WanVideoSampler:
control_camera_end_percent = control_embeds.get("control_camera_end_percent", 1.0)
drop_last = image_embeds.get("drop_last", False)
- has_ref = image_embeds.get("has_ref", False)
-
else: #t2v
target_shape = image_embeds.get("target_shape", None)
if target_shape is None:
raise ValueError("Empty image embeds must be provided for T2V models")
- has_ref = image_embeds.get("has_ref", False)
-
# VACE
vace_context = image_embeds.get("vace_context", None)
vace_scale = image_embeds.get("vace_scale", None)
@@ -633,27 +630,34 @@ class WanVideoSampler:
if not isinstance(audio_cfg_scale, list):
audio_cfg_scale = [audio_cfg_scale] * (steps +1)
log.info(f"Audio proj shape: {audio_proj.shape}")
- elif multitalk_embeds is not None:
+
+
+ # MultiTalk
+ multitalk_audio_embeds = audio_emb_slice = audio_features_in = None
+ multitalk_embeds = image_embeds.get("multitalk_embeds", multitalk_embeds)
+
+ if multitalk_embeds is not None:
+ audio_emb_slice = multitalk_embeds.get("audio_emb_slice", None) # if already sliced
+ print("audio_emb_slice:", audio_emb_slice.shape)
# Handle single or multiple speaker embeddings
- audio_features_in = multitalk_embeds.get("audio_features", None)
- if audio_features_in is None:
- multitalk_audio_embeds = None
- else:
+ if audio_emb_slice is None:
+ audio_features_in = multitalk_embeds.get("audio_features", None)
+ if audio_features_in is not None:
if isinstance(audio_features_in, list):
multitalk_audio_embeds = [emb.to(device, dtype) for emb in audio_features_in]
else:
# keep backward-compatibility with single tensor input
multitalk_audio_embeds = [audio_features_in.to(device, dtype)]
+ shapes = [tuple(e.shape) for e in multitalk_audio_embeds]
+ log.info(f"Multitalk audio features shapes (per speaker): {shapes}")
+
audio_scale = multitalk_embeds.get("audio_scale", 1.0)
audio_cfg_scale = multitalk_embeds.get("audio_cfg_scale", 1.0)
ref_target_masks = multitalk_embeds.get("ref_target_masks", None)
if not isinstance(audio_cfg_scale, list):
audio_cfg_scale = [audio_cfg_scale] * (steps + 1)
- shapes = [tuple(e.shape) for e in multitalk_audio_embeds]
- log.info(f"Multitalk audio features shapes (per speaker): {shapes}")
-
# FantasyPortrait
fantasy_portrait_input = None
fantasy_portrait_embeds = image_embeds.get("portrait_embeds", None)
@@ -820,7 +824,7 @@ class WanVideoSampler:
# extra latents (Pusa) and 5b
latents_to_insert = add_index = noise_multipliers = None
extra_latents = image_embeds.get("extra_latents", None)
- all_indices = []
+ clean_latent_indices = []
noise_multiplier_list = image_embeds.get("pusa_noise_multipliers", None)
if noise_multiplier_list is not None:
if len(noise_multiplier_list) != latent_video_length:
@@ -830,7 +834,7 @@ class WanVideoSampler:
log.info(f"Using Pusa noise multipliers: {noise_multipliers}")
if extra_latents is not None and transformer.multitalk_model_type.lower() != "infinitetalk":
if noise_multiplier_list is not None:
- noise_multiplier_list = list(noise_multiplier_list) + [1.0] * (len(all_indices) - len(noise_multiplier_list))
+ noise_multiplier_list = list(noise_multiplier_list) + [1.0] * (len(clean_latent_indices) - len(noise_multiplier_list))
for i, entry in enumerate(extra_latents):
add_index = entry["index"]
num_extra_frames = entry["samples"].shape[2]
@@ -841,9 +845,9 @@ class WanVideoSampler:
if start_step == 0:
noise[:, add_index:add_index+num_extra_frames] = entry["samples"].to(noise)
log.info(f"Adding extra samples to latent indices {add_index} to {add_index+num_extra_frames-1}")
- all_indices.extend(range(add_index, add_index+num_extra_frames))
+ clean_latent_indices.extend(range(add_index, add_index+num_extra_frames))
if noise_multipliers is not None and len(noise_multiplier_list) != latent_video_length:
- for i, idx in enumerate(all_indices):
+ for i, idx in enumerate(clean_latent_indices):
noise_multipliers[idx] = noise_multiplier_list[i]
log.info(f"Using Pusa noise multipliers: {noise_multipliers}")
@@ -869,6 +873,25 @@ class WanVideoSampler:
latent = noise
+ # LongCat-Avatar
+ longcat_ref_latent = None
+ longcat_num_ref_latents = longcat_num_cond_latents = 0
+ longcat_avatar_options = image_embeds.get("longcat_avatar_options", None)
+
+ if longcat_avatar_options is not None:
+ longcat_ref_latent = longcat_avatar_options.get("longcat_ref_latent", None)
+ if longcat_ref_latent is not None:
+ log.info(f"LongCat-Avatar reference latent shape: {longcat_ref_latent.shape}")
+ latent = torch.cat([longcat_ref_latent.to(latent), latent], dim=1)
+ seq_len = math.ceil((latent.shape[2] * latent.shape[3]) / 4 * latent.shape[1])
+ insert_len = longcat_ref_latent.shape[1]
+ clean_latent_indices = list(range(0, insert_len)) + [i + insert_len for i in clean_latent_indices]
+ longcat_num_ref_latents = longcat_ref_latent.shape[1]
+ latent_video_length += insert_len
+ longcat_num_cond_latents = len(clean_latent_indices)
+ log.info(f"LongCat num_cond_latents: {longcat_num_cond_latents} num_ref_latents: {longcat_num_ref_latents}")
+ audio_stride = 2 if transformer.is_longcat else 1
+
#controlnet
controlnet_latents = controlnet = None
if transformer_options is not None:
@@ -1385,27 +1408,30 @@ class WanVideoSampler:
else:
z = torch.cat([z, minimax_latents, minimax_mask_latents], dim=0)
- if not multitalk_sampling and multitalk_audio_embeds is not None:
+ multitalk_audio_input = None
+ if audio_emb_slice is not None:
+ multitalk_audio_input = audio_emb_slice.to(z)
+ elif not multitalk_sampling and multitalk_audio_embeds is not None:
audio_embedding = multitalk_audio_embeds
audio_embs = []
indices = (torch.arange(4 + 1) - 2) * 1
human_num = len(audio_embedding)
# split audio with window size
+ audio_end_idx = latent_video_length * 4 + 1 if add_cond is not None else (latent_video_length-1) * 4 + 1
+ audio_end_idx = audio_end_idx * audio_stride
if context_window is None:
for human_idx in range(human_num):
- center_indices = torch.arange(
- 0,
- latent_video_length * 4 + 1 if add_cond is not None else (latent_video_length-1) * 4 + 1,
- 1).unsqueeze(1) + indices.unsqueeze(0)
+ center_indices = torch.arange(0, audio_end_idx, audio_stride).unsqueeze(1) + indices.unsqueeze(0)
center_indices = torch.clamp(center_indices, min=0, max=audio_embedding[human_idx].shape[0] - 1)
+
audio_emb = audio_embedding[human_idx][center_indices].unsqueeze(0).to(device)
audio_embs.append(audio_emb)
else:
for human_idx in range(human_num):
- audio_start = context_window[0] * 4
- audio_end = context_window[-1] * 4 + 1
+ audio_start = (context_window[0] * 4) * audio_stride
+ audio_end = (context_window[-1] * 4 + 1) * audio_stride
#print("audio_start: ", audio_start, "audio_end: ", audio_end)
- center_indices = torch.arange(audio_start, audio_end, 1).unsqueeze(1) + indices.unsqueeze(0)
+ center_indices = torch.arange(audio_start, audio_end, audio_stride).unsqueeze(1) + indices.unsqueeze(0)
center_indices = torch.clamp(center_indices, min=0, max=audio_embedding[human_idx].shape[0] - 1)
audio_emb = audio_embedding[human_idx][center_indices].unsqueeze(0).to(device)
audio_embs.append(audio_emb)
@@ -1513,7 +1539,7 @@ class WanVideoSampler:
"add_cond": add_cond_input, # additional conditioning input
"nag_params": text_embeds.get("nag_params", {}), # normalized attention guidance
"nag_context": text_embeds.get("nag_prompt_embeds", None), # normalized attention guidance context
- "multitalk_audio": multitalk_audio_input if multitalk_audio_embeds is not None else None, # Multi/InfiniteTalk audio input
+ "multitalk_audio": multitalk_audio_input, # Multi/InfiniteTalk audio input
"ref_target_masks": ref_target_masks if multitalk_audio_embeds is not None else None, # Multi/InfiniteTalk reference target masks
"inner_t": [shot_len] if shot_len else None, # inner timestep for EchoShot
"standin_input": standin_input, # Stand-in reference input
@@ -1543,7 +1569,9 @@ class WanVideoSampler:
"ovi_negative_text_embeds": ovi_negative_text_embeds, # Audio latent model negative text embeds for Ovi
"flashvsr_LQ_latent": flashvsr_LQ_latent, # FlashVSR LQ latent for upsampling
"flashvsr_strength": flashvsr_strength, # FlashVSR strength
- "num_cond_latents": len(all_indices) if transformer.is_longcat else None,
+ "longcat_num_cond_latents": longcat_num_cond_latents,
+ "longcat_num_ref_latents": longcat_num_ref_latents,
+ "longcat_avatar_options": longcat_avatar_options, # LongCat avatar attention options
"sdancer_input": sdancer_input, # SteadyDancer input
"one_to_all_input": one_to_all_data, # One-to-All input
"one_to_all_controlnet_strength": one_to_all_data["controlnet_strength"] if one_to_all_data is not None else 0.0,
@@ -1576,12 +1604,16 @@ class WanVideoSampler:
if use_fresca:
noise_pred_cond = fourier_filter(noise_pred_cond, fresca_scale_low, fresca_scale_high, fresca_freq_cutoff)
if fantasy_portrait_input is not None and not math.isclose(portrait_cfg[idx], 1.0):
- print("Applying Fantasy Portrait CFG...")
base_params["fantasy_portrait_input"] = None
noise_pred_no_portrait, noise_pred_ovi, cache_state_uncond = transformer(context=positive_embeds, pred_id=cache_state[0] if cache_state else None,
vace_data=vace_data, attn_cond=attn_cond, **base_params)
- noise_pred_no_portrait = noise_pred_no_portrait[0]
- return noise_pred_no_portrait + portrait_cfg[idx] * (noise_pred_cond - noise_pred_no_portrait), noise_pred_ovi, [cache_state_cond, cache_state_uncond]
+ return noise_pred_no_portrait[0] + portrait_cfg[idx] * (noise_pred_cond - noise_pred_no_portrait[0]), noise_pred_ovi, [cache_state_cond, cache_state_uncond]
+ elif multitalk_audio_input is not None and not math.isclose(audio_cfg_scale[idx], 1.0):
+ base_params['multitalk_audio'] = torch.zeros_like(multitalk_audio_input)[-1:]
+ noise_pred_uncond_audio, _, cache_state_uncond = transformer(
+ context=positive_embeds, pred_id=cache_state[0] if cache_state else None,
+ vace_data=vace_data, attn_cond=attn_cond, **base_params)
+ return noise_pred_uncond_audio[0] + audio_cfg_scale[idx] * (noise_pred_cond - noise_pred_uncond_audio[0]), noise_pred_ovi, [cache_state_cond, cache_state_uncond]
else:
return noise_pred_cond, noise_pred_ovi, [cache_state_cond]
@@ -1597,12 +1629,12 @@ class WanVideoSampler:
if neg_latent is not None:
base_params['x'] = [torch.cat([z[:, :-humo_reference_count], neg_latent], dim=1)]
- noise_pred_uncond, noise_pred_ovi_uncond, cache_state_uncond = transformer(
+ noise_pred_uncond_text, noise_pred_ovi_uncond, cache_state_uncond = transformer(
context=negative_embeds if humo_audio_input_neg is None else positive_embeds, #ti #t
pred_id=cache_state[1] if cache_state else None,
vace_data=vace_data, attn_cond=attn_cond_neg,
**base_params)
- noise_pred_uncond = noise_pred_uncond[0]
+ noise_pred_uncond_text = noise_pred_uncond_text[0]
noise_pred_ovi_uncond = noise_pred_ovi_uncond[0] if noise_pred_ovi_uncond is not None else None
# HuMo
@@ -1617,8 +1649,8 @@ class WanVideoSampler:
context=negative_embeds, pred_id=cache_state[2] if cache_state else None, vace_data=None,
**base_params)
- noise_pred = (noise_pred_uncond + humo_audio_cfg_scale[idx] * (noise_pred_cond - noise_pred_humo_audio_uncond[0])
- + (cfg_scale - 2.0) * (noise_pred_humo_audio_uncond[0] - noise_pred_uncond))
+ noise_pred = (noise_pred_uncond_text + humo_audio_cfg_scale[idx] * (noise_pred_cond - noise_pred_humo_audio_uncond[0])
+ + (cfg_scale - 2.0) * (noise_pred_humo_audio_uncond[0] - noise_pred_uncond_text))
return noise_pred, None, [cache_state_cond, cache_state_uncond, cache_state_humo]
elif humo_audio_input is not None:
if cache_state is not None and len(cache_state) != 4:
@@ -1634,8 +1666,8 @@ class WanVideoSampler:
context=positive_embeds, pred_id=cache_state[3] if cache_state else None, vace_data=None,
**base_params)
noise_pred = (humo_audio_cfg_scale[idx] * (noise_pred_cond - noise_pred_humo_audio[0])
- + cfg_scale * (noise_pred_humo_audio[0] - noise_pred_uncond)
- + cfg_scale * (noise_pred_uncond - noise_pred_humo_null[0])
+ + cfg_scale * (noise_pred_humo_audio[0] - noise_pred_uncond_text)
+ + cfg_scale * (noise_pred_uncond_text - noise_pred_humo_null[0])
+ noise_pred_humo_null[0])
return noise_pred, None, [cache_state_cond, cache_state_uncond, cache_state_humo, cache_state_humo2]
@@ -1647,32 +1679,28 @@ class WanVideoSampler:
context=negative_embeds, pred_id=cache_state[2] if cache_state else None, vace_data=None,
**base_params)
- noise_pred = (noise_pred_uncond + phantom_cfg_scale[idx] * (noise_pred_phantom[0] - noise_pred_uncond)
+ noise_pred = (noise_pred_uncond_text + phantom_cfg_scale[idx] * (noise_pred_phantom[0] - noise_pred_uncond_text)
+ cfg_scale * (noise_pred_cond - noise_pred_phantom[0]))
return noise_pred, None,[cache_state_cond, cache_state_uncond, cache_state_phantom]
# audio cfg (fantasytalking and multitalk)
- if (fantasytalking_embeds is not None or multitalk_audio_embeds is not None):
+ if (fantasytalking_embeds is not None or multitalk_audio_input is not None):
if not math.isclose(audio_cfg_scale[idx], 1.0):
if cache_state is not None and len(cache_state) != 3:
cache_state.append(None)
- # Set audio parameters to None/zeros based on type
- if fantasytalking_embeds is not None:
- base_params['audio_proj'] = None
- audio_context = positive_embeds
- else: # multitalk
- base_params['multitalk_audio'] = torch.zeros_like(multitalk_audio_input)[-1:]
- audio_context = negative_embeds
+ base_params['audio_proj'] = None
+ base_params['multitalk_audio'] = torch.zeros_like(multitalk_audio_input)[-1:] if multitalk_audio_input is not None else None
base_params['is_uncond'] = False
- noise_pred_no_audio, _, cache_state_audio = transformer(
- context=audio_context,
+ noise_pred_uncond_audio, _, cache_state_audio = transformer(
+ context=negative_embeds,
pred_id=cache_state[2] if cache_state else None,
vace_data=vace_data,
**base_params)
+ noise_pred_uncond_audio = noise_pred_uncond_audio[0]
- noise_pred = (noise_pred_uncond
- + cfg_scale * (noise_pred_no_audio[0] - noise_pred_uncond)
- + audio_cfg_scale[idx] * (noise_pred_cond - noise_pred_no_audio[0]))
+ noise_pred = noise_pred_uncond_audio + cfg_scale * (
+ (noise_pred_cond - noise_pred_uncond_text)
+ + audio_cfg_scale[idx] * (noise_pred_uncond_text - noise_pred_uncond_audio))
return noise_pred, None,[cache_state_cond, cache_state_uncond, cache_state_audio]
# lynx
if lynx_embeds is not None and not math.isclose(lynx_cfg_scale[idx], 1.0):
@@ -1683,7 +1711,7 @@ class WanVideoSampler:
context=negative_embeds, pred_id=cache_state[2] if cache_state else None, vace_data=None,
**base_params)
- noise_pred = (noise_pred_uncond + lynx_cfg_scale[idx] * (noise_pred_lynx[0] - noise_pred_uncond)
+ noise_pred = (noise_pred_uncond_text + lynx_cfg_scale[idx] * (noise_pred_lynx[0] - noise_pred_uncond_text)
+ cfg_scale * (noise_pred_cond - noise_pred_lynx[0]))
return noise_pred, None, [cache_state_cond, cache_state_uncond, cache_state_lynx]
# one-to-all
@@ -1697,7 +1725,7 @@ class WanVideoSampler:
context=negative_embeds, pred_id=cache_state[2] if cache_state else None, vace_data=None,
**base_params)
- noise_pred = (noise_pred_uncond + one_to_all_pose_cfg_scale[idx] * (noise_pred_pose_uncond[0] - noise_pred_uncond)
+ noise_pred = (noise_pred_uncond_text + one_to_all_pose_cfg_scale[idx] * (noise_pred_pose_uncond[0] - noise_pred_uncond_text)
+ cfg_scale * (noise_pred_cond - noise_pred_pose_uncond[0]))
return noise_pred, None, [cache_state_cond, cache_state_uncond, cache_state_ref]
@@ -1727,23 +1755,23 @@ class WanVideoSampler:
noise_pred_uncond.view(batch_size, -1)
).view(batch_size, 1, 1, 1)
- noise_pred_uncond_scaled = noise_pred_uncond * alpha
+ noise_pred_uncond_text = noise_pred_uncond_text * alpha
if use_tangential:
- noise_pred_uncond_scaled = tangential_projection(noise_pred_cond, noise_pred_uncond_scaled)
+ noise_pred_uncond_text = tangential_projection(noise_pred_cond, noise_pred_uncond_text)
# RAAG (RATIO-aware Adaptive Guidance)
if raag_alpha > 0.0:
- cfg_scale = get_raag_guidance(noise_pred_cond, noise_pred_uncond_scaled, cfg_scale, raag_alpha)
+ cfg_scale = get_raag_guidance(noise_pred_cond, noise_pred_uncond_text, cfg_scale, raag_alpha)
log.info(f"RAAG modified cfg: {cfg_scale}")
#https://github.com/WikiChao/FreSca
if use_fresca:
filtered_cond = fourier_filter(noise_pred_cond - noise_pred_uncond, fresca_scale_low, fresca_scale_high, fresca_freq_cutoff)
- noise_pred = noise_pred_uncond_scaled + cfg_scale * filtered_cond * alpha
+ noise_pred = noise_pred_uncond_text + cfg_scale * filtered_cond * alpha
else:
- noise_pred = noise_pred_uncond_scaled + cfg_scale * (noise_pred_cond - noise_pred_uncond_scaled)
- del noise_pred_uncond_scaled, noise_pred_cond, noise_pred_uncond
+ noise_pred = noise_pred_uncond_text + cfg_scale * (noise_pred_cond - noise_pred_uncond_text)
+ del noise_pred_uncond_text, noise_pred_cond
if latent_model_input_ovi is not None:
if ovi_audio_cfg is None:
@@ -1786,7 +1814,6 @@ class WanVideoSampler:
gc.collect()
try:
torch.cuda.reset_peak_memory_stats(device)
- #torch.cuda.memory._record_memory_history(max_entries=100000)
except:
pass
@@ -1842,7 +1869,7 @@ class WanVideoSampler:
# Set latent for denoising
latent = current_latent
- if is_pusa and all_indices:
+ if is_pusa and clean_latent_indices:
pusa_noisy_steps = image_embeds.get("pusa_noisy_steps", -1)
if pusa_noisy_steps == -1:
pusa_noisy_steps = len(timesteps)
@@ -1873,15 +1900,15 @@ class WanVideoSampler:
current_step_percentage = idx / len(timesteps)
timestep = torch.tensor([t]).to(device)
- if is_pusa or ((is_5b or transformer.is_longcat) and all_indices):
+ if is_pusa or ((is_5b or transformer.is_longcat) and clean_latent_indices):
orig_timestep = timestep
timestep = timestep.unsqueeze(1).repeat(1, latent_video_length)
if extra_latents is not None:
- if all_indices and noise_multipliers is not None:
+ if clean_latent_indices and noise_multipliers is not None:
if is_pusa:
- scheduler_step_args["cond_frame_latent_indices"] = all_indices
+ scheduler_step_args["cond_frame_latent_indices"] = clean_latent_indices
scheduler_step_args["noise_multipliers"] = noise_multipliers
- for latent_idx in all_indices:
+ for latent_idx in clean_latent_indices:
timestep[:, latent_idx] = timestep[:, latent_idx] * noise_multipliers[latent_idx]
# add noise for conditioning frames if multiplier > 0
if idx < pusa_noisy_steps and noise_multipliers[latent_idx] > 0:
@@ -1895,7 +1922,7 @@ class WanVideoSampler:
timestep_cond[:, latent_idx:latent_idx+1].to(device),
noise_multiplier=noise_multipliers[latent_idx])
else:
- timestep[:, all_indices] = 0
+ timestep[:, clean_latent_indices] = 0
#print("timestep: ", timestep)
### latent shift
@@ -2265,8 +2292,8 @@ class WanVideoSampler:
is_first_clip = True
arrive_last_frame = False
cur_motion_frames_num = 1
- audio_start_idx = iteration_count = step_iteration_count= 0
- audio_end_idx = audio_start_idx + clip_length
+ audio_start_idx = iteration_count = step_iteration_count = 0
+ audio_end_idx = (audio_start_idx + clip_length) * audio_stride
indices = (torch.arange(4 + 1) - 2) * 1
current_condframe_index = 0
@@ -2308,7 +2335,7 @@ class WanVideoSampler:
audio_embs = []
# split audio with window size
for human_idx in range(human_num):
- center_indices = torch.arange(audio_start_idx, audio_end_idx, 1).unsqueeze(1) + indices.unsqueeze(0)
+ center_indices = torch.arange(audio_start_idx, audio_end_idx, audio_stride).unsqueeze(1) + indices.unsqueeze(0)
center_indices = torch.clamp(center_indices, min=0, max=audio_embedding[human_idx].shape[0]-1)
audio_emb = audio_embedding[human_idx][center_indices].unsqueeze(0).to(device)
audio_embs.append(audio_emb)
@@ -3134,28 +3161,10 @@ class WanVideoSampler:
if transformer.is_longcat:
noise_pred = -noise_pred
- if len(timestep.shape) != 1 and not is_pusa: #5b and longcat
- # all_indices is a list of indices to skip
- total_indices = list(range(latent.shape[1]))
- process_indices = [i for i in total_indices if i not in all_indices]
- if process_indices:
- latent_to_process = latent[:, process_indices]
- noise_pred_to_process = noise_pred[:, process_indices]
- latent_slice = sample_scheduler.step(
- noise_pred_to_process.unsqueeze(0),
- orig_timestep,
- latent_to_process.unsqueeze(0),
- **scheduler_step_args
- )[0].squeeze(0)
- # Reconstruct the latent tensor: keep skipped indices as-is, update others
- new_latent = []
- for i in total_indices:
- if i in all_indices:
- new_latent.append(latent[:, i:i+1])
- else:
- j = process_indices.index(i)
- new_latent.append(latent_slice[:, j:j+1])
- latent = torch.cat(new_latent, dim=1)
+ if len(timestep.shape) != 1 and clean_latent_indices and not is_pusa: #5b and longcat, skip clean latents for scheduler step
+ step_process_indices = [i for i in range(latent.shape[1]) if i not in clean_latent_indices]
+ latent[:, step_process_indices] = sample_scheduler.step(noise_pred[:, step_process_indices].unsqueeze(0), orig_timestep,
+ latent[:, step_process_indices].unsqueeze(0), **scheduler_step_args)[0].squeeze(0)
else:
if latents_to_not_step > 0:
raw_latent = latent[:, :latents_to_not_step]
@@ -3168,11 +3177,7 @@ class WanVideoSampler:
noise_pred_in = noise_pred
latent = sample_scheduler.step(noise_pred_in.unsqueeze(0), timestep, latent.unsqueeze(0), **scheduler_step_args)[0].squeeze(0)
if noise_pred_flipped is not None:
- latent_backwards = sample_scheduler_flipped.step(
- noise_pred_flipped.unsqueeze(0),
- timestep,
- latent_flipped.unsqueeze(0),
- **scheduler_step_args)[0].squeeze(0)
+ latent_backwards = sample_scheduler_flipped.step(noise_pred_flipped.unsqueeze(0), timestep, latent_flipped.unsqueeze(0), **scheduler_step_args)[0].squeeze(0)
latent_backwards = torch.flip(latent_backwards, dims=[1])
latent = latent * 0.5 + latent_backwards * 0.5
if latents_to_not_step > 0:
@@ -3242,6 +3247,8 @@ class WanVideoSampler:
latent = latent[:,:-phantom_latents.shape[1]]
if humo_reference_count > 0:
latent = latent[:,:-humo_reference_count]
+ if longcat_ref_latent is not None:
+ latent = latent[:, longcat_ref_latent.shape[1]:]
cache_states = None
if cache_args is not None:
@@ -3260,8 +3267,6 @@ class WanVideoSampler:
try:
print_memory(device)
- #torch.cuda.memory._dump_snapshot("wanvideowrapper_memory_dump.pt")
- #torch.cuda.memory._record_memory_history(enabled=None)
torch.cuda.reset_peak_memory_stats(device)
except:
pass
@@ -3380,6 +3385,7 @@ class WanVideoScheduler:
},
"optional": {
"sigmas": ("SIGMAS", ),
+ "enhance_hf": ("BOOLEAN", {"default": False, "tooltip": "Enhanced high-frequency denoising schedule"}),
},
"hidden": {
"unique_id": "UNIQUE_ID",
@@ -3392,9 +3398,9 @@ class WanVideoScheduler:
CATEGORY = "WanVideoWrapper"
EXPERIMENTAL = True
- def process(self, scheduler, steps, start_step, end_step, shift, unique_id, sigmas=None):
+ def process(self, scheduler, steps, start_step, end_step, shift, unique_id, sigmas=None, enhance_hf=False):
sample_scheduler, timesteps, start_idx, end_idx = get_scheduler(
- scheduler, steps, start_step, end_step, shift, device, sigmas=sigmas, log_timesteps=True)
+ scheduler, steps, start_step, end_step, shift, device, sigmas=sigmas, log_timesteps=True, enhance_hf=enhance_hf)
scheduler_dict = {
"sample_scheduler": sample_scheduler,
@@ -3479,6 +3485,7 @@ class WanVideoSchedulerv2(WanVideoScheduler):
},
"optional": {
"sigmas": ("SIGMAS", ),
+ "enhance_hf": ("BOOLEAN", {"default": False, "tooltip": "Enhanced high-frequency denoising schedule"}),
},
"hidden": {
"unique_id": "UNIQUE_ID",
diff --git a/wanvideo/modules/model.py b/wanvideo/modules/model.py
index c920440..89f180f 100644
--- a/wanvideo/modules/model.py
+++ b/wanvideo/modules/model.py
@@ -633,15 +633,15 @@ class WanT2VCrossAttention(WanSelfAttention):
def forward(self, x, context, grid_sizes=None, clip_embed=None, audio_proj=None, audio_scale=1.0,
num_latent_frames=21, nag_params={}, nag_context=None, rope_func="comfy",
inner_t=None, inner_c=None, cross_freqs=None,
- adapter_proj=None, adapter_attn_mask=None, ip_scale=1.0, orig_seq_len=None, lynx_x_ip=None, lynx_ip_scale=1.0, num_cond_latents=None, **kwargs):
+ adapter_proj=None, adapter_attn_mask=None, ip_scale=1.0, orig_seq_len=None, lynx_x_ip=None, lynx_ip_scale=1.0, longcat_num_cond_latents=None, **kwargs):
b, n, d = x.size(0), self.num_heads, self.head_dim
s = x.size(1)
# compute query
is_longcat = x.shape[-1] == 4096
if is_longcat:
- if num_cond_latents is not None and num_cond_latents > 0:
- num_cond_latents_thw = num_cond_latents * (s // num_latent_frames)
+ if longcat_num_cond_latents is not None and longcat_num_cond_latents > 0:
+ num_cond_latents_thw = longcat_num_cond_latents * (s // num_latent_frames)
x = x[:, num_cond_latents_thw:]
q = self.norm_q(self.q(x).view(b, -1, n, d))
else:
@@ -712,7 +712,7 @@ class WanT2VCrossAttention(WanSelfAttention):
x = x.add(target_x)
- if is_longcat and num_cond_latents is not None and num_cond_latents > 0:
+ if is_longcat and longcat_num_cond_latents > 0:
return torch.cat([torch.zeros((b, num_cond_latents_thw, x.shape[-1]), dtype=x.dtype, device=x.device), self.o(x)], dim=1).contiguous()
return self.o(x)
@@ -914,7 +914,7 @@ class WanAttentionBlock(nn.Module):
from ...LongCat.layers import FeedForwardSwiGLU
mlp_ratio = 4
self.ffn = FeedForwardSwiGLU(dim=self.dim, hidden_dim=int(self.dim * mlp_ratio))
-
+
# modulation
if not is_longcat:
self.modulation = nn.Parameter(torch.randn(1, 6, out_features) / in_features**0.5)
@@ -1003,7 +1003,7 @@ class WanAttentionBlock(nn.Module):
humo_audio_input=None, humo_audio_scale=1.0, #humo audio
lynx_x_ip=None, lynx_ref_feature=None, lynx_ip_scale=1.0, lynx_ref_scale=1.0, #lynx
x_ovi=None, e_ovi=None, freqs_ovi=None, context_ovi=None, seq_lens_ovi=None, grid_sizes_ovi=None,
- num_cond_latents=None, #longcat image cond amount
+ longcat_num_cond_latents=0, longcat_avatar_options=None, #longcat image cond amount
x_onetoall_ref=None, onetoall_freqs=None, onetoall_ref=None, onetoall_ref_scale=1.0, #one-to-all
e_tr=None, tr_num=0, tr_start=0, #token replacement
):
@@ -1015,6 +1015,11 @@ class WanAttentionBlock(nn.Module):
grid_sizes(Tensor): Shape [B, 3], the second dimension contains (F, H, W)
freqs(Tensor): Rope freqs, shape [1024, C / num_heads / 2]
"""
+ input_dtype = x.dtype
+ B, N, C = x.shape
+ T = num_latent_frames
+ is_longcat = C == 4096
+
zero_timestep = len(e) == 2
if zero_timestep: #s2v zero timestep
self.seg_idx = e[1]
@@ -1030,11 +1035,10 @@ class WanAttentionBlock(nn.Module):
tr_end = tr_start + (tr_num or 0)
shift_msa, scale_msa, gate_msa, shift_mlp, scale_mlp, gate_mlp = self.get_mod(e.to(x.device), self.modulation)
+ if multitalk_audio_embedding is not None and is_longcat:
+ audio_shift_mca, audio_scale_mca, audio_gate_mca = self.audio_modulation(e[:, longcat_num_cond_latents:]).unsqueeze(2).chunk(3, dim=-1)
del e
- input_dtype = x.dtype
- B, N, C = x.shape
- T = num_latent_frames
- is_longcat = C == 4096
+
if is_longcat:
input_x = self.modulate(self.norm1(x.view(B, T, -1, C).to(shift_msa.dtype)), shift_msa, scale_msa, seg_idx=self.seg_idx).to(input_dtype).view(B, N, C)
elif use_token_replace:
@@ -1181,22 +1185,67 @@ class WanAttentionBlock(nn.Module):
full_k = torch.cat([k, k_ip], dim=1)
full_v = torch.cat([v, v_ip], dim=1)
y = self.self_attn.forward(q, full_k, full_v, seq_lens)
- elif is_longcat and num_cond_latents is not None and num_cond_latents > 0:
- num_cond_latents_thw = num_cond_latents * (N // num_latent_frames)
- # process the condition tokens
- x_cond = self.self_attn.forward(
- q[:, :num_cond_latents_thw].contiguous(),
- k[:, :num_cond_latents_thw].contiguous(),
- v[:, :num_cond_latents_thw].contiguous(),
- seq_lens)
- # process the noise tokens
- x_noise = self.self_attn.forward(q[:, num_cond_latents_thw:].contiguous(), k, v, seq_lens)
- # merge x_cond and x_noise
- y = torch.cat([x_cond, x_noise], dim=1).contiguous()
+ elif is_longcat and longcat_num_cond_latents > 0:
+ if longcat_num_cond_latents == 1:
+ num_cond_latents_thw = longcat_num_cond_latents * (N // num_latent_frames)
+ # process the noise tokens
+ x_noise = self.self_attn.forward(q[:, num_cond_latents_thw:].contiguous(), k, v, seq_lens)
+ # process the condition tokens
+ x_cond = self.self_attn.forward(
+ q[:, :num_cond_latents_thw].contiguous(),
+ k[:, :num_cond_latents_thw].contiguous(),
+ v[:, :num_cond_latents_thw].contiguous(),
+ seq_lens)
+ # merge x_cond and x_noise
+ y = torch.cat([x_cond, x_noise], dim=1).contiguous()
+ elif longcat_num_cond_latents > 1: # video continuation
+ num_ref_latents_thw = (N // num_latent_frames)
+ num_cond_latents_thw = longcat_num_cond_latents * (N // num_latent_frames)
+ if not longcat_num_cond_latents == num_latent_frames:
+ # process the noise tokens
+ q_noise = q[:, num_cond_latents_thw:].contiguous()
+ start_noise, end_noise, num_noisy_frames = 0, 0, num_latent_frames - longcat_num_cond_latents
+ mask_frame_range = longcat_avatar_options["ref_mask_frame_range"]
+ ref_img_index = longcat_avatar_options["ref_frame_index"]
+ num_ref_latents = 1
+ if mask_frame_range is not None and mask_frame_range > 0:
+ start_noise = ref_img_index - mask_frame_range - longcat_num_cond_latents + num_ref_latents
+ end_noise = ref_img_index + mask_frame_range - longcat_num_cond_latents + num_ref_latents + 1
+
+ if start_noise >= 0 and end_noise > start_noise and end_noise <= num_noisy_frames:
+ # remove attention with the reference image in the target range, preventing repeated actions.
+
+ start_pos = start_noise * (N // num_latent_frames)
+ end_pos = end_noise * (N // num_latent_frames)
+
+ q_noise_front = q_noise[:, :start_pos].contiguous()
+ q_noise_maskref = q_noise[:, start_pos:end_pos].contiguous()
+ q_noise_back = q_noise[:, end_pos:].contiguous()
+ k_non_ref = k[:, num_ref_latents_thw:].contiguous()
+ v_non_ref = v[:, num_ref_latents_thw:].contiguous()
+
+ x_noise_front = self.self_attn.forward(q_noise_front, k, v, seq_lens) # q_front has attention with ref + cond + noisy
+ x_noise_back = self.self_attn.forward(q_noise_back, k, v, seq_lens) # q_back has attention with ref + cond + noisy
+ x_noise_maskref = self.self_attn.forward(q_noise_maskref, k_non_ref, v_non_ref, seq_lens) # q_mask has attention with cond+noisy
+ x_noise = torch.cat([x_noise_front, x_noise_maskref, x_noise_back], dim=1).contiguous()
+ else:
+ x_noise = self.self_attn.forward(q_noise, k, v, seq_lens)
+ # process the condition tokens
+ q_ref = q[:, :num_ref_latents_thw].contiguous()
+ k_ref = k[:, :num_ref_latents_thw].contiguous()
+ v_ref = v[:, :num_ref_latents_thw].contiguous()
+ q_cond = q[:, num_ref_latents_thw:num_cond_latents_thw].contiguous()
+ k_cond = k[:, num_ref_latents_thw:num_cond_latents_thw].contiguous()
+ v_cond = v[:, num_ref_latents_thw:num_cond_latents_thw].contiguous()
+ x_ref = self.self_attn.forward(q_ref, k_ref, v_ref, seq_lens)
+ x_cond = self.self_attn.forward(q_cond, k_cond, v_cond, seq_lens)
+
+ # merge x_cond and x_noise
+ y = torch.cat([x_ref, x_cond, x_noise], dim=1).contiguous()
else:
y = self.self_attn.forward(q, k, v, seq_lens, lynx_ref_feature=lynx_ref_feature, lynx_ref_scale=lynx_ref_scale, onetoall_ref=onetoall_ref, onetoall_ref_scale=onetoall_ref_scale)
- del q, k, v,
+ del q, k, v
# FETA
if enhance_enabled:
@@ -1263,12 +1312,21 @@ class WanAttentionBlock(nn.Module):
x = x + self.cross_attn(self.norm3(x.to(self.norm3.weight.dtype)).to(input_dtype), context, grid_sizes, clip_embed=clip_embed, audio_proj=audio_proj, audio_scale=audio_scale,
num_latent_frames=num_latent_frames, nag_params=nag_params, nag_context=nag_context,
rope_func=self.rope_func, inner_t=inner_t, inner_c=inner_c, cross_freqs=cross_freqs,
- adapter_proj=adapter_proj, ip_scale=ip_scale, orig_seq_len=original_seq_len, lynx_x_ip=lynx_x_ip, lynx_ip_scale=lynx_ip_scale, num_cond_latents=num_cond_latents)
+ adapter_proj=adapter_proj, ip_scale=ip_scale, orig_seq_len=original_seq_len, lynx_x_ip=lynx_x_ip, lynx_ip_scale=lynx_ip_scale, longcat_num_cond_latents=longcat_num_cond_latents)
x = x.to(input_dtype)
# MultiTalk
if multitalk_audio_embedding is not None and not isinstance(self, VaceWanAttentionBlock):
- x_audio = self.audio_cross_attn(self.norm_x(x.to(self.norm_x.weight.dtype)).to(input_dtype), encoder_hidden_states=multitalk_audio_embedding,
- shape=grid_sizes[0], x_ref_attn_map=x_ref_attn_map, human_num=human_num)
+
+ if is_longcat:
+ audio_output_cond, x_audio = self.audio_cross_attn(self.norm_x(x.to(self.norm_x.weight.dtype)).to(input_dtype), multitalk_audio_embedding, num_latent_frames=num_latent_frames,
+ num_cond_latents=longcat_num_cond_latents, x_ref_attn_map=x_ref_attn_map, human_num=human_num)
+ x_audio = self.modulate(self.norm1(x_audio.view(B, T-longcat_num_cond_latents, -1, C).to(audio_shift_mca.dtype)), audio_shift_mca, audio_scale_mca, seg_idx=self.seg_idx).to(input_dtype).view(B, -1, C)
+ x_audio = (x_audio.view(B, T-longcat_num_cond_latents, -1, C).float() * audio_gate_mca).to(input_dtype).view(B, -1, C)
+ if audio_output_cond is not None:
+ x_audio = torch.cat([audio_output_cond, x_audio], dim=1).contiguous()
+ else:
+ x_audio = self.audio_cross_attn(self.norm_x(x.to(self.norm_x.weight.dtype)).to(input_dtype), encoder_hidden_states=multitalk_audio_embedding,
+ shape=grid_sizes[0], x_ref_attn_map=x_ref_attn_map, human_num=human_num)
x = x.add(x_audio, alpha=audio_scale)
# MTV-Crafter Motion Attention
@@ -1282,7 +1340,7 @@ class WanAttentionBlock(nn.Module):
# ffn
- if self.rope_func == "comfy_chunked":
+ if self.rope_func == "comfy_chunked" and not is_longcat and not use_token_replace and not zero_timestep:
mod_x = torch.addcmul(shift_mlp, self.norm2(x.to(shift_mlp.dtype)), 1 + scale_mlp)
x_ffn = self.ffn_chunked(mod_x)
else:
@@ -1308,7 +1366,7 @@ class WanAttentionBlock(nn.Module):
mod_x = torch.addcmul(shift_mlp, self.norm2(x.to(shift_mlp.dtype)), 1 + scale_mlp)
del shift_mlp, scale_mlp
- x_ffn = self.ffn_chunked(mod_x.to(input_dtype), num_chunks=1)
+ x_ffn = self.ffn_chunked(mod_x.to(input_dtype), num_chunks=2 if is_longcat else 1)
del mod_x
# gate_mlp
@@ -2128,7 +2186,8 @@ class WanModel(torch.nn.Module):
def rope_encode_comfy(self, t, h, w, freq_offset=0, t_start=0, ref_frame_shape=None, pose_frame_shape=None,
- steps_t=None, steps_h=None, steps_w=None, ntk_alphas=[1,1,1], device=None, dtype=None):
+ steps_t=None, steps_h=None, steps_w=None, ntk_alphas=[1,1,1], device=None, dtype=None,
+ ref_frame_index=10, longcat_num_ref_latents=None):
patch_size = self.patch_size
t_len = ((t + (patch_size[0] // 2)) // patch_size[0])
@@ -2144,7 +2203,18 @@ class WanModel(torch.nn.Module):
# Main frames position IDs
img_ids = torch.zeros((steps_t, steps_h, steps_w, 3), device=device, dtype=dtype)
- img_ids[:, :, :, 0] = img_ids[:, :, :, 0] + torch.linspace(t_start+freq_offset, t_start+freq_offset + (t_len - 1), steps=steps_t, device=device, dtype=dtype).reshape(-1, 1, 1)
+
+ if longcat_num_ref_latents > 0:
+ # Create temporal grid with ref_frame_index prepended, followed by sequential frames
+ grid_t = torch.cat([
+ torch.tensor([ref_frame_index], dtype=dtype, device=device),
+ torch.arange(0, steps_t - longcat_num_ref_latents, dtype=dtype, device=device)
+ ], dim=0)
+ img_ids[:, :, :, 0] = img_ids[:, :, :, 0] + grid_t.reshape(-1, 1, 1)
+ else:
+ # Standard temporal encoding
+ img_ids[:, :, :, 0] = img_ids[:, :, :, 0] + torch.linspace(t_start+freq_offset, t_start+freq_offset + (t_len - 1), steps=steps_t, device=device, dtype=dtype).reshape(-1, 1, 1)
+
img_ids[:, :, :, 1] = img_ids[:, :, :, 1] + torch.linspace(freq_offset, freq_offset + (h_len - 1), steps=steps_h, device=device, dtype=dtype).reshape(1, -1, 1)
img_ids[:, :, :, 2] = img_ids[:, :, :, 2] + torch.linspace(freq_offset, freq_offset + (w_len - 1), steps=steps_w, device=device, dtype=dtype).reshape(1, 1, -1)
img_ids = img_ids.reshape(1, -1, img_ids.shape[-1])
@@ -2243,7 +2313,7 @@ class WanModel(torch.nn.Module):
lynx_embeds=None,
x_ovi=None, seq_len_ovi=None, ovi_negative_text_embeds=None,
flashvsr_LQ_latent=None, flashvsr_strength=1.0,
- num_cond_latents=None,
+ longcat_num_cond_latents=0, longcat_num_ref_latents=0, longcat_avatar_options=None, # for LongCat
add_text_emb=None,
sdancer_input=None, # SteadyDancer
one_to_all_input=None, one_to_all_controlnet_strength=0.0, # One-to-All
@@ -2559,6 +2629,7 @@ class WanModel(torch.nn.Module):
tuple(pose_frame_shape) if pose_frame_shape is not None else None,
self.rope_embedder.k,
tuple(ntk_alphas),
+ longcat_num_ref_latents,
)
# Check cache using key comparison
@@ -2567,16 +2638,17 @@ class WanModel(torch.nn.Module):
self.cached_key == cache_key):
freqs = self.cached_freqs
else:
- log.info("Generating new RoPE frequencies")
freqs = self.rope_encode_comfy(
F, H, W,
freq_offset=freq_offset,
ntk_alphas=ntk_alphas,
ref_frame_shape=ref_frame_shape,
pose_frame_shape=pose_frame_shape,
+ longcat_num_ref_latents=longcat_num_ref_latents,
device=x.device,
dtype=x.dtype
)
+ log.info("Generated new RoPE frequencies")
if s2v_ref_latent is not None:
freqs_ref = self.rope_encode_comfy(
@@ -2641,8 +2713,8 @@ class WanModel(torch.nn.Module):
if len(t.shape) == 1:
t = t.unsqueeze(1).expand(-1, F) # [B, T]
self.time_embedding.to(torch.float32)
- e = e0 = self.time_embedding(t.float().flatten(), dtype=torch.float32).reshape(1, F, -1)
-
+ e = e0 = self.time_embedding(t.float().flatten(), dtype=torch.float32)#.reshape(1, F, -1)
+ e = e0 = e0.reshape(1, F, -1)
if self.audio_model is not None:
#if t.dim() == 1:
@@ -2777,10 +2849,24 @@ class WanModel(torch.nn.Module):
latter_middle_frame_audio_emb = latter_frame_audio_emb[:, :, 1:-1, middle_index:middle_index+1, ...]
latter_middle_frame_audio_emb = rearrange(latter_middle_frame_audio_emb, "b n_t n w s c -> b n_t (n w) s c")
latter_frame_audio_emb_s = torch.concat([latter_first_frame_audio_emb, latter_middle_frame_audio_emb, latter_last_frame_audio_emb], dim=2)
- multitalk_audio_embedding = self.multitalk_audio_proj(first_frame_audio_emb_s, latter_frame_audio_emb_s)
- human_num = len(multitalk_audio_embedding)
- multitalk_audio_embedding = torch.concat(multitalk_audio_embedding.split(1), dim=2).to(self.base_dtype)
+ multitalk_audio_embedding = self.multitalk_audio_proj(first_frame_audio_emb_s, latter_frame_audio_emb_s)
self.multitalk_audio_proj.to(self.offload_device)
+ human_num = len(multitalk_audio_embedding)
+
+ # LongCat-Avatar specific
+ if longcat_num_ref_latents > 0:
+ audio_start_ref = multitalk_audio_embedding[:, [0], :, :] # padding
+ multitalk_audio_embedding = torch.cat([audio_start_ref, multitalk_audio_embedding], dim=1).contiguous()
+
+ if longcat_num_cond_latents > 0:
+ multitalk_audio_embedding = multitalk_audio_embedding[:, (-F // self.patch_size[0]):]
+
+ if ref_target_masks is not None:
+ multitalk_audio_embedding = torch.concat(multitalk_audio_embedding.split(1), dim=2).to(self.base_dtype)
+ multitalk_audio_embedding = multitalk_audio_embedding.squeeze(0)
+ else:
+ multitalk_audio_embedding = rearrange(multitalk_audio_embedding, "b t n c -> (b t) n c")
+
# convert ref_target_masks to token_ref_target_masks
token_ref_target_masks = None
@@ -2974,7 +3060,8 @@ class WanModel(torch.nn.Module):
lynx_x_ip=lynx_x_ip,
lynx_ip_scale=lynx_ip_scale,
lynx_ref_scale=lynx_ref_scale,
- num_cond_latents=num_cond_latents,
+ longcat_num_cond_latents=longcat_num_cond_latents,
+ longcat_avatar_options=longcat_avatar_options,
onetoall_ref_scale=onetoall_ref_scale,
e_tr=e0_token_replace if use_token_replace else None,
tr_start=token_replace_start,
diff --git a/wanvideo/schedulers/__init__.py b/wanvideo/schedulers/__init__.py
index e39e7e4..abb450f 100644
--- a/wanvideo/schedulers/__init__.py
+++ b/wanvideo/schedulers/__init__.py
@@ -1,9 +1,11 @@
import torch
+import numpy as np
from .fm_solvers import (FlowDPMSolverMultistepScheduler)
from .fm_solvers_unipc import FlowUniPCMultistepScheduler
from .basic_flowmatch import FlowMatchScheduler
from .flowmatch_pusa import FlowMatchSchedulerPusa
from .flowmatch_res_multistep import FlowMatchSchedulerResMultistep
+from .ersde_scheduler import ERSDEScheduler
from .scheduling_flow_match_lcm import FlowMatchLCMScheduler
from .fm_sa_ode import FlowMatchSAODEStableScheduler
from .fm_rcm import rCMFlowMatchScheduler
@@ -25,6 +27,7 @@ scheduler_list = [
"deis",
"lcm", "lcm/beta",
"res_multistep",
+ "er_sde",
"flowmatch_causvid",
"flowmatch_distill",
"flowmatch_pusa",
@@ -39,7 +42,7 @@ def _apply_custom_sigmas(sample_scheduler, sigmas, device):
sample_scheduler.timesteps = (sample_scheduler.sigmas[:-1] * 1000).to(torch.int64).to(device)
sample_scheduler.num_inference_steps = len(sample_scheduler.timesteps)
-def get_scheduler(scheduler, steps, start_step, end_step, shift, device, transformer_dim=5120, flowedit_args=None, denoise_strength=1.0, sigmas=None, log_timesteps=False, **kwargs):
+def get_scheduler(scheduler, steps, start_step, end_step, shift, device, transformer_dim=5120, flowedit_args=None, denoise_strength=1.0, sigmas=None, log_timesteps=False, enhance_hf=False, **kwargs):
timesteps = None
if sigmas is not None:
steps = len(sigmas) - 1
@@ -136,6 +139,12 @@ def get_scheduler(scheduler, steps, start_step, end_step, shift, device, transfo
sample_scheduler.set_timesteps(steps, denoising_strength=denoise_strength)
else:
_apply_custom_sigmas(sample_scheduler, sigmas, device)
+ elif scheduler == 'er_sde':
+ sample_scheduler = ERSDEScheduler(shift=shift)
+ if sigmas is None:
+ sample_scheduler.set_timesteps(steps, denoising_strength=denoise_strength)
+ else:
+ _apply_custom_sigmas(sample_scheduler, sigmas, device)
elif "sa_ode_stable" in scheduler:
sample_scheduler = FlowMatchSAODEStableScheduler(shift=shift, **kwargs)
if sigmas is None:
@@ -152,6 +161,18 @@ def get_scheduler(scheduler, steps, start_step, end_step, shift, device, transfo
if timesteps is None:
timesteps = sample_scheduler.timesteps
+ if enhance_hf:
+ num_tail_uniform_steps = max(3, min(15, int(len(timesteps) * 0.2))) # Use 20% of steps for uniform tail (minimum 3, maximum 15)
+ tail_uniform_start = float(timesteps.max()) * 0.5 # Split at 50% of the timestep range
+ tail_uniform_end = 0
+
+ timesteps_uniform_tail = list(np.linspace(tail_uniform_start, tail_uniform_end, num_tail_uniform_steps, dtype=np.float32, endpoint=(tail_uniform_end != 0)))
+ timesteps_uniform_tail = [torch.tensor(t, device=device).unsqueeze(0) for t in timesteps_uniform_tail]
+ filtered_timesteps = [timestep.unsqueeze(0).to(device) for timestep in timesteps if timestep > tail_uniform_start]
+ timesteps = torch.cat(filtered_timesteps + timesteps_uniform_tail)
+ sample_scheduler.timesteps = timesteps
+ sample_scheduler.sigmas = torch.cat([timesteps / 1000, torch.zeros(1, device=timesteps.device)])
+
steps = len(timesteps)
if (isinstance(start_step, int) and end_step != -1 and start_step >= end_step) or (not isinstance(start_step, int) and start_step != -1 and end_step >= start_step):
raise ValueError("start_step must be less than end_step")
diff --git a/wanvideo/schedulers/ersde_scheduler.py b/wanvideo/schedulers/ersde_scheduler.py
new file mode 100644
index 0000000..6abd7a7
--- /dev/null
+++ b/wanvideo/schedulers/ersde_scheduler.py
@@ -0,0 +1,154 @@
+import torch
+
+class ERSDEScheduler():
+ """Extended Reverse-Time SDE solver (VP ER-SDE-Solver-3).
+
+ Based on: arXiv: https://arxiv.org/abs/2309.06169
+ Code reference: https://github.com/QinpengCui/ER-SDE-Solver/blob/main/er_sde_solver.py
+ """
+
+ def __init__(self, num_inference_steps=100, num_train_timesteps=1000, shift=3.0,
+ sigma_max=1.0, sigma_min=0.003 / 1.002, max_stage=3, s_noise=1.0,
+ num_integration_points=200):
+ self.num_train_timesteps = num_train_timesteps
+ self.shift = shift
+ self.sigma_max = sigma_max
+ self.sigma_min = sigma_min
+ self.max_stage = max_stage
+ self.s_noise = s_noise
+ self.num_integration_points = num_integration_points
+ self.set_timesteps(num_inference_steps)
+ self.old_denoised = None
+ self.old_denoised_d = None
+ self.step_index = 0
+
+ def set_timesteps(self, num_inference_steps=100, denoising_strength=1.0, sigmas=None):
+ """Generate the full sigma schedule (from max to min)."""
+ full_sigmas = torch.linspace(self.sigma_max, self.sigma_min, self.num_train_timesteps)
+ ss = len(full_sigmas) / num_inference_steps
+ if sigmas is None:
+ sigmas = []
+ for x in range(num_inference_steps):
+ idx = int(round(x * ss))
+ sigmas.append(float(full_sigmas[idx]))
+ sigmas.append(0.0)
+ self.sigmas = torch.FloatTensor(sigmas)
+ self.sigmas = self.shift * self.sigmas / (1 + (self.shift - 1) * self.sigmas)
+ self.timesteps = self.sigmas * self.num_train_timesteps
+ self.step_index = 0
+ self.old_denoised = None
+ self.old_denoised_d = None
+
+ def default_er_sde_noise_scaler(self, x):
+ return x * ((x ** 0.3).exp() + 10.0)
+
+ def step(self, model_output, timestep, sample, generator):
+
+ if timestep.ndim == 2:
+ timestep = timestep.flatten(0, 1)
+
+ self.sigmas = self.sigmas.to(model_output.device)
+ self.timesteps = self.timesteps.to(model_output.device)
+
+ if timestep.ndim == 0:
+ timestep_id = torch.argmin((self.timesteps - timestep).abs(), dim=0)
+ else:
+ timestep_id = torch.argmin((self.timesteps.unsqueeze(0) - timestep.unsqueeze(1)).abs(), dim=1)
+
+ noise_scaler = self.default_er_sde_noise_scaler
+
+ # Get current and next sigma
+ sigma = self.sigmas[timestep_id].reshape(-1, 1, 1, 1)
+ if (timestep_id + 1 >= len(self.sigmas)).any():
+ sigma_next = torch.zeros_like(sigma)
+ else:
+ sigma_next = self.sigmas[timestep_id + 1].reshape(-1, 1, 1, 1)
+
+ er_lambda_s = sigma
+ er_lambda_t = sigma_next
+
+ # Calculate alpha values
+ alpha_s = sigma / (er_lambda_s + 1e-10)
+ alpha_t = sigma_next / (er_lambda_t + 1e-10)
+ r_alpha = alpha_t / (alpha_s + 1e-10)
+
+ # Denoised prediction (x_0 estimate)
+ denoised = sample - sigma * model_output
+
+ # Determine which stage to use
+ stage_used = min(self.max_stage, self.step_index + 1)
+
+ if sigma_next == 0 or (sigma_next == 0.0).all():
+ # Final step - return denoised
+ x = denoised
+ else:
+ r = noise_scaler(er_lambda_t) / (noise_scaler(er_lambda_s) + 1e-10)
+
+ # Stage 1: Euler step
+ x = r_alpha * r * sample + alpha_t * (1 - r) * denoised
+
+ if stage_used >= 2 and self.old_denoised is not None:
+ dt = er_lambda_t - er_lambda_s
+ lambda_step_size = -dt / self.num_integration_points
+
+ # Create integration points
+ point_indice = torch.arange(0, self.num_integration_points,
+ dtype=torch.float32, device=sample.device)
+ lambda_pos = er_lambda_t + point_indice * lambda_step_size
+ scaled_pos = noise_scaler(lambda_pos)
+
+ # Stage 2: Second-order correction
+ s = torch.sum(1 / (scaled_pos + 1e-10)) * lambda_step_size
+
+ # Get previous sigma for derivative calculation
+ if timestep_id > 0:
+ sigma_prev = self.sigmas[timestep_id - 1].reshape(-1, 1, 1, 1)
+ er_lambda_prev = sigma_prev
+ else:
+ er_lambda_prev = er_lambda_s
+
+ denoised_d = (denoised - self.old_denoised) / ((er_lambda_s - er_lambda_prev) + 1e-10)
+ x = x + alpha_t * (dt + s * noise_scaler(er_lambda_t)) * denoised_d
+
+ if stage_used >= 3 and self.old_denoised_d is not None:
+ # Stage 3: Third-order correction
+ s_u = torch.sum((lambda_pos - er_lambda_s) / (scaled_pos + 1e-10)) * lambda_step_size
+
+ # Get sigma from two steps ago
+ if timestep_id > 1:
+ sigma_prev_prev = self.sigmas[timestep_id - 2].reshape(-1, 1, 1, 1)
+ er_lambda_prev_prev = sigma_prev_prev
+ else:
+ er_lambda_prev_prev = er_lambda_prev
+
+ denoised_u = (denoised_d - self.old_denoised_d) / (((er_lambda_s - er_lambda_prev_prev) / 2) + 1e-10)
+ x = x + alpha_t * ((dt ** 2) / 2 + s_u * noise_scaler(er_lambda_t)) * denoised_u
+
+ self.old_denoised_d = denoised_d
+
+ # Add stochastic noise
+ if self.s_noise > 0:
+ noise_term = (er_lambda_t ** 2 - er_lambda_s ** 2 * r ** 2).sqrt()
+ noise_term = torch.nan_to_num(noise_term, nan=0.0)
+ noise = torch.randn(*x.shape, dtype=torch.float32, device=torch.device("cpu"), generator=generator).to(x)
+ x = x + alpha_t * noise * self.s_noise * noise_term
+
+ # Store current denoised for next iteration
+ self.old_denoised = denoised
+ self.step_index += 1
+
+ return x
+
+ def add_noise(self, original_samples, noise, timestep):
+ if timestep.ndim == 2:
+ timestep = timestep.flatten(0, 1)
+
+ self.sigmas = self.sigmas.to(noise.device)
+ self.timesteps = self.timesteps.to(noise.device)
+
+ timestep_id = torch.argmin(
+ (self.timesteps.unsqueeze(0) - timestep.unsqueeze(1)).abs(), dim=1)
+ sigma = self.sigmas[timestep_id].reshape(-1, 1, 1, 1)
+
+ sample = (1 - sigma) * original_samples + sigma * noise
+ return sample.type_as(noise)