From d00abe52d720a5845343ad0dd2927275bad399e7 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Wed, 28 Jan 2026 02:09:50 +0200 Subject: [PATCH 1/9] version 1.4.6 --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 5629a50..1156788 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "ComfyUI-WanVideoWrapper" description = "ComfyUI wrapper nodes for WanVideo" -version = "1.4.5" +version = "1.4.6" license = {file = "LICENSE"} dependencies = ["accelerate >= 1.2.1", "diffusers >= 0.33.0", "peft >= 0.17.0", "ftfy", "gguf >= 0.17.1", "pyloudnorm"] From 2c5a04cc6332f56f28e9eb4331d6163161af3770 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Fri, 30 Jan 2026 18:59:41 +0200 Subject: [PATCH 2/9] Init image_cond_mask --- nodes_sampler.py | 1 + 1 file changed, 1 insertion(+) diff --git a/nodes_sampler.py b/nodes_sampler.py index 0e6b7a8..93c3fd6 100644 --- a/nodes_sampler.py +++ b/nodes_sampler.py @@ -225,6 +225,7 @@ class WanVideoSampler: #I2V story_mem_latents = image_embeds.get("story_mem_latents", None) image_cond = image_embeds.get("image_embeds", None) + image_cond_mask = None if image_cond is not None: if transformer.in_dim == 16: raise ValueError("T2V (text to video) model detected, encoded images only work with I2V (Image to video) models") From e21fe20a4d2bc67370426be0e533a0e56c9f5cb6 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Sat, 31 Jan 2026 23:24:29 +0200 Subject: [PATCH 3/9] Support SkyReels TalkingAvatar (A2V) --- ...V_SkyReelsV3_TalkingAvatar_example_01.json | 3835 +++++++++++++++++ multitalk/multitalk.py | 12 +- multitalk/multitalk_loop.py | 144 +- multitalk/nodes.py | 90 +- nodes_model_loading.py | 90 +- nodes_sampler.py | 11 +- utils.py | 52 + 7 files changed, 4145 insertions(+), 89 deletions(-) create mode 100644 example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json diff --git a/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json b/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json new file mode 100644 index 0000000..4350292 --- /dev/null +++ b/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json @@ -0,0 +1,3835 @@ +{ + "id": "8b7a9a57-2303-4ef5-9fc2-bf41713bd1fc", + "revision": 0, + "last_node_id": 349, + "last_link_id": 614, + "nodes": [ + { + "id": 129, + "type": "WanVideoVAELoader", + "pos": [ + 1379.0877685546875, + -3038.031494140625 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + } + ], + "outputs": [ + { + "name": "vae", + "type": "WANVAE", + "slot_index": 0, + "links": [ + 436 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "c3ee35f3ece76e38099dc516182d69b406e16772", + "Node name for S&R": "WanVideoVAELoader" + }, + "widgets_values": [ + "wanvideo\\Wan2_1_VAE_bf16.safetensors", + "bf16", + false, + false + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 240, + "type": "SetNode", + "pos": [ + 1761.82666015625, + -3007.7861328125 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 42, + "mode": 0, + "inputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "link": 436 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_VAE", + "properties": { + "previousName": "VAE" + }, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 247, + "type": "SetNode", + "pos": [ + 2301.934352826879, + -2373.917964797528 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 48, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 441 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_width", + "properties": { + "previousName": "width" + }, + "widgets_values": [ + "width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 248, + "type": "SetNode", + "pos": [ + 2300.428005170629, + -2197.305904250653 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 49, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 442 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_height", + "properties": { + "previousName": "height" + }, + "widgets_values": [ + "height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 282, + "type": "GetNode", + "pos": [ + 1036.8980712890625, + -1798.782958984375 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 495 + ] + } + ], + "title": "Get_height", + "properties": {}, + "widgets_values": [ + "height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 294, + "type": "SetNode", + "pos": [ + 1676.046630859375, + -1182.43310546875 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 62, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 519 + } + ], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 520 + ] + } + ], + "title": "Set_actual_audio_frames", + "properties": { + "previousName": "actual_audio_frames" + }, + "widgets_values": [ + "actual_audio_frames" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 293, + "type": "PreviewAny", + "pos": [ + 1696.14794921875, + -1108.7257080078125 + ], + "size": [ + 210, + 112 + ], + "flags": {}, + "order": 64, + "mode": 0, + "inputs": [ + { + "name": "source", + "type": "*", + "link": 520 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.50", + "Node name for S&R": "PreviewAny" + }, + "widgets_values": [ + null, + null, + null + ] + }, + { + "id": 271, + "type": "SetNode", + "pos": [ + 2464.989284467504, + -2206.392574172528 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 47, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 472 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_max_frames", + "properties": { + "previousName": "max_frames" + }, + "widgets_values": [ + "max_frames" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 263, + "type": "Note", + "pos": [ + 1643.58642578125, + -963.624755859375 + ], + "size": [ + 344.82928466796875, + 143.7332305908203 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "increase audio_scale for stronger effect\n\naudio_cfg is only used when sampler is using cfg as well\n\nnum_frames is the maximum frame count to generate, if the value is higher than your audio length would be, the audio length is used " + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 272, + "type": "GetNode", + "pos": [ + 1098.8834228515625, + -1057.341064453125 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 529 + ] + } + ], + "title": "Get_max_frames", + "properties": {}, + "widgets_values": [ + "max_frames" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 283, + "type": "GetNode", + "pos": [ + 1039.0777587890625, + -1849.6192626953125 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 494 + ] + } + ], + "title": "Get_width", + "properties": {}, + "widgets_values": [ + "width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 284, + "type": "LoadImage", + "pos": [ + 634.8666381835938, + -1939.31591796875 + ], + "size": [ + 274.080078125, + 314 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 496 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.50", + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "woman.jpg", + "image" + ] + }, + { + "id": 299, + "type": "Note", + "pos": [ + 1209.535120064439, + -2243.6752374861803 + ], + "size": [ + 290.9361267089844, + 88 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Clip vision is not strictly necessary\n\nAny I2V model should work, MAGREF can be interesting to play with as well." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 253, + "type": "SetNode", + "pos": [ + 909.3299560546875, + -1326.90576171875 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 44, + "mode": 0, + "inputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "link": 542 + } + ], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 545 + ] + } + ], + "title": "Set_input_audio", + "properties": { + "previousName": "input_audio" + }, + "widgets_values": [ + "input_audio" + ] + }, + { + "id": 137, + "type": "DownloadAndLoadWav2VecModel", + "pos": [ + 453.353515625, + -1010.69482421875 + ], + "size": [ + 330.96728515625, + 106 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "wav2vec_model", + "type": "WAV2VECMODEL", + "links": [ + 334 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0", + "Node name for S&R": "DownloadAndLoadWav2VecModel" + }, + "widgets_values": [ + "TencentGameMate/chinese-wav2vec2-base", + "fp16", + "main_device" + ] + }, + { + "id": 300, + "type": "Wav2VecModelLoader", + "pos": [ + 438.4236145019531, + -846.713623046875 + ], + "size": [ + 346.9834289550781, + 106 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "wav2vec_model", + "type": "WAV2VECMODEL", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "6fce0e2d3bb976b0006bc6d8e37e1f23460938ee", + "Node name for S&R": "Wav2VecModelLoader" + }, + "widgets_values": [ + "wav2vec2-chinese-base_fp16.safetensors", + "fp16", + "main_device" + ] + }, + { + "id": 302, + "type": "MelBandRoFormerSampler", + "pos": [ + 1110.198486328125, + -1350.2908935546875 + ], + "size": [ + 231.2720703125, + 46 + ], + "flags": {}, + "order": 55, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MELROFORMERMODEL", + "link": 543 + }, + { + "name": "audio", + "type": "AUDIO", + "link": 545 + } + ], + "outputs": [ + { + "name": "vocals", + "type": "AUDIO", + "links": [ + 544 + ] + }, + { + "name": "instruments", + "type": "AUDIO", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-MelBandRoFormer", + "ver": "b68d9077815387b64d596f8c39607052b95b6eba", + "Node name for S&R": "MelBandRoFormerSampler" + }, + "widgets_values": [] + }, + { + "id": 303, + "type": "MarkdownNote", + "pos": [ + 875.717529296875, + -854.7509155273438 + ], + "size": [ + 355.3899841308594, + 125.51774597167969 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Wav2vec2 safetensors", + "properties": {}, + "widgets_values": [ + "Alternative to the download node is to use single .safetensors file from:\n\n[https://huggingface.co/Kijai/wav2vec2_safetensors/tree/main](https://huggingface.co/Kijai/wav2vec2_safetensors/tree/main)\n\nThe model goes to `ComfyUI/models/wav2vec2`" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 304, + "type": "MarkdownNote", + "pos": [ + 461.80609130859375, + -1254.9613037109375 + ], + "size": [ + 345.2374267578125, + 91.6759033203125 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Wav2vec2 safetensors", + "properties": {}, + "widgets_values": [ + "Vocal separator model\n\n[https://huggingface.co/Kijai/MelBandRoFormer_comfy/tree/main](https://huggingface.co/Kijai/MelBandRoFormer_comfy/tree/main)\n\nThe model goes to `ComfyUI/models/diffusion_models`" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 301, + "type": "MelBandRoFormerModelLoader", + "pos": [ + 454.7294006347656, + -1117.843994140625 + ], + "size": [ + 400.5706481933594, + 58 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MELROFORMERMODEL", + "links": [ + 543 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-MelBandRoFormer", + "ver": "b68d9077815387b64d596f8c39607052b95b6eba", + "Node name for S&R": "MelBandRoFormerModelLoader" + }, + "widgets_values": [ + "MelBandRoFormer\\MelBandRoformer_fp16.safetensors" + ] + }, + { + "id": 310, + "type": "Note", + "pos": [ + 3483.0307547147418, + -643.371899655399 + ], + "size": [ + 340.64837646484375, + 183.92144775390625 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "This node makes the sampling into a loop to process all the needed frames. Each window is the same size, if the frame count doesn't divide evenly the audio will be padded with silence and extra frames generated.\n\nBetween each window the frames have to be re-encoded, thus no latents are saved and end result is already decoded.\n\nYou can also set an output_path to save each window's result as images during the process." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 125, + "type": "LoadAudio", + "pos": [ + 453.9859313964844, + -1425.9039306640625 + ], + "size": [ + 404.9698791503906, + 136 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 542 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.41", + "Node name for S&R": "LoadAudio" + }, + "widgets_values": [ + "woman_speech.mp3", + null, + null + ] + }, + { + "id": 254, + "type": "GetNode", + "pos": [ + 5164.1694521769205, + -1323.1747312290904 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 14, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 449 + ] + } + ], + "title": "Get_input_audio", + "properties": {}, + "widgets_values": [ + "input_audio" + ], + "color": "#323", + "bgcolor": "#535" + }, + { + "id": 244, + "type": "GetNode", + "pos": [ + 3518.147358651261, + -1141.4585574004934 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 15, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 554 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 321, + "type": "WanVideoDecode", + "pos": [ + 4372.743072913519, + -2160.453351448422 + ], + "size": [ + 270, + 198 + ], + "flags": {}, + "order": 65, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 575 + }, + { + "name": "samples", + "type": "LATENT", + "link": 576 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 577, + 586 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 327, + "type": "GetNode", + "pos": [ + 2865.1261295050163, + -2739.0457728978404 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 16, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 581 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 261, + "type": "GetNode", + "pos": [ + 3451.1101237602147, + -2342.3152871019543 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 17, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 569 + ] + } + ], + "title": "Get_wanmodel", + "properties": {}, + "widgets_values": [ + "wanmodel" + ] + }, + { + "id": 320, + "type": "WanVideoSamplerv2", + "pos": [ + 3588.907432893376, + -2150.424837725686 + ], + "size": [ + 295.37109375, + 478.57789522058823 + ], + "flags": {}, + "order": 63, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 569 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 580 + }, + { + "name": "scheduler", + "type": "WANVIDEOSCHEDULER", + "link": 588 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 572 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "extra_args", + "shape": 7, + "type": "WANVIDSAMPLEREXTRAARGS", + "link": 607 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 576 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSamplerv2" + }, + "widgets_values": [ + 1, + 42, + "fixed", + true, + false + ] + }, + { + "id": 329, + "type": "Note", + "pos": [ + 3399.811668121435, + -2817.225543033241 + ], + "size": [ + 286.291473419416, + 88 + ], + "flags": {}, + "order": 18, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "First we generate normal I2V to get frames to use as reference for the audio part" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 328, + "type": "VHS_VideoCombine", + "pos": [ + 4759.630440472775, + -2700.899077815229 + ], + "size": [ + 555.5285458590442, + 334 + ], + "flags": {}, + "order": 67, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 586 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 25, + "loop_count": 0, + "filename_prefix": "WanVideo2_1_InfiniteTalk", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo2_1_InfiniteTalk_00004.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 25, + "workflow": "WanVideo2_1_InfiniteTalk_00004.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_InfiniteTalk_00004.mp4" + } + } + } + }, + { + "id": 322, + "type": "GetNode", + "pos": [ + 4373.832860059409, + -2210.595785335668 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 19, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 575 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 330, + "type": "SetNode", + "pos": [ + 2536.549141667446, + -2050.930477545556 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 45, + "mode": 0, + "inputs": [ + { + "name": "WANVIDEOSCHEDULER", + "type": "WANVIDEOSCHEDULER", + "link": 587 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_scheduler", + "properties": { + "previousName": "scheduler" + }, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 331, + "type": "GetNode", + "pos": [ + 3593.07408684267, + -2203.6913804983656 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 20, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOSCHEDULER", + "type": "WANVIDEOSCHEDULER", + "links": [ + 588 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 317, + "type": "WanVideoSamplerv2", + "pos": [ + 4616.34947527915, + -1078.5985054729583 + ], + "size": [ + 295.37109375, + 478.57789522058823 + ], + "flags": {}, + "order": 68, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 579 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 561 + }, + { + "name": "scheduler", + "type": "WANVIDEOSCHEDULER", + "link": 589 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 578 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "extra_args", + "shape": 7, + "type": "WANVIDSAMPLEREXTRAARGS", + "link": 567 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 563 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSamplerv2" + }, + "widgets_values": [ + 1, + 42, + "fixed", + true, + false + ] + }, + { + "id": 324, + "type": "GetNode", + "pos": [ + 4612.87150554424, + -1132.7470262951058 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 21, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 579 + ] + } + ], + "title": "Get_wanmodel", + "properties": {}, + "widgets_values": [ + "wanmodel" + ] + }, + { + "id": 332, + "type": "GetNode", + "pos": [ + 4611.425495704412, + -1182.7435520831307 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 22, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOSCHEDULER", + "type": "WANVIDEOSCHEDULER", + "links": [ + 589 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 318, + "type": "WanVideoSchedulerv2", + "pos": [ + 2204.3924969534974, + -2116.8787371046265 + ], + "size": [ + 296.1612785322468, + 391.5079908265427 + ], + "flags": {}, + "order": 23, + "mode": 0, + "inputs": [ + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + } + ], + "outputs": [ + { + "name": "scheduler", + "type": "WANVIDEOSCHEDULER", + "links": [ + 587 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSchedulerv2" + }, + "widgets_values": [ + "flowmatch_distill", + 4, + 5, + 0, + -1, + false + ] + }, + { + "id": 333, + "type": "SetNode", + "pos": [ + 1521.7838059469866, + -1887.3118378345364 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 53, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 590 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_gen_width", + "properties": { + "previousName": "gen_width" + }, + "widgets_values": [ + "gen_width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 334, + "type": "SetNode", + "pos": [ + 1525.381091522264, + -1836.9514864370556 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 54, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 591 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_gen_height", + "properties": { + "previousName": "gen_height" + }, + "widgets_values": [ + "gen_height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 335, + "type": "GetNode", + "pos": [ + 2847.5230307912925, + -2522.4959217305127 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 24, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 592 + ] + } + ], + "title": "Get_gen_width", + "properties": {}, + "widgets_values": [ + "gen_width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 336, + "type": "GetNode", + "pos": [ + 2852.3131604497885, + -2469.5211481898514 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 25, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 593 + ] + } + ], + "title": "Get_gen_height", + "properties": {}, + "widgets_values": [ + "gen_height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 337, + "type": "GetNode", + "pos": [ + 3323.3406026465395, + -959.4317258310826 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 26, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 594 + ] + } + ], + "title": "Get_gen_width", + "properties": {}, + "widgets_values": [ + "gen_width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 338, + "type": "GetNode", + "pos": [ + 3328.1307323050355, + -906.4569522904214 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 27, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 595 + ] + } + ], + "title": "Get_gen_height", + "properties": {}, + "widgets_values": [ + "gen_height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 265, + "type": "GetNode", + "pos": [ + 2258.578655088333, + -1616.3396762172404 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 28, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 467 + ] + } + ], + "title": "Get_clip_vision_model", + "properties": {}, + "widgets_values": [ + "clip_vision_model" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 281, + "type": "ImageResizeKJv2", + "pos": [ + 1207.9007568359375, + -1937.6312255859375 + ], + "size": [ + 270, + 336 + ], + "flags": {}, + "order": 43, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 496 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 494 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 495 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 596, + 598 + ] + }, + { + "name": "width", + "type": "INT", + "links": [ + 590 + ] + }, + { + "name": "height", + "type": "INT", + "links": [ + 591 + ] + }, + { + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "f7eb33abc80a2aded1b46dff0dd14d07856a7d50", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 832, + 480, + "lanczos", + "crop", + "0, 0, 0", + "top", + 16, + "cpu" + ] + }, + { + "id": 339, + "type": "SetNode", + "pos": [ + 1523.1228522480642, + -1935.4511901013743 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 52, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 598 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_start_image", + "properties": { + "previousName": "start_image" + }, + "widgets_values": [ + "start_image" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 340, + "type": "GetNode", + "pos": [ + 3322.701739101817, + -1043.5894219565967 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 29, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 599 + ] + } + ], + "title": "Get_start_image", + "properties": {}, + "widgets_values": [ + "start_image" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 326, + "type": "WanVideoImageToVideoEncode", + "pos": [ + 3027.753741759624, + -2687.8185084072566 + ], + "size": [ + 331.822265625, + 434 + ], + "flags": {}, + "order": 57, + "mode": 0, + "inputs": [ + { + "name": "vae", + "shape": 7, + "type": "WANVAE", + "link": 581 + }, + { + "name": "clip_embeds", + "shape": 7, + "type": "WANVIDIMAGE_CLIPEMBEDS", + "link": 582 + }, + { + "name": "start_image", + "shape": 7, + "type": "IMAGE", + "link": 600 + }, + { + "name": "end_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "control_embeds", + "shape": 7, + "type": "WANVIDIMAGE_EMBEDS", + "link": null + }, + { + "name": "temporal_mask", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "extra_latents", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "add_cond_latents", + "shape": 7, + "type": "ADD_COND_LATENTS", + "link": null + }, + { + "name": "empty_frame_pad_image", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 592 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 593 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 580 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoImageToVideoEncode" + }, + "widgets_values": [ + 832, + 480, + 81, + 0, + 1, + 1, + true, + true, + false, + 0 + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 341, + "type": "GetNode", + "pos": [ + 3036.3150762585738, + -2737.1358843947623 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 30, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 600 + ] + } + ], + "title": "Get_start_image", + "properties": {}, + "widgets_values": [ + "start_image" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 237, + "type": "WanVideoClipVisionEncode", + "pos": [ + 2271.749465310757, + -1562.7251353577965 + ], + "size": [ + 280.9771423339844, + 262 + ], + "flags": {}, + "order": 51, + "mode": 0, + "inputs": [ + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": 467 + }, + { + "name": "image_1", + "type": "IMAGE", + "link": 596 + }, + { + "name": "image_2", + "shape": 7, + "type": "IMAGE", + "link": null + }, + { + "name": "negative_image", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_CLIPEMBEDS", + "links": [ + 557, + 582 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "ff779c91714d8ee3484cd4119b082c72a1734b72", + "Node name for S&R": "WanVideoClipVisionEncode" + }, + "widgets_values": [ + 1, + 1, + "center", + "average", + true, + 0, + 0.5 + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 177, + "type": "WanVideoTorchCompileSettings", + "pos": [ + 625.0336883950495, + -3059.9361098215068 + ], + "size": [ + 342.74609375, + 250 + ], + "flags": {}, + "order": 31, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "torch_compile_args", + "type": "WANCOMPILEARGS", + "links": [] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "f3614e6720744247f3211d60f7b9333f43572384", + "Node name for S&R": "WanVideoTorchCompileSettings" + }, + "widgets_values": [ + "inductor", + false, + "default", + false, + 64, + true, + 128, + false, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 238, + "type": "CLIPVisionLoader", + "pos": [ + 1191.416223580064, + -2352.1171320174303 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 32, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 466 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.50", + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "clip_vision_h.safetensors" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 264, + "type": "SetNode", + "pos": [ + 1495.3445320067444, + -2322.200415573926 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 46, + "mode": 0, + "inputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "link": 466 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_clip_vision_model", + "properties": { + "previousName": "clip_vision_model" + }, + "widgets_values": [ + "clip_vision_model" + ] + }, + { + "id": 342, + "type": "WanVideoSetBlockSwap", + "pos": [ + 1383.985433748381, + -2753.0135059095337 + ], + "size": [ + 209.6841796875, + 46 + ], + "flags": {}, + "order": 56, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 601 + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": 603 + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 602 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSetBlockSwap" + }, + "widgets_values": [], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 260, + "type": "SetNode", + "pos": [ + 1644.5757062717614, + -2729.110530629212 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 59, + "mode": 0, + "inputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "link": 602 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_wanmodel", + "properties": { + "previousName": "wanmodel" + }, + "widgets_values": [ + "wanmodel" + ] + }, + { + "id": 343, + "type": "GetImageRangeFromBatch", + "pos": [ + 5621.818483098485, + -1718.9453295399908 + ], + "size": [ + 349.7056640625, + 102 + ], + "flags": {}, + "order": 70, + "mode": 0, + "inputs": [ + { + "name": "images", + "shape": 7, + "type": "IMAGE", + "link": 604 + }, + { + "name": "masks", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 605 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "40a37dfea5976ef2f744556280e9b34f6c8670f4", + "Node name for S&R": "GetImageRangeFromBatch" + }, + "widgets_values": [ + 0, + 17 + ] + }, + { + "id": 323, + "type": "WanVideoTextEncodeCached", + "pos": [ + 3948.0255220003833, + -1093.5091570796537 + ], + "size": [ + 465.5179138183594, + 386.354248046875 + ], + "flags": {}, + "order": 33, + "mode": 0, + "inputs": [ + { + "name": "extender_args", + "shape": 7, + "type": "WANVIDEOPROMPTEXTENDER_ARGS", + "link": null + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 578 + ] + }, + { + "name": "negative_text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": null + }, + { + "name": "positive_prompt", + "type": "STRING", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "ff779c91714d8ee3484cd4119b082c72a1734b72", + "Node name for S&R": "WanVideoTextEncodeCached" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "a person is talking", + "bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards", + "disabled", + true, + "gpu" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 270, + "type": "INTConstant", + "pos": [ + 2460.418727826879, + -2315.170406203778 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 34, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 472 + ] + } + ], + "title": "Max frames", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 500 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 319, + "type": "WanVideoSamplerExtraArgs", + "pos": [ + 4151.609307406882, + -1221.7156868610175 + ], + "size": [ + 314.0220703125, + 262 + ], + "flags": { + "collapsed": true + }, + "order": 60, + "mode": 0, + "inputs": [ + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": 568 + } + ], + "outputs": [ + { + "name": "extra_args", + "type": "WANVIDSAMPLEREXTRAARGS", + "links": [ + 567 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSamplerExtraArgs" + }, + "widgets_values": [ + 0, + "comfy" + ] + }, + { + "id": 194, + "type": "MultiTalkWav2VecEmbeds", + "pos": [ + 1324.7750244140625, + -1235.9822998046875 + ], + "size": [ + 291.08203125, + 326 + ], + "flags": {}, + "order": 58, + "mode": 0, + "inputs": [ + { + "name": "wav2vec_model", + "type": "WAV2VECMODEL", + "link": 334 + }, + { + "name": "audio_1", + "type": "AUDIO", + "link": 544 + }, + { + "name": "audio_2", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "audio_3", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "audio_4", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "ref_target_masks", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "num_frames", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "link": 529 + } + ], + "outputs": [ + { + "name": "multitalk_embeds", + "type": "MULTITALK_EMBEDS", + "links": [ + 568, + 606 + ] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null + }, + { + "name": "num_frames", + "type": "INT", + "links": [ + 519 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "3d7801cee4c8e3106078dd9b9f146caee95069ba", + "Node name for S&R": "MultiTalkWav2VecEmbeds" + }, + "widgets_values": [ + true, + 400, + 25, + 1, + 1, + "para", + false, + false + ], + "color": "#323", + "bgcolor": "#535" + }, + { + "id": 314, + "type": "WanVideoImageToVideoSkyreelsv3_audio", + "pos": [ + 3508.4890772013023, + -1086.9038917751445 + ], + "size": [ + 317.6041015625, + 310 + ], + "flags": {}, + "order": 66, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 554 + }, + { + "name": "start_image", + "shape": 7, + "type": "IMAGE", + "link": 599 + }, + { + "name": "reference_video", + "shape": 7, + "type": "IMAGE", + "link": 577 + }, + { + "name": "clip_embeds", + "shape": 7, + "type": "WANVIDIMAGE_CLIPEMBEDS", + "link": 557 + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 594 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 595 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 561 + ] + }, + { + "name": "output_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoImageToVideoSkyreelsv3_audio" + }, + "widgets_values": [ + 720, + 720, + 81, + 5, + 12, + false, + false, + "reinhard_torch", + "" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 344, + "type": "PreviewImage", + "pos": [ + 6348.568222204387, + -1788.8504056330976 + ], + "size": [ + 1631.6291903858546, + 742.4102601493835 + ], + "flags": {}, + "order": 72, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 605 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.11.1", + "Node name for S&R": "PreviewImage" + }, + "widgets_values": [] + }, + { + "id": 131, + "type": "VHS_VideoCombine", + "pos": [ + 5343.551732247999, + -1415.6189615290073 + ], + "size": [ + 991.5499877929688, + 334 + ], + "flags": {}, + "order": 71, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 612 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": 449 + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 25, + "loop_count": 0, + "filename_prefix": "WanVideo2_1_InfiniteTalk", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo2_1_InfiniteTalk_00005-audio.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 25, + "workflow": "WanVideo2_1_InfiniteTalk_00005.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_InfiniteTalk_00005-audio.mp4" + } + } + } + }, + { + "id": 245, + "type": "INTConstant", + "pos": [ + 2206.803005170629, + -2491.531246047528 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 35, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 441 + ] + } + ], + "title": "Width", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 1088 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 246, + "type": "INTConstant", + "pos": [ + 2209.472927045629, + -2312.119136672528 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 36, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 442 + ] + } + ], + "title": "Height", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "e435e999e4b1a828a6b5f6d8f037e66f4a798324", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 832 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 309, + "type": "WanVideoPassImagesFromSamples", + "pos": [ + 4961.612077494903, + -1420.846852410617 + ], + "size": [ + 321.5958557128906, + 46 + ], + "flags": {}, + "order": 69, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 563 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 604, + 612 + ] + }, + { + "name": "output_path", + "type": "STRING", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "e5955d83958f1c808c86e5be3804ca42b39ab3fa", + "Node name for S&R": "WanVideoPassImagesFromSamples" + }, + "widgets_values": [] + }, + { + "id": 348, + "type": "WanVideoLoraSelect", + "pos": [ + 173.93886457374074, + -2627.1008778348328 + ], + "size": [ + 270, + 150 + ], + "flags": {}, + "order": 37, + "mode": 0, + "inputs": [ + { + "name": "prev_lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "blocks", + "shape": 7, + "type": "SELECTEDBLOCKS", + "link": null + } + ], + "outputs": [ + { + "name": "lora", + "type": "WANVIDLORA", + "links": [ + 613 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoLoraSelect" + }, + "widgets_values": [ + "WanVideo\\Lightx2v\\lightx2v_I2V_14B_480p_cfg_step_distill_rank64_bf16.safetensors", + 1.2, + false, + false + ] + }, + { + "id": 122, + "type": "WanVideoModelLoader", + "pos": [ + 628.1759603903312, + -2759.144023786062 + ], + "size": [ + 668.1777954101562, + 338 + ], + "flags": {}, + "order": 50, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": null + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": 613 + }, + { + "name": "vram_management_args", + "shape": 7, + "type": "VRAM_MANAGEMENTARGS", + "link": null + }, + { + "name": "extra_model", + "shape": 7, + "type": "VACEPATH", + "link": null + }, + { + "name": "fantasytalking_model", + "shape": 7, + "type": "FANTASYTALKINGMODEL", + "link": null + }, + { + "name": "multitalk_model", + "shape": 7, + "type": "MULTITALKMODEL", + "link": 614 + }, + { + "name": "fantasyportrait_model", + "shape": 7, + "type": "FANTASYPORTRAITMODEL", + "link": null + }, + { + "name": "vace_model", + "shape": 7, + "type": "VACEPATH", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 601 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0", + "Node name for S&R": "WanVideoModelLoader" + }, + "widgets_values": [ + "WanVideo\\fp8_scaled_kj\\I2V\\Wan2_1-I2V-14B-720p_fp8_e4m3fn_scaled_KJ.safetensors", + "fp16_fast", + "disabled", + "offload_device", + "sageattn", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 345, + "type": "WanVideoSamplerExtraArgs", + "pos": [ + 3176.296713917537, + -1589.6459743336152 + ], + "size": [ + 314.0220703125, + 262 + ], + "flags": { + "collapsed": true + }, + "order": 61, + "mode": 0, + "inputs": [ + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": 606 + } + ], + "outputs": [ + { + "name": "extra_args", + "type": "WANVIDSAMPLEREXTRAARGS", + "links": [ + 607 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "WanVideoSamplerExtraArgs" + }, + "widgets_values": [ + 0, + "comfy" + ] + }, + { + "id": 349, + "type": "MultiTalkModelLoader", + "pos": [ + 179.38914756751848, + -2391.649371044611 + ], + "size": [ + 303.515234375, + 58 + ], + "flags": {}, + "order": 38, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MULTITALKMODEL", + "links": [ + 614 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "2c5a04cc6332f56f28e9eb4331d6163161af3770", + "Node name for S&R": "MultiTalkModelLoader" + }, + "widgets_values": [ + "WanVideo\\WanVideo_2_1_Multitalk_14B_fp8_e4m3fn.safetensors" + ] + }, + { + "id": 134, + "type": "WanVideoBlockSwap", + "pos": [ + 1342.6580196085042, + -2642.559739705351 + ], + "size": [ + 281.404296875, + 202 + ], + "flags": {}, + "order": 39, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "block_swap_args", + "type": "BLOCKSWAPARGS", + "links": [ + 603 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "058286fc0f3b0651a2f6b68309df3f06e8332cc0", + "Node name for S&R": "WanVideoBlockSwap" + }, + "widgets_values": [ + 30, + false, + false, + true, + 0, + 1, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 241, + "type": "WanVideoTextEncodeCached", + "pos": [ + 2996.8845862858484, + -2121.732010096998 + ], + "size": [ + 465.5179138183594, + 386.354248046875 + ], + "flags": {}, + "order": 40, + "mode": 0, + "inputs": [ + { + "name": "extender_args", + "shape": 7, + "type": "WANVIDEOPROMPTEXTENDER_ARGS", + "link": null + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 572 + ] + }, + { + "name": "negative_text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": null + }, + { + "name": "positive_prompt", + "type": "STRING", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "ff779c91714d8ee3484cd4119b082c72a1734b72", + "Node name for S&R": "WanVideoTextEncodeCached" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "A woman is giving a speech. She is confident, poised, and joyful. Use a static shot.", + "bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards", + "disabled", + true, + "gpu" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 239, + "type": "MarkdownNote", + "pos": [ + 634.3194976886969, + -2365.4932972379856 + ], + "size": [ + 466.23426467749596, + 135.29214174312165 + ], + "flags": {}, + "order": 41, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Model links", + "properties": {}, + "widgets_values": [ + "You can't mix GGUF MultiTalk with non-GGUF main model, but you can mix different GGUF Qtypes with eachother.\n\nFp8:\n\n[https://huggingface.co/Kijai/WanVideo_comfy_fp8_scaled/blob/main/SkyReelsV3/Wan21-SkyReelsV3-A2V_fp8_scaled_mixed.safetensors](https://huggingface.co/Kijai/WanVideo_comfy_fp8_scaled/blob/main/SkyReelsV3/Wan21-SkyReelsV3-A2V_fp8_scaled_mixed.safetensors)\n\n" + ], + "color": "#432", + "bgcolor": "#653" + } + ], + "links": [ + [ + 334, + 137, + 0, + 194, + 0, + "WAV2VECMODEL" + ], + [ + 436, + 129, + 0, + 240, + 0, + "*" + ], + [ + 441, + 245, + 0, + 247, + 0, + "*" + ], + [ + 442, + 246, + 0, + 248, + 0, + "*" + ], + [ + 449, + 254, + 0, + 131, + 1, + "AUDIO" + ], + [ + 466, + 238, + 0, + 264, + 0, + "*" + ], + [ + 467, + 265, + 0, + 237, + 0, + "CLIP_VISION" + ], + [ + 472, + 270, + 0, + 271, + 0, + "*" + ], + [ + 494, + 283, + 0, + 281, + 2, + "INT" + ], + [ + 495, + 282, + 0, + 281, + 3, + "INT" + ], + [ + 496, + 284, + 0, + 281, + 0, + "IMAGE" + ], + [ + 519, + 194, + 2, + 294, + 0, + "*" + ], + [ + 520, + 294, + 0, + 293, + 0, + "*" + ], + [ + 529, + 272, + 0, + 194, + 6, + "INT" + ], + [ + 542, + 125, + 0, + 253, + 0, + "AUDIO" + ], + [ + 543, + 301, + 0, + 302, + 0, + "MELROFORMERMODEL" + ], + [ + 544, + 302, + 0, + 194, + 1, + "AUDIO" + ], + [ + 545, + 253, + 0, + 302, + 1, + "AUDIO" + ], + [ + 554, + 244, + 0, + 314, + 0, + "WANVAE" + ], + [ + 557, + 237, + 0, + 314, + 3, + "WANVIDIMAGE_CLIPEMBEDS" + ], + [ + 561, + 314, + 0, + 317, + 1, + "WANVIDIMAGE_EMBEDS" + ], + [ + 563, + 317, + 0, + 309, + 0, + "LATENT" + ], + [ + 567, + 319, + 0, + 317, + 5, + "WANVIDSAMPLEREXTRAARGS" + ], + [ + 568, + 194, + 0, + 319, + 9, + "MULTITALK_EMBEDS" + ], + [ + 569, + 261, + 0, + 320, + 0, + "WANVIDEOMODEL" + ], + [ + 572, + 241, + 0, + 320, + 3, + "WANVIDEOTEXTEMBEDS" + ], + [ + 575, + 322, + 0, + 321, + 0, + "WANVAE" + ], + [ + 576, + 320, + 0, + 321, + 1, + "LATENT" + ], + [ + 577, + 321, + 0, + 314, + 2, + "IMAGE" + ], + [ + 578, + 323, + 0, + 317, + 3, + "WANVIDEOTEXTEMBEDS" + ], + [ + 579, + 324, + 0, + 317, + 0, + "WANVIDEOMODEL" + ], + [ + 580, + 326, + 0, + 320, + 1, + "WANVIDIMAGE_EMBEDS" + ], + [ + 581, + 327, + 0, + 326, + 0, + "WANVAE" + ], + [ + 582, + 237, + 0, + 326, + 1, + "WANVIDIMAGE_CLIPEMBEDS" + ], + [ + 586, + 321, + 0, + 328, + 0, + "IMAGE" + ], + [ + 587, + 318, + 0, + 330, + 0, + "WANVIDEOSCHEDULER" + ], + [ + 588, + 331, + 0, + 320, + 2, + "WANVIDEOSCHEDULER" + ], + [ + 589, + 332, + 0, + 317, + 2, + "WANVIDEOSCHEDULER" + ], + [ + 590, + 281, + 1, + 333, + 0, + "INT" + ], + [ + 591, + 281, + 2, + 334, + 0, + "INT" + ], + [ + 592, + 335, + 0, + 326, + 9, + "INT" + ], + [ + 593, + 336, + 0, + 326, + 10, + "INT" + ], + [ + 594, + 337, + 0, + 314, + 4, + "INT" + ], + [ + 595, + 338, + 0, + 314, + 5, + "INT" + ], + [ + 596, + 281, + 0, + 237, + 1, + "IMAGE" + ], + [ + 598, + 281, + 0, + 339, + 0, + "IMAGE" + ], + [ + 599, + 340, + 0, + 314, + 1, + "IMAGE" + ], + [ + 600, + 341, + 0, + 326, + 2, + "IMAGE" + ], + [ + 601, + 122, + 0, + 342, + 0, + "WANVIDEOMODEL" + ], + [ + 602, + 342, + 0, + 260, + 0, + "WANVIDEOMODEL" + ], + [ + 603, + 134, + 0, + 342, + 1, + "BLOCKSWAPARGS" + ], + [ + 604, + 309, + 0, + 343, + 0, + "IMAGE" + ], + [ + 605, + 343, + 0, + 344, + 0, + "IMAGE" + ], + [ + 606, + 194, + 0, + 345, + 9, + "MULTITALK_EMBEDS" + ], + [ + 607, + 345, + 0, + 320, + 5, + "WANVIDSAMPLEREXTRAARGS" + ], + [ + 612, + 309, + 0, + 131, + 0, + "IMAGE" + ], + [ + 613, + 348, + 0, + 122, + 2, + "WANVIDLORA" + ], + [ + 614, + 349, + 0, + 122, + 6, + "MULTITALKMODEL" + ] + ], + "groups": [ + { + "id": 1, + "title": "Models", + "bounding": [ + 480.1058044433594, + -3147.42529296875, + 1615.13037109375, + 1070.3359375 + ], + "color": "#88A", + "font_size": 24, + "flags": {} + }, + { + "id": 4, + "title": "Audio", + "bounding": [ + 386.6732482910156, + -1527.863525390625, + 1782.2769775390625, + 833.420166015625 + ], + "color": "#a1309b", + "font_size": 24, + "flags": {} + }, + { + "id": 5, + "title": "Input image", + "bounding": [ + 462.7972412109375, + -2044.41552734375, + 1652.696533203125, + 495.40948486328125 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 6, + "title": "Reference video gen", + "bounding": [ + 2820.5476987924117, + -2901.1407084147545, + 2556.757182928247, + 1244.3950267595374 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 0.5559917313492756, + "offset": [ + 241.21669606034698, + 2972.8400275859303 + ] + }, + "frontendVersion": "1.39.3", + "node_versions": { + "ComfyUI-WanVideoWrapper": "0a11c67a0c0062b534178920a0d6dcaa75e7b5fe", + "comfy-core": "0.3.43", + "audio-separation-nodes-comfyui": "31a4567726e035097cc2d1f767767908a6fda2ea", + "ComfyUI-KJNodes": "f7eb33abc80a2aded1b46dff0dd14d07856a7d50", + "comfyui-videohelpersuite": "a7ce59e381934733bfae03b1be029756d6ce936d" + }, + "VHS_latentpreview": true, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true, + "workflowRendererVersion": "LG" + }, + "version": 0.4 +} \ No newline at end of file diff --git a/multitalk/multitalk.py b/multitalk/multitalk.py index b0aad90..f785cff 100644 --- a/multitalk/multitalk.py +++ b/multitalk/multitalk.py @@ -148,13 +148,13 @@ class AudioProjModel(nn.Module): def __init__( self, seq_len=5, - seq_len_vf=12, - blocks=12, - channels=768, + seq_len_vf=8, + blocks=12, + channels=768, intermediate_dim=512, output_dim=768, context_tokens=32, - norm_output_audio=False, + norm_output_audio=True, ): super().__init__() @@ -278,9 +278,9 @@ class SingleStreamMultiAttention(SingleStreamAttention): def __init__( self, dim: int, - encoder_hidden_states_dim: int, num_heads: int, - qkv_bias: bool, + qkv_bias: bool = True, + encoder_hidden_states_dim: int = 768, class_range: int = 24, class_interval: int = 4, attention_mode: str = 'sdpa', diff --git a/multitalk/multitalk_loop.py b/multitalk/multitalk_loop.py index 513edcc..d7d12ad 100644 --- a/multitalk/multitalk_loop.py +++ b/multitalk/multitalk_loop.py @@ -6,7 +6,7 @@ import numpy as np from ..latent_preview import prepare_callback from ..wanvideo.schedulers import get_scheduler from .multitalk import timestep_transform, add_noise -from ..utils import log, print_memory, temporal_score_rescaling, offload_transformer, init_blockswap +from ..utils import log, print_memory, temporal_score_rescaling, offload_transformer, init_blockswap, match_and_blend_colors from comfy.utils import load_torch_file from ..nodes_model_loading import load_weights from ..HuMo.nodes import get_audio_emb_window @@ -48,7 +48,13 @@ def multitalk_loop(self, **kwargs): mode = image_embeds.get("multitalk_mode", "multitalk") if mode == "auto": mode = transformer.multitalk_model_type.lower() + elif mode == "skyreelsv3": + num_pseudo_frames = 5 + pseudo_frames = reference_keyframes = None + keyframe_index = 0 + reference_video = image_embeds.get("reference_video", None) log.info(f"Multitalk mode: {mode}") + drop_frames = image_embeds.get("drop_frames", 0) cond_frame = None offload = image_embeds.get("force_offload", False) offloaded = False @@ -62,7 +68,9 @@ def multitalk_loop(self, **kwargs): motion_frame = image_embeds.get("motion_frame", 25) target_w = image_embeds.get("target_w", None) target_h = image_embeds.get("target_h", None) - original_images = cond_image = image_embeds.get("multitalk_start_image", None) + original_images = image_embeds.get("multitalk_start_image", None) + cond_image = original_images.clone() if original_images is not None else None + original_color_reference = cond_image.clone() if cond_image is not None else None if original_images is None: original_images = torch.zeros([noise.shape[0], 1, target_h, target_w], device=device) @@ -94,7 +102,6 @@ def multitalk_loop(self, **kwargs): audio_embedding = multitalk_audio_embeds human_num = len(audio_embedding) audio_embs = None - cond_frame = None uni3c_data = None if uni3c_embeds is not None: @@ -110,9 +117,56 @@ def multitalk_loop(self, **kwargs): log.warning("No encoded silence file found, padding with end of audio embedding instead.") total_frames = len(audio_embedding[0]) - estimated_iterations = total_frames // (frame_num - motion_frame) + 1 + estimated_iterations = total_frames // (frame_num - motion_frame - drop_frames) + 1 callback = prepare_callback(patcher, estimated_iterations) + # If reference_video is provided, extract keyframes from it + if mode == "skyreelsv3" and reference_video is not None: + ref_video_length = reference_video.shape[1] # (C, T, H, W) + if colormatch == "reinhard_torch": + reference_video = match_and_blend_colors(reference_video, original_color_reference, 1.0) + + if ref_video_length >= total_frames: + # Reference is long enough - extract keyframes at the expected positions + segment_interval = frame_num - motion_frame - drop_frames + generate_idx = [] + current_idx = frame_num - 1 + while current_idx < total_frames: + generate_idx.append(min(current_idx, ref_video_length - 1)) + current_idx += segment_interval + else: + # Calculate target indices then map to reference video + audio_length = total_frames + generate_idx_target = [0] + segment_interval = frame_num - motion_frame - drop_frames + current_idx = frame_num - 1 + while current_idx < audio_length - 1: + generate_idx_target.append(current_idx) + current_idx += segment_interval + if generate_idx_target[-1] != audio_length - 1: + generate_idx_target.append(audio_length - 1) + + # Map target indices to reference video + generate_idx_target = np.array(generate_idx_target, dtype=np.int16) + original_max = generate_idx_target[-1] + original_min = generate_idx_target[0] + if original_max > original_min: + generate_idx_float = (generate_idx_target.astype(np.float64) - original_min) * (ref_video_length - 1) / (original_max - original_min) + generate_idx = np.clip(np.round(generate_idx_float), 0, ref_video_length - 1).astype(np.int32).tolist() + else: + generate_idx = [0] + + generate_idx = generate_idx[1:] + log.info(f"Reference video ({ref_video_length} frames) mapped to target ({total_frames} frames). Keyframe indices: {generate_idx}") + + # Extract keyframes from reference video + # reference_video shape: (C, T, H, W) from nodes.py processing + # Select keyframes and add batch dimension: (C, num_keyframes, H, W) -> (1, C, num_keyframes, H, W) + selected_keyframes = reference_video[:, generate_idx] # (C, num_keyframes, H, W) + reference_keyframes = selected_keyframes.unsqueeze(0).cpu() # (1, C, num_keyframes, H, W) + log.info(f"Extracted {len(generate_idx)} keyframes from provided reference video at indices {generate_idx}, shape: {reference_keyframes.shape}") + log.info(f"Reference video total frames: {reference_video.shape[1]}, will generate {total_frames} total frames with {estimated_iterations} windows") + if frame_num >= total_frames: arrive_last_frame = True estimated_iterations = 1 @@ -122,6 +176,14 @@ def multitalk_loop(self, **kwargs): while True: # start video generation iteratively self.cache_state = [None, None] + if mode == "skyreelsv3" and reference_keyframes is not None: + clamped_index = min(keyframe_index, reference_keyframes.shape[2] - 1) # Clamp keyframe_index to reuse last keyframe if we run out + pseudo_frames = reference_keyframes[:, :, clamped_index:clamped_index+1].repeat(1, 1, num_pseudo_frames, 1, 1) # Use one keyframe and repeat it 5 times + log.info(f"Window {iteration_count}: using keyframe {clamped_index}/{reference_keyframes.shape[2]-1} for pseudo frames.") + keyframe_index += 1 + else: + pseudo_frames = None + cur_motion_frames_latent_num = int(1 + (cur_motion_frames_num-1) // 4) if mode == "infinitetalk": cond_image = original_images[:, :, current_condframe_index:current_condframe_index+1] if cond_image is not None else None @@ -133,15 +195,13 @@ def multitalk_loop(self, **kwargs): center_indices = torch.clamp(center_indices, min=0, max=audio_embedding[human_idx].shape[0]-1) audio_emb = audio_embedding[human_idx][center_indices].unsqueeze(0).to(device) audio_embs.append(audio_emb) - audio_embs = torch.concat(audio_embs, dim=0).to(dtype) + audio_embs = torch.cat(audio_embs, dim=0).to(dtype) h, w = (cond_image.shape[-2], cond_image.shape[-1]) if cond_image is not None else (target_h, target_w) lat_h, lat_w = h // VAE_STRIDE[1], w // VAE_STRIDE[2] latent_frame_num = (frame_num - 1) // 4 + 1 - noise = torch.randn( - 16, latent_frame_num, - lat_h, lat_w, dtype=torch.float32, device=torch.device("cpu"), generator=seed_g).to(device) + noise = torch.randn(16, latent_frame_num, lat_h, lat_w, dtype=torch.float32, device=torch.device("cpu"), generator=seed_g).to(device) # Calculate the correct latent slice based on current iteration if is_first_clip: @@ -198,24 +258,41 @@ def multitalk_loop(self, **kwargs): if cond_image is not None or cond_frame is not None: cond_ = cond_image if (is_first_clip or humo_image_cond is None) else cond_frame cond_frame_num = cond_.shape[2] - video_frames = torch.zeros(1, 3, frame_num-cond_frame_num, target_h, target_w, device=device, dtype=vae.dtype) - padding_frames_pixels_values = torch.concat([cond_.to(device, vae.dtype), video_frames], dim=2) + + # Prepare pseudo frames if enabled and available from reference_video + if mode == "skyreelsv3" and pseudo_frames is not None: + video_frames = torch.zeros(1, 3, frame_num-cond_frame_num-num_pseudo_frames, target_h, target_w, device=device, dtype=vae.dtype) + padding_frames_pixels_values = torch.cat([cond_.to(device, vae.dtype), video_frames, pseudo_frames.to(device, vae.dtype)], dim=2) + else: + video_frames = torch.zeros(1, 3, frame_num-cond_frame_num, target_h, target_w, device=device, dtype=vae.dtype) + padding_frames_pixels_values = torch.cat([cond_.to(device, vae.dtype), video_frames], dim=2) # encode vae.to(device) y = vae.encode(padding_frames_pixels_values, device=device, tiled=tiled_vae, pbar=False).to(dtype)[0] - if mode == "multitalk": - latent_motion_frames = y[:, :cur_motion_frames_latent_num] # C T H W - else: + if mode == "infinitetalk": cond_ = cond_image if is_first_clip else cond_frame latent_motion_frames = vae.encode(cond_.to(device, vae.dtype), device=device, tiled=tiled_vae, pbar=False).to(dtype)[0] + else: + latent_motion_frames = y[:, :cur_motion_frames_latent_num] # C T H W vae.to(offload_device) #motion_frame_index = cur_motion_frames_latent_num if mode == "infinitetalk" else 1 - msk = torch.zeros(4, latent_frame_num, lat_h, lat_w, device=device, dtype=dtype) - msk[:, :1] = 1 + if mode == "skyreelsv3" and pseudo_frames is not None: + # create mask in pixel space, then transform + msk_pixel = torch.ones(1, frame_num, lat_h, lat_w, device=device) + msk_pixel[:, cur_motion_frames_num : -num_pseudo_frames] = 0 + msk_pixel = torch.cat([ + torch.repeat_interleave(msk_pixel[:, 0:1], repeats=4, dim=1), + msk_pixel[:, 1:], + ], dim=1) + msk_pixel = msk_pixel.view(1, msk_pixel.shape[1] // 4, 4, lat_h, lat_w) + msk = msk_pixel.transpose(1, 2).squeeze(0).to(dtype) # 4 T H W + else: + msk = torch.zeros(4, latent_frame_num, lat_h, lat_w, device=device, dtype=dtype) + msk[:, :1] = 1 y = torch.cat([msk, y]) # 4+C T H W mm.soft_empty_cache() else: @@ -258,7 +335,7 @@ def multitalk_loop(self, **kwargs): latent = noise # injecting motion frames - if not is_first_clip and mode == "multitalk": + if not is_first_clip and mode != "infinitetalk": latent_motion_frames = latent_motion_frames.to(latent.dtype).to(device) motion_add_noise = torch.randn(latent_motion_frames.shape, device=torch.device("cpu"), generator=seed_g).to(device).contiguous() add_latent = add_noise(latent_motion_frames, motion_add_noise, timesteps[0]) @@ -370,12 +447,12 @@ def multitalk_loop(self, **kwargs): latent = image_latent * mask + latent * (1-mask) # injecting motion frames - if not is_first_clip and mode == "multitalk": + if not is_first_clip and mode != "infinitetalk": latent_motion_frames = latent_motion_frames.to(latent.dtype).to(device) motion_add_noise = torch.randn(latent_motion_frames.shape, device=torch.device("cpu"), generator=seed_g).to(device).contiguous() add_latent = add_noise(latent_motion_frames, motion_add_noise, timesteps[i+1]) latent[:, :add_latent.shape[1]] = add_latent - else: + elif mode == "infinitetalk": if humo_image_cond is None or not is_first_clip: latent[:, :cur_motion_frames_latent_num] = latent_motion_frames @@ -392,20 +469,27 @@ def multitalk_loop(self, **kwargs): sampling_pbar.close() + # crop drop_frames from end if enabled + if mode == "skyreelsv3" and drop_frames > 0 and not arrive_last_frame: + videos = videos[:, :-drop_frames] + # optional color correction (less relevant for InfiniteTalk) if colormatch != "disabled": - videos = videos.permute(1, 2, 3, 0).float().numpy() - from color_matcher import ColorMatcher - cm = ColorMatcher() - cm_result_list = [] - for img in videos: - if mode == "multitalk": - cm_result = cm.transfer(src=img, ref=original_images[0].permute(1, 2, 3, 0).squeeze(0).cpu().float().numpy(), method=colormatch) - else: - cm_result = cm.transfer(src=img, ref=cond_image[0].permute(1, 2, 3, 0).squeeze(0).cpu().float().numpy(), method=colormatch) - cm_result_list.append(torch.from_numpy(cm_result).to(vae.dtype)) + if colormatch == "reinhard_torch": + videos = match_and_blend_colors(videos, original_color_reference, 1.0) + else: + videos = videos.permute(1, 2, 3, 0).float().numpy() + from color_matcher import ColorMatcher + cm = ColorMatcher() + cm_result_list = [] + for img in videos: + if mode == "infinitetalk": + cm_result = cm.transfer(src=img, ref=cond_image[0].permute(1, 2, 3, 0).squeeze(0).cpu().float().numpy(), method=colormatch) + else: + cm_result = cm.transfer(src=img, ref=original_images[0].permute(1, 2, 3, 0).squeeze(0).cpu().float().numpy(), method=colormatch) + cm_result_list.append(torch.from_numpy(cm_result).to(vae.dtype)) - videos = torch.stack(cm_result_list, dim=0).permute(3, 0, 1, 2) + videos = torch.stack(cm_result_list, dim=0).permute(3, 0, 1, 2) # optionally save generated samples to disk if output_path: @@ -441,7 +525,7 @@ def multitalk_loop(self, **kwargs): # Repeat audio emb if multitalk_embeds is not None: - audio_start_idx += (frame_num - cur_motion_frames_num - humo_reference_count) + audio_start_idx += (frame_num - cur_motion_frames_num - humo_reference_count - drop_frames) audio_end_idx = audio_start_idx + clip_length if audio_end_idx >= len(audio_embedding[0]): arrive_last_frame = True diff --git a/multitalk/nodes.py b/multitalk/nodes.py index 837a763..8360656 100644 --- a/multitalk/nodes.py +++ b/multitalk/nodes.py @@ -461,13 +461,100 @@ class WanVideoImageToVideoMultiTalk: } return (image_embeds, output_path) - + +class WanVideoImageToVideoSkyreelsv3_audio: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "vae": ("WANVAE",), + "width": ("INT", {"default": 832, "min": 64, "max": 2048, "step": 8, "tooltip": "Width of the generation"}), + "height": ("INT", {"default": 480, "min": 64, "max": 29048, "step": 8, "tooltip": "Height of the generation"}), + "frame_window_size": ("INT", {"default": 81, "min": 1, "max": 10000, "step": 4, "tooltip": "The number of frames to process at once, should be a value the model is generally good at."}), + "motion_frame": ("INT", {"default": 5, "min": 1, "max": 10000, "step": 1, "tooltip": "Driven frame length used in the long video generation. Basically the overlap length."}), + "drop_frames": ("INT", {"default": 12, "min": 0, "max": 10000, "step": 1, "tooltip": "Additional frames to drop when advancing the audio window. Higher values = less overlap = faster generation but potentially less smooth transitions."}), + "tiled_vae": ("BOOLEAN", {"default": False, "tooltip": "Use tiled VAE encoding for reduced memory use"}), + "force_offload": ("BOOLEAN", {"default": False, "tooltip": "Whether to force offload the model within the loop for VAE operations, enable if you encounter memory issues."}), + "colormatch": ( + [ + 'disabled', + 'reinhard_torch', + 'mkl', + 'hm', + 'reinhard', + 'mvgd', + 'hm-mvgd-hm', + 'hm-mkl-hm', + ], { + "default": 'disabled', "tooltip": "Color matching method to use between the windows" + },), + }, + "optional": { + "start_image": ("IMAGE", {"tooltip": "Images to encode"}), + "reference_video": ("IMAGE", {"tooltip": "Optional: Pre-generated reference video to use for keyframes instead of extracting from first generation. Should be color-matched to source image."}), + "clip_embeds": ("WANVIDIMAGE_CLIPEMBEDS", {"tooltip": "Clip vision encoded image"}), + "output_path": ("STRING", {"default": "", "tooltip": "If set, will save each window's resulting frames to this folder, also DISABLES returning the final video tensor to save memory"}), + } + } + + RETURN_TYPES = ("WANVIDIMAGE_EMBEDS", "STRING",) + RETURN_NAMES = ("image_embeds", "output_path") + FUNCTION = "process" + CATEGORY = "WanVideoWrapper" + DESCRIPTION = "Enables Multi/InfiniteTalk long video generation sampling method, the video is created in windows with overlapping frames. Not compatible or necessary to be used with context windows and many other features besides Multi/InfiniteTalk." + + def process(self, vae, width, height, frame_window_size, motion_frame, drop_frames, force_offload, colormatch, start_image=None, + tiled_vae=False, clip_embeds=None, mode="multitalk", output_path="", reference_video=None): + + H, W = height, width + num_frames = ((frame_window_size - 1) // 4) * 4 + 1 + + # Resize and rearrange the input image dimensions + if start_image is not None: + resized_start_image = common_upscale(start_image.movedim(-1, 1), W, H, "lanczos", "disabled").movedim(0, 1) + resized_start_image = resized_start_image * 2 - 1 + resized_start_image = resized_start_image.unsqueeze(0) + + target_shape = (16, (num_frames - 1) // 4 + 1, height // 8, width // 8) + + if output_path: + timestamp = datetime.datetime.now().strftime("%Y%m%d_%H%M%S") + output_path = os.path.join(output_path, f"{timestamp}_{mode}_output") + os.makedirs(output_path, exist_ok=True) + + processed_reference_video = None + if reference_video is not None: + processed_reference_video = common_upscale(reference_video.movedim(-1, 1), W, H, "lanczos", "disabled").movedim(0, 1) + processed_reference_video = processed_reference_video * 2 - 1 + + image_embeds = { + "multitalk_sampling": True, + "multitalk_start_image": resized_start_image if start_image is not None else None, + "frame_window_size": num_frames, + "motion_frame": motion_frame, + "drop_frames": drop_frames, + "use_pseudo_frames": True, + "reference_video": processed_reference_video, + "target_h": H, + "target_w": W, + "tiled_vae": tiled_vae, + "force_offload": force_offload, + "vae": vae, + "target_shape": target_shape, + "clip_context": clip_embeds.get("clip_embeds", None) if clip_embeds is not None else None, + "colormatch": colormatch, + "multitalk_mode": "skyreelsv3", + "output_path": output_path + } + + return (image_embeds, output_path) + NODE_CLASS_MAPPINGS = { "MultiTalkModelLoader": MultiTalkModelLoader, "MultiTalkWav2VecEmbeds": MultiTalkWav2VecEmbeds, "WanVideoImageToVideoMultiTalk": WanVideoImageToVideoMultiTalk, "Wav2VecModelLoader": Wav2VecModelLoader, "MultiTalkSilentEmbeds": MultiTalkSilentEmbeds, + "WanVideoImageToVideoSkyreelsv3_audio": WanVideoImageToVideoSkyreelsv3_audio, } NODE_DISPLAY_NAME_MAPPINGS = { @@ -476,4 +563,5 @@ NODE_DISPLAY_NAME_MAPPINGS = { "WanVideoImageToVideoMultiTalk": "WanVideo Long I2V Multi/InfiniteTalk", "Wav2VecModelLoader": "Wav2vec2 Model Loader", "MultiTalkSilentEmbeds": "MultiTalk Silent Embeds", + "WanVideoImageToVideoSkyreelsv3_audio": "WanVideo Long SkyReelsV3 A2V", } \ No newline at end of file diff --git a/nodes_model_loading.py b/nodes_model_loading.py index 8e28176..eae4f48 100644 --- a/nodes_model_loading.py +++ b/nodes_model_loading.py @@ -810,7 +810,6 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, "adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "face_encoder", "fuser_block"} param_count = sum(1 for _ in transformer.named_parameters()) pbar = ProgressBar(param_count) - cnt = 0 block_idx = vace_block_idx = None if gguf: @@ -920,9 +919,7 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, load_device = offload_device # Set tensor to device set_module_tensor_to_device(transformer, name, device=load_device, dtype=dtype_to_use, value=value) - cnt += 1 - if cnt % 100 == 0: - pbar.update(100) + pbar.update(1) #[print(name, param.device, param.dtype) for name, param in transformer.named_parameters()] memory_on_device = get_module_memory_mb_per_device(transformer) @@ -931,6 +928,8 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, for dev, mem_mb in memory_on_device.items(): log.info(f"Device: {dev:8s} | Memory: {mem_mb:,.2f} MB") + if hasattr(pbar, "_last_sent_value"): + pbar._last_sent_value = -1 pbar.update_absolute(0) def patch_control_lora(transformer, device): @@ -1512,7 +1511,45 @@ class WanVideoModelLoader: block.cross_attn.ip_adapter_single_stream_k_proj = nn.Linear(context_dim, dim, bias=False) block.cross_attn.ip_adapter_single_stream_v_proj = nn.Linear(context_dim, dim, bias=False) - if multitalk_model is not None: + # LongCat Avatar + if "multitalk_audio_proj.proj1.weight" in sd and "blocks.0.audio_cross_attn.q_norm.weight" in sd: + log.info("MultiTalk/InfiniteTalk model detected, patching model...") + from .multitalk.multitalk import AudioProjModel + from .wanvideo.modules.model import WanLayerNorm + from .LongCat.layers import SingleStreamAttention + + + for block in transformer.blocks: + with init_empty_weights(): + if "blocks.0.audio_modulation.1.weight" in sd: + block.audio_modulation = nn.Sequential(nn.SiLU(), nn.Linear(512, 3 * dim, bias=True)) + block.norm_x = WanLayerNorm(dim, transformer.eps, elementwise_affine=True) + block.audio_cross_attn = SingleStreamAttention( + dim=dim, + encoder_hidden_states_dim=768, + num_heads=num_heads, + qkv_bias=True, + qk_norm=True, + class_range=24, + class_interval=4, + attention_mode=attention_mode, + ) + multitalk_proj_model = AudioProjModel() + transformer.multitalk_audio_proj = multitalk_proj_model + # SkyreelsV3 + elif "blocks.1.audio_cross_attn.kv_linear.weight" in sd and "audio_proj.proj1.weight" in sd: + sd = {k.replace("audio_proj", "multitalk_audio_proj"): v for k, v in sd.items()} + # init audio module + from .multitalk.multitalk import SingleStreamMultiAttention, AudioProjModel + from .wanvideo.modules.model import WanLayerNorm + + for block in transformer.blocks: + with init_empty_weights(): + block.norm_x = WanLayerNorm(dim, transformer.eps, elementwise_affine=True) + block.audio_cross_attn = SingleStreamMultiAttention(dim=dim, num_heads=num_heads, attention_mode=attention_mode) + + transformer.multitalk_audio_proj = AudioProjModel() + elif multitalk_model is not None: multitalk_model_type = multitalk_model.get("model_type", "MultiTalk") log.info(f"{multitalk_model_type} detected, patching model...") @@ -1529,15 +1566,7 @@ class WanVideoModelLoader: for block in transformer.blocks: with init_empty_weights(): block.norm_x = WanLayerNorm(dim, transformer.eps, elementwise_affine=True) - block.audio_cross_attn = SingleStreamMultiAttention( - dim=dim, - encoder_hidden_states_dim=768, - num_heads=num_heads, - qkv_bias=True, - class_range=24, - class_interval=4, - attention_mode=attention_mode, - ) + block.audio_cross_attn = SingleStreamMultiAttention(dim=dim, num_heads=num_heads, attention_mode=attention_mode) transformer.multitalk_audio_proj = multitalk_model["proj_model"] transformer.multitalk_model_type = multitalk_model_type @@ -1555,39 +1584,6 @@ class WanVideoModelLoader: sd.update(extra_sd) del extra_sd - elif "multitalk_audio_proj.proj1.weight" in sd: - log.info("MultiTalk/InfiniteTalk model detected, patching model...") - from .multitalk.multitalk import AudioProjModel - from .wanvideo.modules.model import WanLayerNorm - from .LongCat.layers import SingleStreamAttention - - audio_window = 5 - vae_scale = 4 - - for block in transformer.blocks: - with init_empty_weights(): - if "blocks.0.audio_modulation.1.weight" in sd: - block.audio_modulation = nn.Sequential(nn.SiLU(), nn.Linear(512, 3 * dim, bias=True)) - block.norm_x = WanLayerNorm(dim, transformer.eps, elementwise_affine=True) - block.audio_cross_attn = SingleStreamAttention( - dim=dim, - encoder_hidden_states_dim=768, - num_heads=num_heads, - qkv_bias=True, - qk_norm=True, - class_range=24, - class_interval=4, - attention_mode=attention_mode, - ) - multitalk_proj_model = AudioProjModel( - seq_len=audio_window, - seq_len_vf=audio_window+vae_scale-1, - intermediate_dim=512, - output_dim=768, - context_tokens=32, - norm_output_audio=True, - ) - transformer.multitalk_audio_proj = multitalk_proj_model sd = {k.replace(".weight_scale", ".scale_weight"): v for k, v in sd.items()} diff --git a/nodes_sampler.py b/nodes_sampler.py index 93c3fd6..42638d8 100644 --- a/nodes_sampler.py +++ b/nodes_sampler.py @@ -185,11 +185,12 @@ class WanVideoSampler: is_pusa = "pusa" in sample_scheduler.__class__.__name__.lower() - scheduler_step_args = {"generator": seed_g} - step_sig = inspect.signature(sample_scheduler.step) - for arg in list(scheduler_step_args.keys()): - if arg not in step_sig.parameters: - scheduler_step_args.pop(arg) + if scheduler != "multitalk": + scheduler_step_args = {"generator": seed_g} + step_sig = inspect.signature(sample_scheduler.step) + for arg in list(scheduler_step_args.keys()): + if arg not in step_sig.parameters: + scheduler_step_args.pop(arg) # Ovi if transformer.audio_model is not None: # temporary workaround (...nothing more permanent) diff --git a/utils.py b/utils.py index eff0aa7..bc6e2f7 100644 --- a/utils.py +++ b/utils.py @@ -718,3 +718,55 @@ def temporal_score_rescaling(model_output, sample, timestep, k=1.0, tsr_sigma=0. if not t == 1.0: model_output = (ratio * ((1-t) * model_output + sample) - sample) / (1 - t) return model_output + +def match_and_blend_colors( + source_chunk: torch.Tensor, # (C, T, H, W), range [-1, 1] + reference_image: torch.Tensor, # (C, 1, H, W), range [-1, 1] + strength: float, +) -> torch.Tensor: + import kornia + if strength == 0.0: + return source_chunk + source_chunk = source_chunk.unsqueeze(0) # (1, C, T, H, W) + + # shapes + B, C, T, H, W = source_chunk.shape + input_dtype = source_chunk.dtype + + # [-1,1] -> [0,1] + src_01 = (source_chunk + 1.0) * 0.5 + ref_01 = (reference_image + 1.0) * 0.5 + + src32 = src_01.to(torch.float32) + ref32 = ref_01.to(torch.float32) + + # (B, C, T, H, W) -> (B*T, C, H, W) + src_bt = src32.permute(0, 2, 1, 3, 4).contiguous().view(B * T, C, H, W) + ref_bchw = ref32[:, :, 0, :, :].contiguous() + + # RGB->Lab + src_lab = kornia.color.rgb_to_lab(src_bt) # (B*T, C, H, W) + ref_lab = kornia.color.rgb_to_lab(ref_bchw) # (B, C, H, W) + + src_lab_flat = src_lab.view(B * T, C, -1) # (B*T, C, HW) + ref_lab_flat = ref_lab.view(B, C, -1) # (B, C, HW) + src_std, src_mean = torch.std_mean(src_lab_flat, dim=-1, keepdim=True, unbiased=False) + ref_std, ref_mean = torch.std_mean(ref_lab_flat, dim=-1, keepdim=True, unbiased=False) + src_std = src_std.clamp_min_(1e-6) + + ref_mean_bt = ref_mean.repeat_interleave(T, dim=0) # (B*T, C, 1) + ref_std_bt = ref_std.repeat_interleave(T, dim=0) # (B*T, C, 1) + + corrected_lab_flat = (src_lab_flat - src_mean) * (ref_std_bt / src_std) + ref_mean_bt + corrected_lab = corrected_lab_flat.view(B * T, C, H, W) + + # Lab->RGB + corrected_rgb_01 = kornia.color.lab_to_rgb(corrected_lab) # (B*T, C, H, W) + + blended_rgb_01 = (1.0 - strength) * src_bt + strength * corrected_rgb_01 + + # (B, C, T, H, W) + blended_rgb_01 = blended_rgb_01.view(B, T, C, H, W).permute(0, 2, 1, 3, 4).contiguous() + + # [0,1] -> [-1,1] + return (blended_rgb_01 * 2.0 - 1.0)[0].to(dtype=input_dtype) From e4e7f413f756fbae6b5d8aa1d58e5b1b32c35cb6 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Sun, 1 Feb 2026 15:15:27 +0200 Subject: [PATCH 4/9] Make compatible with latest ComfyUI version --- nodes_model_loading.py | 40 ++++++++++++++++++---------------------- 1 file changed, 18 insertions(+), 22 deletions(-) diff --git a/nodes_model_loading.py b/nodes_model_loading.py index eae4f48..3895c4a 100644 --- a/nodes_model_loading.py +++ b/nodes_model_loading.py @@ -52,17 +52,6 @@ def update_folder_names_and_paths(key, targets=[]): log.warning(f"Unknown file list already present on key {key}: {base}") update_folder_names_and_paths("unet_gguf", ["diffusion_models", "unet"]) -class WanVideoModel(comfy.model_base.BaseModel): - def __init__(self, *args, **kwargs): - super().__init__(*args, **kwargs) - self.pipeline = {} - - def __getitem__(self, k): - return self.pipeline[k] - - def __setitem__(self, k, v): - self.pipeline[k] = v - try: from comfy.latent_formats import Wan21, Wan22 latent_format = Wan21 @@ -71,16 +60,27 @@ except: #for backwards compatibility from comfy.latent_formats import HunyuanVideo latent_format = HunyuanVideo +class WanVideoModel(torch.nn.Module): + def __init__(self, model_config, transformer, device=None): + super().__init__() + self.latent_format = model_config.latent_format + self.model_config = model_config + self.device = device + self.current_patcher = None + self.diffusion_model = transformer + self.pipeline = {} + + def __getitem__(self, k): + return self.pipeline[k] + + def __setitem__(self, k, v): + self.pipeline[k] = v + class WanVideoModelConfig: - def __init__(self, dtype, latent_format=latent_format): + def __init__(self, latent_format=latent_format): self.unet_config = {} self.unet_extra_config = {} self.latent_format = latent_format - #self.latent_format.latent_channels = 16 - self.manual_cast_dtype = dtype - self.sampling_settings = {"multiplier": 1.0} - self.memory_usage_factor = 2.0 - self.unet_config["disable_unet_model_creation"] = True def filter_state_dict_by_blocks(state_dict, blocks_mapping, layer_filter=[]): filtered_dict = {} @@ -1609,11 +1609,7 @@ class WanVideoModelLoader: transformer.text_projection = nn.Sequential(nn.Linear(sd["text_projection.0.weight"].shape[1], text_dim), nn.GELU(approximate='tanh'), nn.Linear(text_dim, text_dim)) latent_format=Wan22 if dim == 3072 else Wan21 - comfy_model = WanVideoModel( - WanVideoModelConfig(base_dtype, latent_format=latent_format), - model_type=comfy.model_base.ModelType.FLOW, - device=device, - ) + comfy_model = WanVideoModel(WanVideoModelConfig(latent_format=latent_format), device=device, transformer=transformer) # SteadyDancer if "condition_embedding_align.cross_attn.in_proj_bias" in sd: From f55b7b3d893b66e9a4dc907a285f20959ebfabf5 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Sun, 1 Feb 2026 15:15:33 +0200 Subject: [PATCH 5/9] Update wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json --- ...V_SkyReelsV3_TalkingAvatar_example_01.json | 1206 ++++++++--------- 1 file changed, 531 insertions(+), 675 deletions(-) diff --git a/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json b/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json index 4350292..9ff62cf 100644 --- a/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json +++ b/example_workflows/wanvideo_2_1_14B_I2V_SkyReelsV3_TalkingAvatar_example_01.json @@ -1,7 +1,7 @@ { "id": "8b7a9a57-2303-4ef5-9fc2-bf41713bd1fc", "revision": 0, - "last_node_id": 349, + "last_node_id": 351, "last_link_id": 614, "nodes": [ { @@ -64,7 +64,7 @@ "flags": { "collapsed": true }, - "order": 42, + "order": 43, "mode": 0, "inputs": [ { @@ -104,7 +104,7 @@ "flags": { "collapsed": true }, - "order": 48, + "order": 49, "mode": 0, "inputs": [ { @@ -144,7 +144,7 @@ "flags": { "collapsed": true }, - "order": 49, + "order": 50, "mode": 0, "inputs": [ { @@ -293,7 +293,7 @@ "flags": { "collapsed": true }, - "order": 47, + "order": 48, "mode": 0, "inputs": [ { @@ -486,7 +486,7 @@ "flags": { "collapsed": true }, - "order": 44, + "order": 45, "mode": 0, "inputs": [ { @@ -592,7 +592,7 @@ 46 ], "flags": {}, - "order": 55, + "order": 56, "mode": 0, "inputs": [ { @@ -1028,106 +1028,6 @@ false ] }, - { - "id": 329, - "type": "Note", - "pos": [ - 3399.811668121435, - -2817.225543033241 - ], - "size": [ - 286.291473419416, - 88 - ], - "flags": {}, - "order": 18, - "mode": 0, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "First we generate normal I2V to get frames to use as reference for the audio part" - ], - "color": "#432", - "bgcolor": "#653" - }, - { - "id": 328, - "type": "VHS_VideoCombine", - "pos": [ - 4759.630440472775, - -2700.899077815229 - ], - "size": [ - 555.5285458590442, - 334 - ], - "flags": {}, - "order": 67, - "mode": 0, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 586 - }, - { - "name": "audio", - "shape": 7, - "type": "AUDIO", - "link": null - }, - { - "name": "meta_batch", - "shape": 7, - "type": "VHS_BatchManager", - "link": null - }, - { - "name": "vae", - "shape": 7, - "type": "VAE", - "link": null - } - ], - "outputs": [ - { - "name": "Filenames", - "type": "VHS_FILENAMES", - "links": null - } - ], - "properties": { - "cnr_id": "comfyui-videohelpersuite", - "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", - "Node name for S&R": "VHS_VideoCombine" - }, - "widgets_values": { - "frame_rate": 25, - "loop_count": 0, - "filename_prefix": "WanVideo2_1_InfiniteTalk", - "format": "video/h264-mp4", - "pix_fmt": "yuv420p", - "crf": 19, - "save_metadata": true, - "trim_to_audio": false, - "pingpong": false, - "save_output": false, - "videopreview": { - "hidden": false, - "paused": false, - "params": { - "filename": "WanVideo2_1_InfiniteTalk_00004.mp4", - "subfolder": "", - "type": "temp", - "format": "video/h264-mp4", - "frame_rate": 25, - "workflow": "WanVideo2_1_InfiniteTalk_00004.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_InfiniteTalk_00004.mp4" - } - } - } - }, { "id": 322, "type": "GetNode", @@ -1142,7 +1042,7 @@ "flags": { "collapsed": true }, - "order": 19, + "order": 18, "mode": 0, "inputs": [], "outputs": [ @@ -1176,7 +1076,7 @@ "flags": { "collapsed": true }, - "order": 45, + "order": 46, "mode": 0, "inputs": [ { @@ -1214,7 +1114,7 @@ "flags": { "collapsed": true }, - "order": 20, + "order": 19, "mode": 0, "inputs": [], "outputs": [ @@ -1322,7 +1222,7 @@ "flags": { "collapsed": true }, - "order": 21, + "order": 20, "mode": 0, "inputs": [], "outputs": [ @@ -1354,7 +1254,7 @@ "flags": { "collapsed": true }, - "order": 22, + "order": 21, "mode": 0, "inputs": [], "outputs": [ @@ -1384,7 +1284,7 @@ 391.5079908265427 ], "flags": {}, - "order": 23, + "order": 22, "mode": 0, "inputs": [ { @@ -1414,7 +1314,8 @@ 5, 0, -1, - false + false, + "" ] }, { @@ -1431,7 +1332,7 @@ "flags": { "collapsed": true }, - "order": 53, + "order": 54, "mode": 0, "inputs": [ { @@ -1471,7 +1372,7 @@ "flags": { "collapsed": true }, - "order": 54, + "order": 55, "mode": 0, "inputs": [ { @@ -1511,7 +1412,7 @@ "flags": { "collapsed": true }, - "order": 24, + "order": 23, "mode": 0, "inputs": [], "outputs": [ @@ -1545,7 +1446,7 @@ "flags": { "collapsed": true }, - "order": 25, + "order": 24, "mode": 0, "inputs": [], "outputs": [ @@ -1579,7 +1480,7 @@ "flags": { "collapsed": true }, - "order": 26, + "order": 25, "mode": 0, "inputs": [], "outputs": [ @@ -1613,7 +1514,7 @@ "flags": { "collapsed": true }, - "order": 27, + "order": 26, "mode": 0, "inputs": [], "outputs": [ @@ -1647,7 +1548,7 @@ "flags": { "collapsed": true }, - "order": 28, + "order": 27, "mode": 0, "inputs": [], "outputs": [ @@ -1679,7 +1580,7 @@ 336 ], "flags": {}, - "order": 43, + "order": 44, "mode": 0, "inputs": [ { @@ -1752,7 +1653,8 @@ "0, 0, 0", "top", 16, - "cpu" + "cpu", + "