diff --git a/__init__.py b/__init__.py index ebb563c..f6db32e 100644 --- a/__init__.py +++ b/__init__.py @@ -84,6 +84,13 @@ except Exception as e: STEADYDANCER_NODE_CLASS_MAPPINGS = {} STEADYDANCER_NODE_DISPLAY_NAME_MAPPINGS = {} +try: + from .onetoall.nodes import NODE_CLASS_MAPPINGS as ONETOALL_NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as ONETOALL_NODE_DISPLAY_NAME_MAPPINGS +except Exception as e: + log.warning(f"WanVideoWrapper WARNING: OneToAll nodes not available due to error in importing them: {e}") + ONETOALL_NODE_CLASS_MAPPINGS = {} + ONETOALL_NODE_DISPLAY_NAME_MAPPINGS = {} + NODE_CLASS_MAPPINGS.update(RECAM_MASTER_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(UNIANIMATE_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(SKYREELS_NODE_CLASS_MAPPINGS) @@ -108,6 +115,8 @@ NODE_CLASS_MAPPINGS.update(OVI_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(FLASHVSR_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(MOCHA_NODE_CLASS_MAPPINGS) NODE_CLASS_MAPPINGS.update(STEADYDANCER_NODE_CLASS_MAPPINGS) +NODE_CLASS_MAPPINGS.update(ONETOALL_NODE_CLASS_MAPPINGS) + NODE_DISPLAY_NAME_MAPPINGS.update(RECAM_MASTER_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(UNIANIMATE_NODE_DISPLAY_NAME_MAPPINGS) @@ -133,5 +142,6 @@ NODE_DISPLAY_NAME_MAPPINGS.update(OVI_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(FLASHVSR_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(MOCHA_NODE_DISPLAY_NAME_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(STEADYDANCER_NODE_DISPLAY_NAME_MAPPINGS) +NODE_DISPLAY_NAME_MAPPINGS.update(ONETOALL_NODE_DISPLAY_NAME_MAPPINGS) __all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] \ No newline at end of file diff --git a/controlnet/wan_controlnet.py b/controlnet/wan_controlnet.py index 9a982c5..242b64b 100644 --- a/controlnet/wan_controlnet.py +++ b/controlnet/wan_controlnet.py @@ -10,18 +10,13 @@ from diffusers.utils import USE_PEFT_BACKEND, logging, scale_lora_layers, unscal from diffusers.models.modeling_outputs import Transformer2DModelOutput from diffusers.models.modeling_utils import ModelMixin from diffusers.models.transformers.transformer_wan import ( - WanTimeTextImageEmbedding, - WanRotaryPosEmbed, + WanTimeTextImageEmbedding, + WanRotaryPosEmbed, WanTransformerBlock ) logger = logging.get_logger(__name__) # pylint: disable=invalid-name -def zero_module(module): - for p in module.parameters(): - nn.init.zeros_(p) - return module - class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModelMixin): r""" @@ -69,7 +64,7 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel _no_split_modules = ["WanTransformerBlock"] _keep_in_fp32_modules = ["time_embedder", "scale_shift_table", "norm1", "norm2", "norm3"] _keys_to_ignore_on_load_unexpected = ["norm_added_q"] - + @register_to_config def __init__( self, @@ -100,10 +95,10 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel ## Spatial compression with time awareness nn.Sequential( nn.Conv3d( - in_channels, - input_channels[0], + in_channels, + input_channels[0], kernel_size=(3, downscale_coef + 1, downscale_coef + 1), - stride=(1, downscale_coef, downscale_coef), + stride=(1, downscale_coef, downscale_coef), padding=(1, downscale_coef // 2, downscale_coef // 2) ), nn.GELU(approximate="tanh"), @@ -122,9 +117,9 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel nn.GroupNorm(2, input_channels[2]), ) ]) - + inner_dim = num_attention_heads * attention_head_dim - + # 1. Patch & position embedding self.rope = WanRotaryPosEmbed(attention_head_dim, patch_size, rope_max_seq_len) self.patch_embedding = nn.Conv3d(vae_channels + input_channels[2], inner_dim, kernel_size=patch_size, stride=patch_size) @@ -153,11 +148,10 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel for _ in range(len(self.blocks)): controlnet_block = nn.Linear(inner_dim, out_proj_dim) - controlnet_block = zero_module(controlnet_block) self.controlnet_blocks.append(controlnet_block) - + self.gradient_checkpointing = False - + def forward( self, hidden_states: torch.Tensor, @@ -187,7 +181,7 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel # 0. Controlnet encoder for control_encoder_block in self.control_encoder: controlnet_states = control_encoder_block(controlnet_states) - + hidden_states = torch.cat([hidden_states, controlnet_states], dim=1) ## 1. Patch embedding and stack @@ -216,7 +210,7 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel if encoder_hidden_states_image is not None: encoder_hidden_states = torch.concat([encoder_hidden_states_image, encoder_hidden_states], dim=1) - + # 4. Transformer blocks controlnet_hidden_states = () if torch.is_grad_enabled() and self.gradient_checkpointing: @@ -239,43 +233,4 @@ class WanControlnet(ModelMixin, ConfigMixin, PeftAdapterMixin, FromOriginalModel return (controlnet_hidden_states,) return Transformer2DModelOutput(sample=controlnet_hidden_states) - - -if __name__ == "__main__": - parameters = { - "added_kv_proj_dim": None, - "attention_head_dim": 128, - "cross_attn_norm": True, - "eps": 1e-06, - "ffn_dim": 8960, - "freq_dim": 256, - "image_dim": None, - "in_channels": 3, - "num_attention_heads": 12, - "num_layers": 2, - "patch_size": [1, 2, 2], - "qk_norm": "rms_norm_across_heads", - "rope_max_seq_len": 1024, - "text_dim": 4096, - "downscale_coef": 8, - "out_proj_dim": 12 * 128, - "vae_channels": 16 - } - controlnet = WanControlnet(**parameters) - - hidden_states = torch.rand(1, 16, 13, 60, 90) - timestep = torch.tensor([1000]).repeat(17550).unsqueeze(0) #torch.randint(low=0, high=1000, size=(1,), dtype=torch.long) - encoder_hidden_states = torch.rand(1, 512, 4096) - controlnet_states = torch.rand(1, 3, 49, 480, 720) - - controlnet_hidden_states = controlnet( - hidden_states=hidden_states, - timestep=timestep, - encoder_hidden_states=encoder_hidden_states, - controlnet_states=controlnet_states, - return_dict=False - ) - print("Output states count", len(controlnet_hidden_states[0])) - for out_hidden_states in controlnet_hidden_states[0]: - print(out_hidden_states.shape) diff --git a/example_workflows/Wan21_OneToAllAnimation_example_01.json b/example_workflows/Wan21_OneToAllAnimation_example_01.json new file mode 100644 index 0000000..c484de8 --- /dev/null +++ b/example_workflows/Wan21_OneToAllAnimation_example_01.json @@ -0,0 +1,9315 @@ +{ + "id": "c6e410bc-5e2c-460b-ae81-c91b6094fbb1", + "revision": 0, + "last_node_id": 311, + "last_link_id": 503, + "nodes": [ + { + "id": 11, + "type": "LoadWanVideoT5TextEncoder", + "pos": [ + -795.1616201345697, + -458.50687233733436 + ], + "size": [ + 377.1661376953125, + 130 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "wan_t5_model", + "type": "WANTEXTENCODER", + "slot_index": 0, + "links": [ + 15 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "LoadWanVideoT5TextEncoder" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "offload_device", + "disabled" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 98, + "type": "WanVideoAddOneToAllPoseEmbeds", + "pos": [ + 1448.8516747518884, + -759.5892906813722 + ], + "size": [ + 344.2642578125, + 146 + ], + "flags": {}, + "order": 98, + "mode": 0, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 332 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 352 + }, + { + "name": "pose_prefix_image", + "shape": 7, + "type": "IMAGE", + "link": 329 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 330 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllPoseEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 176, + "type": "GetNode", + "pos": [ + 1845.0940790302232, + -587.3412125215413 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 327 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 106, + "type": "LoadImage", + "pos": [ + -1193.755718041252, + -1749.5456298423485 + ], + "size": [ + 274.080078125, + 314 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 277 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "title": "Load Image: Reference", + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.76", + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "pasted/image (1035).png", + "image" + ] + }, + { + "id": 128, + "type": "OnnxDetectionModelLoader", + "pos": [ + -335.45014980720333, + -1988.997571942917 + ], + "size": [ + 292.8853515625, + 106 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "POSEMODEL", + "links": [ + 255 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanAnimatePreprocess", + "ver": "2fcbcae7eec637fdc712fdec18e6266feb8ba3a7", + "Node name for S&R": "OnnxDetectionModelLoader" + }, + "widgets_values": [ + "vitpose-l-wholebody.onnx", + "onnx\\yolov10m.onnx", + "CUDAExecutionProvider" + ] + }, + { + "id": 178, + "type": "SetNode", + "pos": [ + 1473.2361587929145, + -820.675404155246 + ], + "size": [ + 332.15625, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 96, + "mode": 0, + "inputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "link": 331 + } + ], + "outputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 332 + ] + } + ], + "title": "Set_ref_embeds", + "properties": { + "previousName": "ref_embeds" + }, + "widgets_values": [ + "ref_embeds" + ] + }, + { + "id": 105, + "type": "WanVideoAddOneToAllReferenceEmbeds", + "pos": [ + 1457.4042220746367, + -1063.546662053317 + ], + "size": [ + 344.1879986998439, + 166 + ], + "flags": {}, + "order": 92, + "mode": 0, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 184 + }, + { + "name": "vae", + "type": "WANVAE", + "link": 325 + }, + { + "name": "ref_image", + "type": "IMAGE", + "link": 266 + }, + { + "name": "ref_mask", + "shape": 7, + "type": "MASK", + "link": 346 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 331 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllReferenceEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 186, + "type": "GetNode", + "pos": [ + 1259.014517799361, + -1216.4976702178271 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 341 + ] + } + ], + "title": "Get_gen_width", + "properties": {}, + "widgets_values": [ + "gen_width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 187, + "type": "GetNode", + "pos": [ + 1264.2817772003318, + -1167.3920575846996 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 342 + ] + } + ], + "title": "Get_gen_height", + "properties": {}, + "widgets_values": [ + "gen_height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 174, + "type": "GetNode", + "pos": [ + 1280.707343991687, + -1059.8792465156102 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 325 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 99, + "type": "WanVideoEmptyEmbeds", + "pos": [ + 1481.5779452001077, + -1246.7221935731116 + ], + "size": [ + 272.431640625, + 126 + ], + "flags": {}, + "order": 73, + "mode": 0, + "inputs": [ + { + "name": "control_embeds", + "shape": 7, + "type": "WANVIDIMAGE_EMBEDS", + "link": null + }, + { + "name": "extra_latents", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 341 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 342 + }, + { + "name": "num_frames", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "link": 345 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 184 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoEmptyEmbeds" + }, + "widgets_values": [ + 480, + 832, + 81 + ] + }, + { + "id": 177, + "type": "GetNode", + "pos": [ + 1261.5325627482212, + -689.8011114700616 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 329 + ] + } + ], + "title": "Get_ref_pose", + "properties": {}, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 195, + "type": "GetImageRangeFromBatch", + "pos": [ + 1115.2031139178644, + -849.6709330969245 + ], + "size": [ + 302.1748267547342, + 102 + ], + "flags": {}, + "order": 74, + "mode": 0, + "inputs": [ + { + "name": "images", + "shape": 7, + "type": "IMAGE", + "link": 351 + }, + { + "name": "masks", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "num_frames", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "link": 353 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 352 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "GetImageRangeFromBatch" + }, + "widgets_values": [ + 0, + 81 + ] + }, + { + "id": 193, + "type": "GetNode", + "pos": [ + 1076.403053714897, + -692.5111625250113 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 351 + ] + } + ], + "title": "Get_pose_images", + "properties": {}, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 198, + "type": "SetNode", + "pos": [ + 130.43125675802116, + -398.9326244460033 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 78, + "mode": 0, + "inputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "link": 356 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_text_embeds", + "properties": { + "previousName": "text_embeds" + }, + "widgets_values": [ + "text_embeds" + ] + }, + { + "id": 35, + "type": "WanVideoTorchCompileSettings", + "pos": [ + -1203.10748054892, + -1113.158179825764 + ], + "size": [ + 390.5999755859375, + 250 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "torch_compile_args", + "type": "WANCOMPILEARGS", + "slot_index": 0, + "links": [ + 111 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoTorchCompileSettings" + }, + "widgets_values": [ + "inductor", + false, + "default", + false, + 64, + true, + 128, + false, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 22, + "type": "WanVideoModelLoader", + "pos": [ + -783.0013257514316, + -1113.471824131216 + ], + "size": [ + 477.4410095214844, + 338 + ], + "flags": {}, + "order": 67, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": 111 + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": null + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "vram_management_args", + "shape": 7, + "type": "VRAM_MANAGEMENTARGS", + "link": null + }, + { + "name": "extra_model", + "shape": 7, + "type": "VACEPATH", + "link": null + }, + { + "name": "fantasytalking_model", + "shape": 7, + "type": "FANTASYTALKINGMODEL", + "link": null + }, + { + "name": "multitalk_model", + "shape": 7, + "type": "MULTITALKMODEL", + "link": null + }, + { + "name": "fantasyportrait_model", + "shape": 7, + "type": "FANTASYPORTRAITMODEL", + "link": null + }, + { + "name": "vace_model", + "shape": 7, + "type": "VACEPATH", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "slot_index": 0, + "links": [ + 155 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoModelLoader" + }, + "widgets_values": [ + "WanVideo\\OneToAll\\Wan21-OneToAllAnimation_fp8_e4m3fn_scaled_KJ.safetensors", + "fp16_fast", + "disabled", + "offload_device", + "sageattn", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 92, + "type": "WanVideoSetBlockSwap", + "pos": [ + -255.9436070031603, + -826.7734108769189 + ], + "size": [ + 201.7681640625, + 46 + ], + "flags": {}, + "order": 79, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 155 + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": 156 + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 157 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7e290c67bff1f906cdab84523018573f6c9d4d7f", + "Node name for S&R": "WanVideoSetBlockSwap" + }, + "widgets_values": [], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 39, + "type": "WanVideoBlockSwap", + "pos": [ + -257.58717966392953, + -1104.7483557480432 + ], + "size": [ + 315, + 202 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "block_swap_args", + "type": "BLOCKSWAPARGS", + "slot_index": 0, + "links": [ + 156 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoBlockSwap" + }, + "widgets_values": [ + 32, + false, + false, + true, + 1, + 1, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 80, + "type": "WanVideoSetLoRAs", + "pos": [ + -32.57836520831636, + -834.969415937952 + ], + "size": [ + 222.27981567382812, + 46 + ], + "flags": {}, + "order": 83, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 157 + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": 110 + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 354 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSetLoRAs" + }, + "widgets_values": [], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 56, + "type": "WanVideoLoraSelect", + "pos": [ + -922.1241028789763, + -702.6640915803675 + ], + "size": [ + 625.6896274024377, + 154.3601465571362 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "prev_lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "blocks", + "shape": 7, + "type": "SELECTEDBLOCKS", + "link": null + } + ], + "outputs": [ + { + "name": "lora", + "type": "WANVIDLORA", + "links": [ + 110 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoLoraSelect" + }, + "widgets_values": [ + "WanVideo\\Lightx2v\\lightx2v_T2V_14B_cfg_step_distill_v2_lora_rank64_bf16_.safetensors", + 1, + false, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 38, + "type": "WanVideoVAELoader", + "pos": [ + -204.09056596688924, + -659.7661559633536 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + } + ], + "outputs": [ + { + "name": "vae", + "type": "WANVAE", + "slot_index": 0, + "links": [ + 324 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoVAELoader" + }, + "widgets_values": [ + "wanvideo\\Wan2_1_VAE_bf16.safetensors", + "bf16", + false + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 173, + "type": "SetNode", + "pos": [ + 154.4563446113392, + -627.81676576093 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 68, + "mode": 0, + "inputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "link": 324 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_VAE", + "properties": { + "previousName": "VAE" + }, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 16, + "type": "WanVideoTextEncode", + "pos": [ + -378.1559707543689, + -458.9345010498894 + ], + "size": [ + 474.3573303222656, + 316.48370361328125 + ], + "flags": {}, + "order": 66, + "mode": 0, + "inputs": [ + { + "name": "t5", + "shape": 7, + "type": "WANTEXTENCODER", + "link": 15 + }, + { + "name": "model_to_offload", + "shape": 7, + "type": "WANVIDEOMODEL", + "link": null + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "slot_index": 0, + "links": [ + 356 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoTextEncode" + }, + "widgets_values": [ + "video of anime witch dancing, she's wearing a big hat and a gray robe, the style is soft japanese 2d animation style", + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走", + true, + true, + "gpu" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 201, + "type": "GetNode", + "pos": [ + 1294.38274542039, + -504.06178368527384 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 13, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 358 + ] + } + ], + "title": "Get_model", + "properties": {}, + "widgets_values": [ + "model" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 202, + "type": "GetNode", + "pos": [ + 1275.91716248651, + -444.70838301523736 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 14, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 359 + ] + } + ], + "title": "Get_text_embeds", + "properties": {}, + "widgets_values": [ + "text_embeds" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 107, + "type": "ImageResizeKJv2", + "pos": [ + -804.2149849730151, + -1794.963947440175 + ], + "size": [ + 270, + 336 + ], + "flags": {}, + "order": 84, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 277 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 367 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 368 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 257 + ] + }, + { + "name": "width", + "type": "INT", + "links": null + }, + { + "name": "height", + "type": "INT", + "links": null + }, + { + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 480, + 832, + "lanczos", + "crop", + "0, 0, 0", + "center", + 16, + "cpu" + ] + }, + { + "id": 131, + "type": "ImageResizeKJv2", + "pos": [ + -759.188756073471, + -2598.358844649257 + ], + "size": [ + 270, + 336 + ], + "flags": {}, + "order": 69, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 234 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 363 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 364 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 256 + ] + }, + { + "name": "width", + "type": "INT", + "links": [ + 339 + ] + }, + { + "name": "height", + "type": "INT", + "links": [ + 340 + ] + }, + { + "name": "mask", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageResizeKJv2" + }, + "widgets_values": [ + 480, + 832, + "lanczos", + "crop", + "0, 0, 0", + "center", + 16, + "cpu" + ] + }, + { + "id": 208, + "type": "GetNode", + "pos": [ + -909.9808041495439, + -2486.8707987818093 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 15, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 363 + ] + } + ], + "title": "Get_width", + "properties": {}, + "widgets_values": [ + "width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 209, + "type": "GetNode", + "pos": [ + -908.6774957957107, + -2430.8284532659322 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 16, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 364 + ] + } + ], + "title": "Get_height", + "properties": {}, + "widgets_values": [ + "height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 185, + "type": "SetNode", + "pos": [ + -747.0643077401129, + -2006.229576866948 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 81, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 340 + } + ], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 366, + 368 + ] + } + ], + "title": "Set_gen_height", + "properties": { + "previousName": "gen_height" + }, + "widgets_values": [ + "gen_height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 184, + "type": "SetNode", + "pos": [ + -745.5383861445993, + -2053.9665955386185 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 80, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 339 + } + ], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 365, + 367 + ] + } + ], + "title": "Set_gen_width", + "properties": { + "previousName": "gen_width" + }, + "widgets_values": [ + "gen_width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 138, + "type": "PreviewImage", + "pos": [ + 50.86321713314622, + -1725.2912649416003 + ], + "size": [ + 213.66310348004185, + 374.6435470605736 + ], + "flags": {}, + "order": 91, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 371 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.76", + "Node name for S&R": "PreviewImage" + }, + "widgets_values": [] + }, + { + "id": 137, + "type": "VHS_VideoCombine", + "pos": [ + 782.285623186155, + -2444.691047269818 + ], + "size": [ + 287.28398784391584, + 811.2922455961209 + ], + "flags": {}, + "order": 94, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 372 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "onetotall_pose", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "onetotall_pose_00019.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "onetotall_pose_00019.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\onetotall_pose_00019.mp4" + } + } + } + }, + { + "id": 192, + "type": "SetNode", + "pos": [ + 105.59627674852838, + -1873.056615227936 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 89, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 347 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 372 + ] + } + ], + "title": "Set_pose_images", + "properties": { + "previousName": "pose_images" + }, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 171, + "type": "SetNode", + "pos": [ + 110.23802159526953, + -1806.6275008459781 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 90, + "mode": 0, + "inputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "link": 321 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 370 + ] + } + ], + "title": "Set_ref_pose", + "properties": { + "previousName": "ref_pose" + }, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 145, + "type": "PreviewImage", + "pos": [ + 515.6198450949249, + -2023.7646316057644 + ], + "size": [ + 210, + 382.14765702148634 + ], + "flags": {}, + "order": 95, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 370 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.76", + "Node name for S&R": "PreviewImage" + }, + "widgets_values": [] + }, + { + "id": 162, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 2839.654759342869, + 1811.7067208541482 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 76, + "mode": 2, + "inputs": [ + { + "name": "source_images", + "type": "IMAGE", + "link": 301 + }, + { + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "name": "source_images", + "type": "IMAGE", + "links": [ + 337 + ] + }, + { + "name": "start_images", + "type": "IMAGE", + "links": [ + 307 + ] + }, + { + "name": "extended_images", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 172, + "type": "GetNode", + "pos": [ + 3180.6995393079064, + 1966.0259917746791 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 17, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 322 + ] + } + ], + "title": "Get_ref_pose", + "properties": {}, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 175, + "type": "GetNode", + "pos": [ + 2852.4906338185747, + 2038.74029062336 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 18, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 326 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 181, + "type": "GetNode", + "pos": [ + 4073.3641637652363, + 2009.979582277341 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 19, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 334 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 182, + "type": "WanVideoDecode", + "pos": [ + 4068.4959841143555, + 2073.9183839707684 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 99, + "mode": 2, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 334 + }, + { + "name": "samples", + "type": "LATENT", + "link": 335 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 336 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 163, + "type": "WanVideoSampler", + "pos": [ + 3715.56305599809, + 1698.0255429406568 + ], + "size": [ + 315, + 1215.3333333333335 + ], + "flags": {}, + "order": 97, + "mode": 2, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 355 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 401 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 357 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + }, + { + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 416 + }, + { + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 412 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 335 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 7, + 0, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + "" + ] + }, + { + "id": 130, + "type": "VHS_LoadVideo", + "pos": [ + -1184.0544782492284, + -2597.8271314720423 + ], + "size": [ + 247.455078125, + 727.9202266483517 + ], + "flags": {}, + "order": 20, + "mode": 0, + "inputs": [ + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 234 + ] + }, + { + "name": "frame_count", + "type": "INT", + "links": [] + }, + { + "name": "audio", + "type": "AUDIO", + "links": null + }, + { + "name": "video_info", + "type": "VHS_VIDEOINFO", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "8550981384301e9bc5bfea83e5c2c75258102593", + "Node name for S&R": "VHS_LoadVideo" + }, + "widgets_values": { + "video": "vid.mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 300, + "skip_first_frames": 0, + "select_every_nth": 1, + "format": "AnimateDiff", + "choose video to upload": "image", + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "vid.mp4", + "type": "input", + "format": "video/mp4", + "force_rate": 0, + "custom_width": 0, + "custom_height": 0, + "frame_load_cap": 300, + "skip_first_frames": 0, + "select_every_nth": 1 + } + } + } + }, + { + "id": 141, + "type": "PoseDetectionOneToAllAnimation", + "pos": [ + -331.6582810980512, + -1816.9919378688412 + ], + "size": [ + 321.68515625, + 214 + ], + "flags": {}, + "order": 87, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "POSEMODEL", + "link": 255 + }, + { + "name": "images", + "type": "IMAGE", + "link": 256 + }, + { + "name": "ref_image", + "shape": 7, + "type": "IMAGE", + "link": 257 + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 365 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 366 + } + ], + "outputs": [ + { + "name": "pose_images", + "type": "IMAGE", + "links": [ + 347 + ] + }, + { + "name": "ref_pose_image", + "type": "IMAGE", + "links": [ + 321 + ] + }, + { + "name": "ref_image", + "type": "IMAGE", + "links": [ + 266, + 371 + ] + }, + { + "name": "ref_mask", + "type": "MASK", + "links": [ + 346 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanAnimatePreprocess", + "ver": "8689199bf4fd631bf3e46a64f8304d826f398442", + "Node name for S&R": "PoseDetectionOneToAllAnimation" + }, + "widgets_values": [ + 832, + 480, + "ref", + "weak", + "full" + ] + }, + { + "id": 160, + "type": "VHS_VideoCombine", + "pos": [ + 4441.834579182315, + 1763.159366996609 + ], + "size": [ + 632.7794606974512, + 1410.1510652089155 + ], + "flags": {}, + "order": 103, + "mode": 2, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 338 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo_OneToAllAnimation", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo_OneToAllAnimation_00047.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo_OneToAllAnimation_00047.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo_OneToAllAnimation_00047.mp4" + } + } + } + }, + { + "id": 164, + "type": "WanVideoAddOneToAllPoseEmbeds", + "pos": [ + 3345.021372931639, + 1913.8233243652492 + ], + "size": [ + 344.2642578125, + 146 + ], + "flags": {}, + "order": 93, + "mode": 2, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 402 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 419 + }, + { + "name": "pose_prefix_image", + "shape": 7, + "type": "IMAGE", + "link": 322 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 401 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllPoseEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 240, + "type": "GetNode", + "pos": [ + 1306.119453895459, + -138.80728677446322 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 21, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 415 + ] + } + ], + "title": "Get_cfg", + "properties": {}, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 235, + "type": "GetNode", + "pos": [ + 1278.2742665797873, + -24.151069765455006 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 22, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "COMBO", + "type": "COMBO", + "links": [ + 411 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 234, + "type": "SetNode", + "pos": [ + 879.5245208150117, + -460.4181258180906 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 77, + "mode": 0, + "inputs": [ + { + "name": "COMBO", + "type": "COMBO", + "link": 410 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_scheduler", + "properties": { + "previousName": "scheduler" + }, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 179, + "type": "GetNode", + "pos": [ + 3155.9781622992928, + 1673.137236705533 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 23, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 400 + ] + } + ], + "title": "Get_ref_embeds", + "properties": {}, + "widgets_values": [ + "ref_embeds" + ] + }, + { + "id": 194, + "type": "GetNode", + "pos": [ + 3166.8577051024004, + 1730.0246841323294 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 24, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 395 + ] + } + ], + "title": "Get_pose_images", + "properties": {}, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 189, + "type": "GetNode", + "pos": [ + 3190.356554911142, + 1783.3198143417032 + ], + "size": [ + 210, + 58 + ], + "flags": { + "collapsed": true + }, + "order": 25, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 399 + ] + } + ], + "title": "Get_overlap", + "properties": {}, + "widgets_values": [ + "overlap" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 199, + "type": "GetNode", + "pos": [ + 3504.317826018415, + 2109.7952009762257 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 26, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 357 + ] + } + ], + "title": "Get_text_embeds", + "properties": {}, + "widgets_values": [ + "text_embeds" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 236, + "type": "GetNode", + "pos": [ + 3544.6736548128865, + 2239.4229892791045 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 27, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "COMBO", + "type": "COMBO", + "links": [ + 412 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 241, + "type": "GetNode", + "pos": [ + 3567.846811208546, + 2173.3318832387895 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 28, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 416 + ] + } + ], + "title": "Get_cfg", + "properties": {}, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 153, + "type": "WanVideoEncode", + "pos": [ + 3007.6623258473733, + 2030.480627720721 + ], + "size": [ + 270, + 242 + ], + "flags": {}, + "order": 82, + "mode": 2, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 326 + }, + { + "name": "image", + "type": "IMAGE", + "link": 307 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 288 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoEncode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + 0, + 1 + ] + }, + { + "id": 169, + "type": "INTConstant", + "pos": [ + 544.3234468966637, + -1022.305387520862 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 29, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 343 + ] + } + ], + "title": "overlap", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 5 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 203, + "type": "INTConstant", + "pos": [ + 544.3234468966637, + -908.8989939293682 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 30, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 361 + ] + } + ], + "title": "width", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 480 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 204, + "type": "INTConstant", + "pos": [ + 544.3234468966637, + -798.5693780165677 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 31, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "INT", + "links": [ + 362 + ] + } + ], + "title": "height", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "INTConstant" + }, + "widgets_values": [ + 832 + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 196, + "type": "SetNode", + "pos": [ + 173.52086567372532, + -709.3253033454741 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 86, + "mode": 0, + "inputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "link": 354 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_model", + "properties": { + "previousName": "model" + }, + "widgets_values": [ + "model" + ] + }, + { + "id": 191, + "type": "PrimitiveNode", + "pos": [ + 544.3234468966637, + -1180.336968149296 + ], + "size": [ + 210, + 82 + ], + "flags": {}, + "order": 32, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "links": [ + 345, + 353 + ] + } + ], + "title": "num_frames", + "properties": { + "Run widget replace on values": false + }, + "widgets_values": [ + 81, + "fixed" + ] + }, + { + "id": 238, + "type": "FloatConstant", + "pos": [ + 544.3234468966637, + -686.5652392807289 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 33, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "value", + "type": "FLOAT", + "links": [ + 414 + ] + } + ], + "title": "CFG", + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "FloatConstant" + }, + "widgets_values": [ + 1 + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 239, + "type": "SetNode", + "pos": [ + 778.9942947518144, + -656.4639289760764 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 75, + "mode": 0, + "inputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "link": 414 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_cfg", + "properties": { + "previousName": "cfg" + }, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 207, + "type": "SetNode", + "pos": [ + 778.9942947518144, + -769.5411176924722 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 72, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 362 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_height", + "properties": { + "previousName": "height" + }, + "widgets_values": [ + "height" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 206, + "type": "SetNode", + "pos": [ + 778.9942947518144, + -879.512049160155 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 71, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 361 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_width", + "properties": { + "previousName": "width" + }, + "widgets_values": [ + "width" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 188, + "type": "SetNode", + "pos": [ + 778.9942947518144, + -995.1919218276627 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 70, + "mode": 0, + "inputs": [ + { + "name": "INT", + "type": "INT", + "link": 343 + } + ], + "outputs": [ + { + "name": "*", + "type": "*", + "links": null + } + ], + "title": "Set_overlap", + "properties": { + "previousName": "overlap" + }, + "widgets_values": [ + "overlap" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 197, + "type": "GetNode", + "pos": [ + 3716.001629122847, + 1634.2765097207869 + ], + "size": [ + 210, + 60 + ], + "flags": { + "collapsed": true + }, + "order": 34, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 355 + ] + } + ], + "title": "Get_model", + "properties": {}, + "widgets_values": [ + "model" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 154, + "type": "WanVideoAddOneToAllExtendEmbeds", + "pos": [ + 3352.817251387321, + 1675.4363455786072 + ], + "size": [ + 322.0653377606026, + 170 + ], + "flags": {}, + "order": 85, + "mode": 2, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 400 + }, + { + "name": "prev_latents", + "type": "LATENT", + "link": 288 + }, + { + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 395 + }, + { + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 399 + }, + { + "name": "frames_processed", + "type": "INT", + "widget": { + "name": "frames_processed" + }, + "link": 417 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 402 + ] + }, + { + "name": "pose_slice", + "type": "IMAGE", + "links": [ + 418 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoAddOneToAllExtendEmbeds" + }, + "widgets_values": [ + 81, + 5, + 0, + "pad_with_last" + ] + }, + { + "id": 242, + "type": "GetImageSizeAndCount", + "pos": [ + 3473.4208519308463, + 1518.8930818357805 + ], + "size": [ + 240.41265869140625, + 86 + ], + "flags": {}, + "order": 88, + "mode": 2, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 418 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 419 + ] + }, + { + "label": "480 width", + "name": "width", + "type": "INT", + "links": null + }, + { + "label": "832 height", + "name": "height", + "type": "INT", + "links": null + }, + { + "label": "81 count", + "name": "count", + "type": "INT", + "links": [] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "a6b867b63a29ca48ddb15c589e17a9f2d8530d57", + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1457.3481541123053, + -533.3290636522167 + ], + "size": [ + 315, + 1215.3333333333335 + ], + "flags": {}, + "order": 100, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 358 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 330 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 359 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + }, + { + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 415 + }, + { + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 411 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 177 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 7, + 0, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + "" + ] + }, + { + "id": 252, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -590.2254184864095 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 35, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 444 + ] + } + ], + "title": "Get_ref_embeds", + "properties": {}, + "widgets_values": [ + "ref_embeds" + ] + }, + { + "id": 253, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -542.3150942141725 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 36, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 445 + ] + } + ], + "title": "Get_pose_images", + "properties": {}, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 259, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -398.4394265992232 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 37, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 449 + ] + } + ], + "title": "Get_model", + "properties": {}, + "widgets_values": [ + "model" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 245, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -494.40476994193534 + ], + "size": [ + 210, + 50 + ], + "flags": { + "collapsed": true + }, + "order": 38, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 447 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 255, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -350.52910232698605 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 39, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 450 + ] + } + ], + "title": "Get_text_embeds", + "properties": {}, + "widgets_values": [ + "text_embeds" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 180, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 4052.7555556510197, + 1789.7669668405626 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 101, + "mode": 2, + "inputs": [ + { + "name": "source_images", + "type": "IMAGE", + "link": 337 + }, + { + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": 336 + } + ], + "outputs": [ + { + "name": "source_images", + "type": "IMAGE", + "links": null + }, + { + "name": "start_images", + "type": "IMAGE", + "links": [] + }, + { + "name": "extended_images", + "type": "IMAGE", + "links": [ + 338, + 454 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 264, + "type": "Reroute", + "pos": [ + 5166.39703978643, + 1613.0386198510726 + ], + "size": [ + 75, + 26 + ], + "flags": {}, + "order": 104, + "mode": 2, + "inputs": [ + { + "name": "", + "type": "*", + "link": 454 + } + ], + "outputs": [ + { + "name": "", + "type": "IMAGE", + "links": [] + } + ], + "properties": { + "showOutputText": false, + "horizontal": false + } + }, + { + "id": 69, + "type": "GetImageSizeAndCount", + "pos": [ + 2838.884742729907, + 1626.3187741511504 + ], + "size": [ + 240.41265869140625, + 86 + ], + "flags": {}, + "order": 40, + "mode": 2, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 301 + ] + }, + { + "label": "480 width", + "name": "width", + "type": "INT", + "links": null + }, + { + "label": "832 height", + "name": "height", + "type": "INT", + "links": null + }, + { + "label": "81 count", + "name": "count", + "type": "INT", + "links": [ + 417 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "a6b867b63a29ca48ddb15c589e17a9f2d8530d57", + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + }, + { + "id": 250, + "type": "VHS_VideoCombine", + "pos": [ + 2949.04786286762, + -594.0225372282977 + ], + "size": [ + 632.7794606974512, + 1410.1510652089155 + ], + "flags": {}, + "order": 107, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 453 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo_OneToAllAnimation", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo_OneToAllAnimation_00056.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo_OneToAllAnimation_00056.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo_OneToAllAnimation_00056.mp4" + } + } + } + }, + { + "id": 28, + "type": "WanVideoDecode", + "pos": [ + 1841.5450248450798, + -540.548868296255 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 102, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 327 + }, + { + "name": "samples", + "type": "LATENT", + "link": 177 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 373, + 456 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 244, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -446.3497508714604 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 41, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 448 + ] + } + ], + "title": "Get_ref_pose", + "properties": {}, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 254, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -300.6370293433564 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 42, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 446 + ] + } + ], + "title": "Get_overlap", + "properties": {}, + "widgets_values": [ + "overlap" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 257, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -247.7718569870676 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 43, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 451 + ] + } + ], + "title": "Get_cfg", + "properties": {}, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 256, + "type": "GetNode", + "pos": [ + 2434.605317921038, + -196.88856942947686 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 44, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "COMBO", + "type": "COMBO", + "links": [ + 452 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 139, + "type": "VHS_VideoCombine", + "pos": [ + 1850.9720046592752, + -290.57954867365413 + ], + "size": [ + 395.9316825037836, + 999.6149163398916 + ], + "flags": {}, + "order": 105, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 373 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo_OneToAllAnimation", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo_OneToAllAnimation_00055.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo_OneToAllAnimation_00055.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo_OneToAllAnimation_00055.mp4" + } + } + } + }, + { + "id": 231, + "type": "WanVideoScheduler", + "pos": [ + 555.2732475181559, + -548.9801075749042 + ], + "size": [ + 287.11648399982687, + 463.4540736986145 + ], + "flags": {}, + "order": 45, + "mode": 0, + "inputs": [ + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + } + ], + "outputs": [ + { + "name": "sigmas", + "type": "SIGMAS", + "links": null + }, + { + "name": "steps", + "type": "INT", + "links": [] + }, + { + "name": "shift", + "type": "FLOAT", + "links": [] + }, + { + "name": "scheduler", + "type": "COMBO", + "links": [ + 410 + ] + }, + { + "name": "start_step", + "type": "INT", + "links": [] + }, + { + "name": "end_step", + "type": "INT", + "links": [] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoScheduler" + }, + "widgets_values": [ + "euler", + 6, + 7, + 0, + -1 + ] + }, + { + "id": 263, + "type": "dede8476-91a7-4dc1-b735-80cfe6ab21d4", + "pos": [ + 2639.9052582244144, + -588.9242489797474 + ], + "size": [ + 255.099609375, + 671.5059895833333 + ], + "flags": { + "collapsed": false + }, + "order": 106, + "mode": 0, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 444 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 445 + }, + { + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 446 + }, + { + "name": "vae", + "type": "WANVAE", + "link": 447 + }, + { + "name": "pose_prefix_image", + "type": "IMAGE", + "link": 448 + }, + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 449 + }, + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "link": 450 + }, + { + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 451 + }, + { + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 452 + }, + { + "label": "PREVIOUS_IMAGES", + "name": "image", + "type": "IMAGE", + "link": 456 + } + ], + "outputs": [ + { + "name": "extended_images", + "type": "IMAGE", + "links": [ + 453, + 492 + ] + } + ], + "title": "Extend", + "properties": { + "proxyWidgets": [ + [ + "-1", + "overlap" + ], + [ + "-1", + "cfg" + ], + [ + "-1", + "scheduler" + ], + [ + "248", + "seed" + ], + [ + "-1", + "vhslatentpreview" + ] + ], + "cnr_id": "comfy-core", + "ver": "0.3.76" + }, + "widgets_values": [ + 5, + 1, + "euler" + ] + }, + { + "id": 287, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -599.3520873320318 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 46, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 483 + ] + } + ], + "title": "Get_ref_embeds", + "properties": {}, + "widgets_values": [ + "ref_embeds" + ] + }, + { + "id": 288, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -551.4417630597949 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 47, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 484 + ] + } + ], + "title": "Get_pose_images", + "properties": {}, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 289, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -407.56609544484553 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 48, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 488 + ] + } + ], + "title": "Get_model", + "properties": {}, + "widgets_values": [ + "model" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 290, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -503.53143878755765 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 49, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 486 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 291, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -359.65577117260835 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 50, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 489 + ] + } + ], + "title": "Get_text_embeds", + "properties": {}, + "widgets_values": [ + "text_embeds" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 293, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -455.4764197170827 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 51, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 487 + ] + } + ], + "title": "Get_ref_pose", + "properties": {}, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 294, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -309.7636981889787 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 52, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 485 + ] + } + ], + "title": "Get_overlap", + "properties": {}, + "widgets_values": [ + "overlap" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 295, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -256.8985258326899 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 53, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 490 + ] + } + ], + "title": "Get_cfg", + "properties": {}, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 296, + "type": "GetNode", + "pos": [ + 3686.1960800408433, + -206.01523827509914 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 54, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "COMBO", + "type": "COMBO", + "links": [ + 491 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 298, + "type": "Note", + "pos": [ + -1259.9208235943547, + -468.61474065573736 + ], + "size": [ + 409.0687497041629, + 174.74978720726676 + ], + "flags": {}, + "order": 55, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "https://huggingface.co/Kijai/WanVideo_comfy_fp8_scaled/blob/main/OneToAllAnimation/Wan21-OneToAllAnimation_fp8_e4m3fn_scaled_KJ.safetensors" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 300, + "type": "MarkdownNote", + "pos": [ + -327.43742149670334, + -2396.7087563603764 + ], + "size": [ + 513.9422004724022, + 289.83132461985497 + ], + "flags": { + "collapsed": false + }, + "order": 56, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Preprocessor links", + "properties": {}, + "widgets_values": [ + "Nodes:\n\n[https://github.com/kijai/ComfyUI-WanAnimatePreprocess](https://github.com/kijai/ComfyUI-WanAnimatePreprocess)\n\nModels:\n\nYOLO:\n\n[https://huggingface.co/Wan-AI/Wan2.2-Animate-14B/blob/main/process_checkpoint/det/yolov10m.onnx](https://huggingface.co/Wan-AI/Wan2.2-Animate-14B/blob/main/process_checkpoint/det/yolov10m.onnx)\n\nViTPose\n\nLarge:\n\n[https://huggingface.co/JunkyByte/easy_ViTPose/blob/main/onnx/wholebody/vitpose-l-wholebody.onnx](https://huggingface.co/JunkyByte/easy_ViTPose/blob/main/onnx/wholebody/vitpose-l-wholebody.onnx)\n\nHuge (needs both files):\n\n[https://huggingface.co/Kijai/vitpose_comfy/blob/main/onnx/vitpose_h_wholebody_model.onnx](https://huggingface.co/Kijai/vitpose_comfy/blob/main/onnx/vitpose_h_wholebody_model.onnx)\n\n[https://huggingface.co/Kijai/vitpose_comfy/blob/main/onnx/vitpose_h_wholebody_data.bin](https://huggingface.co/Kijai/vitpose_comfy/blob/main/onnx/vitpose_h_wholebody_data.bin)" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 301, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -613.8944600057328 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 57, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDIMAGE_EMBEDS", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 494 + ] + } + ], + "title": "Get_ref_embeds", + "properties": {}, + "widgets_values": [ + "ref_embeds" + ] + }, + { + "id": 302, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -565.9841357334958 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 58, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 495 + ] + } + ], + "title": "Get_pose_images", + "properties": {}, + "widgets_values": [ + "pose_images" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 303, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -422.1084681185465 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 59, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOMODEL", + "type": "WANVIDEOMODEL", + "links": [ + 499 + ] + } + ], + "title": "Get_model", + "properties": {}, + "widgets_values": [ + "model" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 304, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -518.0738114612586 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 60, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVAE", + "type": "WANVAE", + "links": [ + 497 + ] + } + ], + "title": "Get_VAE", + "properties": {}, + "widgets_values": [ + "VAE" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 305, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -374.19814384630934 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 61, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "WANVIDEOTEXTEMBEDS", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 500 + ] + } + ], + "title": "Get_text_embeds", + "properties": {}, + "widgets_values": [ + "text_embeds" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 307, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -470.0187923907837 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 62, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 498 + ] + } + ], + "title": "Get_ref_pose", + "properties": {}, + "widgets_values": [ + "ref_pose" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 308, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -324.3060708626797 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 63, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 496 + ] + } + ], + "title": "Get_overlap", + "properties": {}, + "widgets_values": [ + "overlap" + ], + "color": "#1b4669", + "bgcolor": "#29699c" + }, + { + "id": 309, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -271.4408985063909 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 64, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 501 + ] + } + ], + "title": "Get_cfg", + "properties": {}, + "widgets_values": [ + "cfg" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 310, + "type": "GetNode", + "pos": [ + 4936.5193301956615, + -220.55761094880012 + ], + "size": [ + 210, + 34 + ], + "flags": { + "collapsed": true + }, + "order": 65, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "COMBO", + "type": "COMBO", + "links": [ + 502 + ] + } + ], + "title": "Get_scheduler", + "properties": {}, + "widgets_values": [ + "scheduler" + ] + }, + { + "id": 311, + "type": "70a9c226-a6dc-4e8e-9cb9-f16fd821a543", + "pos": [ + 5141.819270499038, + -612.5932904990707 + ], + "size": [ + 255.099609375, + 671.5059895833333 + ], + "flags": { + "collapsed": false + }, + "order": 110, + "mode": 0, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 494 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 495 + }, + { + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 496 + }, + { + "name": "vae", + "type": "WANVAE", + "link": 497 + }, + { + "name": "pose_prefix_image", + "type": "IMAGE", + "link": 498 + }, + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 499 + }, + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "link": 500 + }, + { + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 501 + }, + { + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 502 + }, + { + "label": "PREVIOUS_IMAGES", + "name": "image", + "type": "IMAGE", + "link": 503 + } + ], + "outputs": [ + { + "name": "extended_images", + "type": "IMAGE", + "links": [ + 493 + ] + } + ], + "title": "Extend", + "properties": { + "proxyWidgets": [ + [ + "-1", + "overlap" + ], + [ + "-1", + "cfg" + ], + [ + "-1", + "scheduler" + ], + [ + "248", + "seed" + ], + [ + "-1", + "vhslatentpreview" + ] + ], + "cnr_id": "comfy-core", + "ver": "0.3.76" + }, + "widgets_values": [ + 5, + 1, + "euler" + ] + }, + { + "id": 297, + "type": "cd88a71a-291e-45fd-9414-2c1f20b86257", + "pos": [ + 3891.4960203442197, + -598.0509178253698 + ], + "size": [ + 255.099609375, + 671.5059895833333 + ], + "flags": { + "collapsed": false + }, + "order": 108, + "mode": 0, + "inputs": [ + { + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 483 + }, + { + "name": "pose_images", + "type": "IMAGE", + "link": 484 + }, + { + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 485 + }, + { + "name": "vae", + "type": "WANVAE", + "link": 486 + }, + { + "name": "pose_prefix_image", + "type": "IMAGE", + "link": 487 + }, + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 488 + }, + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "link": 489 + }, + { + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 490 + }, + { + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 491 + }, + { + "label": "PREVIOUS_IMAGES", + "name": "image", + "type": "IMAGE", + "link": 492 + } + ], + "outputs": [ + { + "name": "extended_images", + "type": "IMAGE", + "links": [ + 482, + 503 + ] + } + ], + "title": "Extend", + "properties": { + "proxyWidgets": [ + [ + "-1", + "overlap" + ], + [ + "-1", + "cfg" + ], + [ + "-1", + "scheduler" + ], + [ + "248", + "seed" + ], + [ + "-1", + "vhslatentpreview" + ] + ], + "cnr_id": "comfy-core", + "ver": "0.3.76" + }, + "widgets_values": [ + 5, + 1, + "euler" + ] + }, + { + "id": 292, + "type": "VHS_VideoCombine", + "pos": [ + 4200.638624987422, + -603.14920607392 + ], + "size": [ + 632.7794606974512, + 1410.1510652089155 + ], + "flags": {}, + "order": 109, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 482 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo_OneToAllAnimation", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo_OneToAllAnimation_00057.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo_OneToAllAnimation_00057.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo_OneToAllAnimation_00057.mp4" + } + } + } + }, + { + "id": 306, + "type": "VHS_VideoCombine", + "pos": [ + 5450.961875142243, + -617.691578747621 + ], + "size": [ + 632.7794606974512, + 1410.1510652089155 + ], + "flags": {}, + "order": 111, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 493 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo_OneToAllAnimation", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo_OneToAllAnimation_00053.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo_OneToAllAnimation_00053.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo_OneToAllAnimation_00053.mp4" + } + } + } + } + ], + "links": [ + [ + 15, + 11, + 0, + 16, + 0, + "WANTEXTENCODER" + ], + [ + 110, + 56, + 0, + 80, + 1, + "WANVIDLORA" + ], + [ + 111, + 35, + 0, + 22, + 0, + "WANCOMPILEARGS" + ], + [ + 155, + 22, + 0, + 92, + 0, + "WANVIDEOMODEL" + ], + [ + 156, + 39, + 0, + 92, + 1, + "BLOCKSWAPARGS" + ], + [ + 157, + 92, + 0, + 80, + 0, + "WANVIDEOMODEL" + ], + [ + 177, + 27, + 0, + 28, + 1, + "LATENT" + ], + [ + 184, + 99, + 0, + 105, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 234, + 130, + 0, + 131, + 0, + "IMAGE" + ], + [ + 255, + 128, + 0, + 141, + 0, + "POSEMODEL" + ], + [ + 256, + 131, + 0, + 141, + 1, + "IMAGE" + ], + [ + 257, + 107, + 0, + 141, + 2, + "IMAGE" + ], + [ + 266, + 141, + 2, + 105, + 2, + "IMAGE" + ], + [ + 277, + 106, + 0, + 107, + 0, + "IMAGE" + ], + [ + 288, + 153, + 0, + 154, + 1, + "LATENT" + ], + [ + 301, + 69, + 0, + 162, + 0, + "IMAGE" + ], + [ + 307, + 162, + 1, + 153, + 1, + "IMAGE" + ], + [ + 321, + 141, + 1, + 171, + 0, + "IMAGE" + ], + [ + 322, + 172, + 0, + 164, + 2, + "IMAGE" + ], + [ + 324, + 38, + 0, + 173, + 0, + "*" + ], + [ + 325, + 174, + 0, + 105, + 1, + "WANVAE" + ], + [ + 326, + 175, + 0, + 153, + 0, + "WANVAE" + ], + [ + 327, + 176, + 0, + 28, + 0, + "WANVAE" + ], + [ + 329, + 177, + 0, + 98, + 2, + "IMAGE" + ], + [ + 330, + 98, + 0, + 27, + 1, + "WANVIDIMAGE_EMBEDS" + ], + [ + 331, + 105, + 0, + 178, + 0, + "*" + ], + [ + 332, + 178, + 0, + 98, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 334, + 181, + 0, + 182, + 0, + "WANVAE" + ], + [ + 335, + 163, + 0, + 182, + 1, + "LATENT" + ], + [ + 336, + 182, + 0, + 180, + 1, + "IMAGE" + ], + [ + 337, + 162, + 0, + 180, + 0, + "IMAGE" + ], + [ + 338, + 180, + 2, + 160, + 0, + "IMAGE" + ], + [ + 339, + 131, + 1, + 184, + 0, + "*" + ], + [ + 340, + 131, + 2, + 185, + 0, + "*" + ], + [ + 341, + 186, + 0, + 99, + 2, + "INT" + ], + [ + 342, + 187, + 0, + 99, + 3, + "INT" + ], + [ + 343, + 169, + 0, + 188, + 0, + "*" + ], + [ + 345, + 191, + 0, + 99, + 4, + "INT" + ], + [ + 346, + 141, + 3, + 105, + 3, + "MASK" + ], + [ + 347, + 141, + 0, + 192, + 0, + "*" + ], + [ + 351, + 193, + 0, + 195, + 0, + "IMAGE" + ], + [ + 352, + 195, + 0, + 98, + 1, + "IMAGE" + ], + [ + 353, + 191, + 0, + 195, + 2, + "INT" + ], + [ + 354, + 80, + 0, + 196, + 0, + "*" + ], + [ + 355, + 197, + 0, + 163, + 0, + "WANVIDEOMODEL" + ], + [ + 356, + 16, + 0, + 198, + 0, + "*" + ], + [ + 357, + 199, + 0, + 163, + 2, + "WANVIDEOTEXTEMBEDS" + ], + [ + 358, + 201, + 0, + 27, + 0, + "WANVIDEOMODEL" + ], + [ + 359, + 202, + 0, + 27, + 2, + "WANVIDEOTEXTEMBEDS" + ], + [ + 361, + 203, + 0, + 206, + 0, + "*" + ], + [ + 362, + 204, + 0, + 207, + 0, + "*" + ], + [ + 363, + 208, + 0, + 131, + 2, + "INT" + ], + [ + 364, + 209, + 0, + 131, + 3, + "INT" + ], + [ + 365, + 184, + 0, + 141, + 3, + "INT" + ], + [ + 366, + 185, + 0, + 141, + 4, + "INT" + ], + [ + 367, + 184, + 0, + 107, + 2, + "INT" + ], + [ + 368, + 185, + 0, + 107, + 3, + "INT" + ], + [ + 370, + 171, + 0, + 145, + 0, + "IMAGE" + ], + [ + 371, + 141, + 2, + 138, + 0, + "IMAGE" + ], + [ + 372, + 192, + 0, + 137, + 0, + "IMAGE" + ], + [ + 373, + 28, + 0, + 139, + 0, + "IMAGE" + ], + [ + 395, + 194, + 0, + 154, + 2, + "IMAGE" + ], + [ + 399, + 189, + 0, + 154, + 3, + "INT" + ], + [ + 400, + 179, + 0, + 154, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 401, + 164, + 0, + 163, + 1, + "WANVIDIMAGE_EMBEDS" + ], + [ + 402, + 154, + 0, + 164, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 410, + 231, + 3, + 234, + 0, + "*" + ], + [ + 411, + 235, + 0, + 27, + 18, + "COMBO" + ], + [ + 412, + 236, + 0, + 163, + 18, + "COMBO" + ], + [ + 414, + 238, + 0, + 239, + 0, + "FLOAT" + ], + [ + 415, + 240, + 0, + 27, + 17, + "FLOAT" + ], + [ + 416, + 241, + 0, + 163, + 17, + "FLOAT" + ], + [ + 417, + 69, + 3, + 154, + 4, + "INT" + ], + [ + 418, + 154, + 1, + 242, + 0, + "IMAGE" + ], + [ + 419, + 242, + 0, + 164, + 1, + "IMAGE" + ], + [ + 444, + 252, + 0, + 263, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 445, + 253, + 0, + 263, + 1, + "IMAGE" + ], + [ + 446, + 254, + 0, + 263, + 2, + "INT" + ], + [ + 447, + 245, + 0, + 263, + 3, + "WANVAE" + ], + [ + 448, + 244, + 0, + 263, + 4, + "IMAGE" + ], + [ + 449, + 259, + 0, + 263, + 5, + "WANVIDEOMODEL" + ], + [ + 450, + 255, + 0, + 263, + 6, + "WANVIDEOTEXTEMBEDS" + ], + [ + 451, + 257, + 0, + 263, + 7, + "FLOAT" + ], + [ + 452, + 256, + 0, + 263, + 8, + "COMBO" + ], + [ + 453, + 263, + 0, + 250, + 0, + "IMAGE" + ], + [ + 454, + 180, + 2, + 264, + 0, + "*" + ], + [ + 456, + 28, + 0, + 263, + 9, + "IMAGE" + ], + [ + 482, + 297, + 0, + 292, + 0, + "IMAGE" + ], + [ + 483, + 287, + 0, + 297, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 484, + 288, + 0, + 297, + 1, + "IMAGE" + ], + [ + 485, + 294, + 0, + 297, + 2, + "INT" + ], + [ + 486, + 290, + 0, + 297, + 3, + "WANVAE" + ], + [ + 487, + 293, + 0, + 297, + 4, + "IMAGE" + ], + [ + 488, + 289, + 0, + 297, + 5, + "WANVIDEOMODEL" + ], + [ + 489, + 291, + 0, + 297, + 6, + "WANVIDEOTEXTEMBEDS" + ], + [ + 490, + 295, + 0, + 297, + 7, + "FLOAT" + ], + [ + 491, + 296, + 0, + 297, + 8, + "COMBO" + ], + [ + 492, + 263, + 0, + 297, + 9, + "IMAGE" + ], + [ + 493, + 311, + 0, + 306, + 0, + "IMAGE" + ], + [ + 494, + 301, + 0, + 311, + 0, + "WANVIDIMAGE_EMBEDS" + ], + [ + 495, + 302, + 0, + 311, + 1, + "IMAGE" + ], + [ + 496, + 308, + 0, + 311, + 2, + "INT" + ], + [ + 497, + 304, + 0, + 311, + 3, + "WANVAE" + ], + [ + 498, + 307, + 0, + 311, + 4, + "IMAGE" + ], + [ + 499, + 303, + 0, + 311, + 5, + "WANVIDEOMODEL" + ], + [ + 500, + 305, + 0, + 311, + 6, + "WANVIDEOTEXTEMBEDS" + ], + [ + 501, + 309, + 0, + 311, + 7, + "FLOAT" + ], + [ + 502, + 310, + 0, + 311, + 8, + "COMBO" + ], + [ + 503, + 297, + 0, + 311, + 9, + "IMAGE" + ] + ], + "groups": [ + { + "id": 1, + "title": "Pose extraction", + "bounding": [ + -1297.8988525849584, + -2715.225264043692, + 2452.14461670954, + 1385.7748147407951 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 2, + "title": "Extend1", + "bounding": [ + 2752.3069334719708, + 1447.0588998628475, + 2468.9655581123916, + 1854.05488995403 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + }, + { + "id": 3, + "title": "Models", + "bounding": [ + -1295.6488105123005, + -1287.6022972649387, + 1661.714401224834, + 1202.7549036080636 + ], + "color": "#88A", + "font_size": 24, + "flags": {} + } + ], + "definitions": { + "subgraphs": [ + { + "id": "dede8476-91a7-4dc1-b735-80cfe6ab21d4", + "version": 1, + "state": { + "lastGroupId": 3, + "lastNodeId": 262, + "lastLinkId": 444, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "New Subgraph", + "inputNode": { + "id": -10, + "bounding": [ + 5361.002774866404, + -161.0907971213036, + 155.365234375, + 240 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 8258.151083451643, + -598.3062215865275, + 134.734375, + 60 + ] + }, + "inputs": [ + { + "id": "50429d80-ba30-4f9f-893b-7af5a760a08e", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "linkIds": [ + 436 + ], + "localized_name": "embeds", + "pos": [ + 5496.368009241404, + -141.0907971213036 + ] + }, + { + "id": "80c34bcf-5aa5-40bc-82a9-cdaf34635cae", + "name": "pose_images", + "type": "IMAGE", + "linkIds": [ + 438 + ], + "localized_name": "pose_images", + "shape": 7, + "pos": [ + 5496.368009241404, + -121.09079712130361 + ] + }, + { + "id": "bfcd6d95-5d8d-4507-a2eb-f7487671379c", + "name": "overlap", + "type": "INT", + "linkIds": [ + 439 + ], + "localized_name": "overlap", + "pos": [ + 5496.368009241404, + -101.09079712130361 + ] + }, + { + "id": "36d3d408-174b-401e-8a90-02c58f22fc54", + "name": "vae", + "type": "WANVAE", + "linkIds": [ + 443, + 434 + ], + "localized_name": "vae", + "pos": [ + 5496.368009241404, + -81.09079712130361 + ] + }, + { + "id": "bee34a36-5ecc-4ea6-b616-fa4a506af34e", + "name": "pose_prefix_image", + "type": "IMAGE", + "linkIds": [ + 433 + ], + "localized_name": "pose_prefix_image", + "shape": 7, + "pos": [ + 5496.368009241404, + -61.09079712130361 + ] + }, + { + "id": "f99de622-035c-4b64-92f0-7d711c0d40b1", + "name": "model", + "type": "WANVIDEOMODEL", + "linkIds": [ + 423 + ], + "localized_name": "model", + "pos": [ + 5496.368009241404, + -41.09079712130361 + ] + }, + { + "id": "69917577-89da-436b-a757-6879194f4eb4", + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "linkIds": [ + 425 + ], + "localized_name": "text_embeds", + "shape": 7, + "pos": [ + 5496.368009241404, + -21.09079712130361 + ] + }, + { + "id": "b0c8083f-d6cd-4e94-bb67-59d3c868fb93", + "name": "cfg", + "type": "FLOAT", + "linkIds": [ + 426 + ], + "localized_name": "cfg", + "pos": [ + 5496.368009241404, + -1.0907971213036092 + ] + }, + { + "id": "0df8140b-7748-485f-9bf6-a7046e63ff74", + "name": "scheduler", + "type": "COMBO", + "linkIds": [ + 427 + ], + "localized_name": "scheduler", + "pos": [ + 5496.368009241404, + 18.90920287869639 + ] + }, + { + "id": "01622287-6689-4efe-b2c1-bd60dd89e6ac", + "name": "image", + "type": "IMAGE", + "linkIds": [ + 444 + ], + "label": "PREVIOUS_IMAGES", + "pos": [ + 5496.368009241404, + 38.90920287869639 + ] + } + ], + "outputs": [ + { + "id": "08f3a254-81f1-4355-bbc2-b6c14b273d2f", + "name": "extended_images", + "type": "IMAGE", + "linkIds": [ + 430 + ], + "localized_name": "extended_images", + "pos": [ + 8278.151083451643, + -578.3062215865275 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 261, + "type": "WanVideoAddOneToAllExtendEmbeds", + "pos": [ + 6288.143610087761, + -509.37998545700503 + ], + "size": [ + 322.0653377606026, + 162 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 436 + }, + { + "localized_name": "prev_latents", + "name": "prev_latents", + "type": "LATENT", + "link": 437 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 438 + }, + { + "localized_name": "overlap", + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 439 + }, + { + "localized_name": "frames_processed", + "name": "frames_processed", + "type": "INT", + "widget": { + "name": "frames_processed" + }, + "link": 440 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 431 + ] + }, + { + "localized_name": "pose_slice", + "name": "pose_slice", + "type": "IMAGE", + "links": [ + 442 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoAddOneToAllExtendEmbeds" + }, + "widgets_values": [ + 81, + 5, + 0, + "pad_with_last" + ] + }, + { + "id": 247, + "type": "WanVideoDecode", + "pos": [ + 7535.005595085545, + -409.46353831411506 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 443 + }, + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "link": 422 + } + ], + "outputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 429 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 249, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 7887.376083451644, + -631.2638716834346 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 428 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": 429 + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": null + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": [ + 430 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 251, + "type": "WanVideoAddOneToAllPoseEmbeds", + "pos": [ + 6670.041703250616, + -510.8047448454783 + ], + "size": [ + 344.2642578125, + 146 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 431 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "type": "IMAGE", + "link": 442 + }, + { + "localized_name": "pose_prefix_image", + "name": "pose_prefix_image", + "shape": 7, + "type": "IMAGE", + "link": 433 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 424 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllPoseEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 248, + "type": "WanVideoSampler", + "pos": [ + 7119.721473431301, + -470.0039494072477 + ], + "size": [ + 315, + 1215.3333333333335 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "WANVIDEOMODEL", + "link": 423 + }, + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 424 + }, + { + "localized_name": "text_embeds", + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 425 + }, + { + "localized_name": "samples", + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "localized_name": "feta_args", + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "localized_name": "context_options", + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "localized_name": "cache_args", + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "localized_name": "flowedit_args", + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "localized_name": "slg_args", + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "localized_name": "loop_args", + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "localized_name": "experimental_args", + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "localized_name": "sigmas", + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "localized_name": "unianimate_poses", + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "localized_name": "fantasytalking_embeds", + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "localized_name": "uni3c_embeds", + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "localized_name": "multitalk_embeds", + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "localized_name": "freeinit_args", + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + }, + { + "localized_name": "cfg", + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 426 + }, + { + "localized_name": "scheduler", + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 427 + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 422 + ] + }, + { + "localized_name": "denoised_samples", + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 7, + 0, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + "" + ] + }, + { + "id": 258, + "type": "WanVideoEncode", + "pos": [ + 5808.149021594793, + -422.2708009425685 + ], + "size": [ + 270, + 242 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 434 + }, + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 435 + }, + { + "localized_name": "mask", + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "links": [ + 437 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoEncode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + 0, + 1 + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 243, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 5810.843837432143, + -645.29589561116 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 420 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": [ + 428 + ] + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [ + 435 + ] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 260, + "type": "GetImageSizeAndCount", + "pos": [ + 5803.201936640504, + -835.3482135315888 + ], + "size": [ + 240.41265869140625, + 86 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 444 + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 420 + ] + }, + { + "label": "480 width", + "localized_name": "width", + "name": "width", + "type": "INT", + "links": null + }, + { + "label": "832 height", + "localized_name": "height", + "name": "height", + "type": "INT", + "links": null + }, + { + "label": "81 count", + "localized_name": "count", + "name": "count", + "type": "INT", + "links": [ + 440 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "a6b867b63a29ca48ddb15c589e17a9f2d8530d57", + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + } + ], + "groups": [], + "links": [ + { + "id": 437, + "origin_id": 258, + "origin_slot": 0, + "target_id": 261, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 440, + "origin_id": 260, + "origin_slot": 3, + "target_id": 261, + "target_slot": 4, + "type": "INT" + }, + { + "id": 422, + "origin_id": 248, + "origin_slot": 0, + "target_id": 247, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 428, + "origin_id": 243, + "origin_slot": 0, + "target_id": 249, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 429, + "origin_id": 247, + "origin_slot": 0, + "target_id": 249, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 431, + "origin_id": 261, + "origin_slot": 0, + "target_id": 251, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 442, + "origin_id": 261, + "origin_slot": 1, + "target_id": 251, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 424, + "origin_id": 251, + "origin_slot": 0, + "target_id": 248, + "target_slot": 1, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 435, + "origin_id": 243, + "origin_slot": 1, + "target_id": 258, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 420, + "origin_id": 260, + "origin_slot": 0, + "target_id": 243, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 436, + "origin_id": -10, + "origin_slot": 0, + "target_id": 261, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 438, + "origin_id": -10, + "origin_slot": 1, + "target_id": 261, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 439, + "origin_id": -10, + "origin_slot": 2, + "target_id": 261, + "target_slot": 3, + "type": "INT" + }, + { + "id": 443, + "origin_id": -10, + "origin_slot": 3, + "target_id": 247, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 434, + "origin_id": -10, + "origin_slot": 3, + "target_id": 258, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 433, + "origin_id": -10, + "origin_slot": 4, + "target_id": 251, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 423, + "origin_id": -10, + "origin_slot": 5, + "target_id": 248, + "target_slot": 0, + "type": "WANVIDEOMODEL" + }, + { + "id": 425, + "origin_id": -10, + "origin_slot": 6, + "target_id": 248, + "target_slot": 2, + "type": "WANVIDEOTEXTEMBEDS" + }, + { + "id": 426, + "origin_id": -10, + "origin_slot": 7, + "target_id": 248, + "target_slot": 17, + "type": "FLOAT" + }, + { + "id": 427, + "origin_id": -10, + "origin_slot": 8, + "target_id": 248, + "target_slot": 18, + "type": "COMBO" + }, + { + "id": 430, + "origin_id": 249, + "origin_slot": 2, + "target_id": -20, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 444, + "origin_id": -10, + "origin_slot": 9, + "target_id": 260, + "target_slot": 0, + "type": "IMAGE" + } + ], + "extra": { + "workflowRendererVersion": "LG" + } + }, + { + "id": "cd88a71a-291e-45fd-9414-2c1f20b86257", + "version": 1, + "state": { + "lastGroupId": 3, + "lastNodeId": 262, + "lastLinkId": 444, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "New Subgraph", + "inputNode": { + "id": -10, + "bounding": [ + 5361.002774866404, + -161.0907971213036, + 155.365234375, + 240 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 8258.151083451643, + -598.3062215865275, + 134.734375, + 60 + ] + }, + "inputs": [ + { + "id": "50429d80-ba30-4f9f-893b-7af5a760a08e", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "linkIds": [ + 436 + ], + "localized_name": "embeds", + "pos": [ + 5496.368009241404, + -141.0907971213036 + ] + }, + { + "id": "80c34bcf-5aa5-40bc-82a9-cdaf34635cae", + "name": "pose_images", + "type": "IMAGE", + "linkIds": [ + 438 + ], + "localized_name": "pose_images", + "shape": 7, + "pos": [ + 5496.368009241404, + -121.09079712130361 + ] + }, + { + "id": "bfcd6d95-5d8d-4507-a2eb-f7487671379c", + "name": "overlap", + "type": "INT", + "linkIds": [ + 439 + ], + "localized_name": "overlap", + "pos": [ + 5496.368009241404, + -101.09079712130361 + ] + }, + { + "id": "36d3d408-174b-401e-8a90-02c58f22fc54", + "name": "vae", + "type": "WANVAE", + "linkIds": [ + 443, + 434 + ], + "localized_name": "vae", + "pos": [ + 5496.368009241404, + -81.09079712130361 + ] + }, + { + "id": "bee34a36-5ecc-4ea6-b616-fa4a506af34e", + "name": "pose_prefix_image", + "type": "IMAGE", + "linkIds": [ + 433 + ], + "localized_name": "pose_prefix_image", + "shape": 7, + "pos": [ + 5496.368009241404, + -61.09079712130361 + ] + }, + { + "id": "f99de622-035c-4b64-92f0-7d711c0d40b1", + "name": "model", + "type": "WANVIDEOMODEL", + "linkIds": [ + 423 + ], + "localized_name": "model", + "pos": [ + 5496.368009241404, + -41.09079712130361 + ] + }, + { + "id": "69917577-89da-436b-a757-6879194f4eb4", + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "linkIds": [ + 425 + ], + "localized_name": "text_embeds", + "shape": 7, + "pos": [ + 5496.368009241404, + -21.09079712130361 + ] + }, + { + "id": "b0c8083f-d6cd-4e94-bb67-59d3c868fb93", + "name": "cfg", + "type": "FLOAT", + "linkIds": [ + 426 + ], + "localized_name": "cfg", + "pos": [ + 5496.368009241404, + -1.0907971213036092 + ] + }, + { + "id": "0df8140b-7748-485f-9bf6-a7046e63ff74", + "name": "scheduler", + "type": "COMBO", + "linkIds": [ + 427 + ], + "localized_name": "scheduler", + "pos": [ + 5496.368009241404, + 18.90920287869639 + ] + }, + { + "id": "01622287-6689-4efe-b2c1-bd60dd89e6ac", + "name": "image", + "type": "IMAGE", + "linkIds": [ + 444 + ], + "label": "PREVIOUS_IMAGES", + "pos": [ + 5496.368009241404, + 38.90920287869639 + ] + } + ], + "outputs": [ + { + "id": "08f3a254-81f1-4355-bbc2-b6c14b273d2f", + "name": "extended_images", + "type": "IMAGE", + "linkIds": [ + 430 + ], + "localized_name": "extended_images", + "pos": [ + 8278.151083451643, + -578.3062215865275 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 261, + "type": "WanVideoAddOneToAllExtendEmbeds", + "pos": [ + 6288.143610087761, + -509.37998545700503 + ], + "size": [ + 322.0653377606026, + 162 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 436 + }, + { + "localized_name": "prev_latents", + "name": "prev_latents", + "type": "LATENT", + "link": 437 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 438 + }, + { + "localized_name": "overlap", + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 439 + }, + { + "localized_name": "frames_processed", + "name": "frames_processed", + "type": "INT", + "widget": { + "name": "frames_processed" + }, + "link": 440 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 431 + ] + }, + { + "localized_name": "pose_slice", + "name": "pose_slice", + "type": "IMAGE", + "links": [ + 442 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoAddOneToAllExtendEmbeds" + }, + "widgets_values": [ + 81, + 5, + 0, + "pad_with_last" + ] + }, + { + "id": 247, + "type": "WanVideoDecode", + "pos": [ + 7535.005595085545, + -409.46353831411506 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 443 + }, + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "link": 422 + } + ], + "outputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 429 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 249, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 7887.376083451644, + -631.2638716834346 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 428 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": 429 + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": null + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": [ + 430 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 251, + "type": "WanVideoAddOneToAllPoseEmbeds", + "pos": [ + 6670.041703250616, + -510.8047448454783 + ], + "size": [ + 344.2642578125, + 146 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 431 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "type": "IMAGE", + "link": 442 + }, + { + "localized_name": "pose_prefix_image", + "name": "pose_prefix_image", + "shape": 7, + "type": "IMAGE", + "link": 433 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 424 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllPoseEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 248, + "type": "WanVideoSampler", + "pos": [ + 7119.721473431301, + -470.0039494072477 + ], + "size": [ + 315, + 1215.3333333333335 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "WANVIDEOMODEL", + "link": 423 + }, + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 424 + }, + { + "localized_name": "text_embeds", + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 425 + }, + { + "localized_name": "samples", + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "localized_name": "feta_args", + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "localized_name": "context_options", + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "localized_name": "cache_args", + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "localized_name": "flowedit_args", + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "localized_name": "slg_args", + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "localized_name": "loop_args", + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "localized_name": "experimental_args", + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "localized_name": "sigmas", + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "localized_name": "unianimate_poses", + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "localized_name": "fantasytalking_embeds", + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "localized_name": "uni3c_embeds", + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "localized_name": "multitalk_embeds", + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "localized_name": "freeinit_args", + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + }, + { + "localized_name": "cfg", + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 426 + }, + { + "localized_name": "scheduler", + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 427 + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 422 + ] + }, + { + "localized_name": "denoised_samples", + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 7, + 0, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + "" + ] + }, + { + "id": 258, + "type": "WanVideoEncode", + "pos": [ + 5808.149021594793, + -422.2708009425685 + ], + "size": [ + 270, + 242 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 434 + }, + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 435 + }, + { + "localized_name": "mask", + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "links": [ + 437 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoEncode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + 0, + 1 + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 243, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 5810.843837432143, + -645.29589561116 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 420 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": [ + 428 + ] + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [ + 435 + ] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 260, + "type": "GetImageSizeAndCount", + "pos": [ + 5803.201936640504, + -835.3482135315888 + ], + "size": [ + 240.41265869140625, + 86 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 444 + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 420 + ] + }, + { + "label": "480 width", + "localized_name": "width", + "name": "width", + "type": "INT", + "links": null + }, + { + "label": "832 height", + "localized_name": "height", + "name": "height", + "type": "INT", + "links": null + }, + { + "label": "157 count", + "localized_name": "count", + "name": "count", + "type": "INT", + "links": [ + 440 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "a6b867b63a29ca48ddb15c589e17a9f2d8530d57", + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + } + ], + "groups": [], + "links": [ + { + "id": 437, + "origin_id": 258, + "origin_slot": 0, + "target_id": 261, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 440, + "origin_id": 260, + "origin_slot": 3, + "target_id": 261, + "target_slot": 4, + "type": "INT" + }, + { + "id": 422, + "origin_id": 248, + "origin_slot": 0, + "target_id": 247, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 428, + "origin_id": 243, + "origin_slot": 0, + "target_id": 249, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 429, + "origin_id": 247, + "origin_slot": 0, + "target_id": 249, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 431, + "origin_id": 261, + "origin_slot": 0, + "target_id": 251, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 442, + "origin_id": 261, + "origin_slot": 1, + "target_id": 251, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 424, + "origin_id": 251, + "origin_slot": 0, + "target_id": 248, + "target_slot": 1, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 435, + "origin_id": 243, + "origin_slot": 1, + "target_id": 258, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 420, + "origin_id": 260, + "origin_slot": 0, + "target_id": 243, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 436, + "origin_id": -10, + "origin_slot": 0, + "target_id": 261, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 438, + "origin_id": -10, + "origin_slot": 1, + "target_id": 261, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 439, + "origin_id": -10, + "origin_slot": 2, + "target_id": 261, + "target_slot": 3, + "type": "INT" + }, + { + "id": 443, + "origin_id": -10, + "origin_slot": 3, + "target_id": 247, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 434, + "origin_id": -10, + "origin_slot": 3, + "target_id": 258, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 433, + "origin_id": -10, + "origin_slot": 4, + "target_id": 251, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 423, + "origin_id": -10, + "origin_slot": 5, + "target_id": 248, + "target_slot": 0, + "type": "WANVIDEOMODEL" + }, + { + "id": 425, + "origin_id": -10, + "origin_slot": 6, + "target_id": 248, + "target_slot": 2, + "type": "WANVIDEOTEXTEMBEDS" + }, + { + "id": 426, + "origin_id": -10, + "origin_slot": 7, + "target_id": 248, + "target_slot": 17, + "type": "FLOAT" + }, + { + "id": 427, + "origin_id": -10, + "origin_slot": 8, + "target_id": 248, + "target_slot": 18, + "type": "COMBO" + }, + { + "id": 430, + "origin_id": 249, + "origin_slot": 2, + "target_id": -20, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 444, + "origin_id": -10, + "origin_slot": 9, + "target_id": 260, + "target_slot": 0, + "type": "IMAGE" + } + ], + "extra": { + "workflowRendererVersion": "LG" + } + }, + { + "id": "70a9c226-a6dc-4e8e-9cb9-f16fd821a543", + "version": 1, + "state": { + "lastGroupId": 3, + "lastNodeId": 262, + "lastLinkId": 444, + "lastRerouteId": 0 + }, + "revision": 0, + "config": {}, + "name": "New Subgraph", + "inputNode": { + "id": -10, + "bounding": [ + 5361.002774866404, + -161.0907971213036, + 155.365234375, + 240 + ] + }, + "outputNode": { + "id": -20, + "bounding": [ + 8258.151083451643, + -598.3062215865275, + 134.734375, + 60 + ] + }, + "inputs": [ + { + "id": "50429d80-ba30-4f9f-893b-7af5a760a08e", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "linkIds": [ + 436 + ], + "localized_name": "embeds", + "pos": [ + 5496.368009241404, + -141.0907971213036 + ] + }, + { + "id": "80c34bcf-5aa5-40bc-82a9-cdaf34635cae", + "name": "pose_images", + "type": "IMAGE", + "linkIds": [ + 438 + ], + "localized_name": "pose_images", + "shape": 7, + "pos": [ + 5496.368009241404, + -121.09079712130361 + ] + }, + { + "id": "bfcd6d95-5d8d-4507-a2eb-f7487671379c", + "name": "overlap", + "type": "INT", + "linkIds": [ + 439 + ], + "localized_name": "overlap", + "pos": [ + 5496.368009241404, + -101.09079712130361 + ] + }, + { + "id": "36d3d408-174b-401e-8a90-02c58f22fc54", + "name": "vae", + "type": "WANVAE", + "linkIds": [ + 443, + 434 + ], + "localized_name": "vae", + "pos": [ + 5496.368009241404, + -81.09079712130361 + ] + }, + { + "id": "bee34a36-5ecc-4ea6-b616-fa4a506af34e", + "name": "pose_prefix_image", + "type": "IMAGE", + "linkIds": [ + 433 + ], + "localized_name": "pose_prefix_image", + "shape": 7, + "pos": [ + 5496.368009241404, + -61.09079712130361 + ] + }, + { + "id": "f99de622-035c-4b64-92f0-7d711c0d40b1", + "name": "model", + "type": "WANVIDEOMODEL", + "linkIds": [ + 423 + ], + "localized_name": "model", + "pos": [ + 5496.368009241404, + -41.09079712130361 + ] + }, + { + "id": "69917577-89da-436b-a757-6879194f4eb4", + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "linkIds": [ + 425 + ], + "localized_name": "text_embeds", + "shape": 7, + "pos": [ + 5496.368009241404, + -21.09079712130361 + ] + }, + { + "id": "b0c8083f-d6cd-4e94-bb67-59d3c868fb93", + "name": "cfg", + "type": "FLOAT", + "linkIds": [ + 426 + ], + "localized_name": "cfg", + "pos": [ + 5496.368009241404, + -1.0907971213036092 + ] + }, + { + "id": "0df8140b-7748-485f-9bf6-a7046e63ff74", + "name": "scheduler", + "type": "COMBO", + "linkIds": [ + 427 + ], + "localized_name": "scheduler", + "pos": [ + 5496.368009241404, + 18.90920287869639 + ] + }, + { + "id": "01622287-6689-4efe-b2c1-bd60dd89e6ac", + "name": "image", + "type": "IMAGE", + "linkIds": [ + 444 + ], + "label": "PREVIOUS_IMAGES", + "pos": [ + 5496.368009241404, + 38.90920287869639 + ] + } + ], + "outputs": [ + { + "id": "08f3a254-81f1-4355-bbc2-b6c14b273d2f", + "name": "extended_images", + "type": "IMAGE", + "linkIds": [ + 430 + ], + "localized_name": "extended_images", + "pos": [ + 8278.151083451643, + -578.3062215865275 + ] + } + ], + "widgets": [], + "nodes": [ + { + "id": 261, + "type": "WanVideoAddOneToAllExtendEmbeds", + "pos": [ + 6288.143610087761, + -509.37998545700503 + ], + "size": [ + 322.0653377606026, + 162 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 436 + }, + { + "localized_name": "prev_latents", + "name": "prev_latents", + "type": "LATENT", + "link": 437 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 438 + }, + { + "localized_name": "overlap", + "name": "overlap", + "type": "INT", + "widget": { + "name": "overlap" + }, + "link": 439 + }, + { + "localized_name": "frames_processed", + "name": "frames_processed", + "type": "INT", + "widget": { + "name": "frames_processed" + }, + "link": 440 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 431 + ] + }, + { + "localized_name": "pose_slice", + "name": "pose_slice", + "type": "IMAGE", + "links": [ + 442 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoAddOneToAllExtendEmbeds" + }, + "widgets_values": [ + 81, + 5, + 0, + "pad_with_last" + ] + }, + { + "id": 247, + "type": "WanVideoDecode", + "pos": [ + 7535.005595085545, + -409.46353831411506 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 443 + }, + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "link": 422 + } + ], + "outputs": [ + { + "localized_name": "images", + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 429 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 249, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 7887.376083451644, + -631.2638716834346 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 428 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": 429 + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": null + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": [ + 430 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 251, + "type": "WanVideoAddOneToAllPoseEmbeds", + "pos": [ + 6670.041703250616, + -510.8047448454783 + ], + "size": [ + 344.2642578125, + 146 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "localized_name": "embeds", + "name": "embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 431 + }, + { + "localized_name": "pose_images", + "name": "pose_images", + "type": "IMAGE", + "link": 442 + }, + { + "localized_name": "pose_prefix_image", + "name": "pose_prefix_image", + "shape": 7, + "type": "IMAGE", + "link": 433 + } + ], + "outputs": [ + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 424 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "0f217be4d8742741b0f89db50138214302a58dc3", + "Node name for S&R": "WanVideoAddOneToAllPoseEmbeds" + }, + "widgets_values": [ + 1, + 0, + 1 + ] + }, + { + "id": 248, + "type": "WanVideoSampler", + "pos": [ + 7119.721473431301, + -470.0039494072477 + ], + "size": [ + 315, + 1215.3333333333335 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "localized_name": "model", + "name": "model", + "type": "WANVIDEOMODEL", + "link": 423 + }, + { + "localized_name": "image_embeds", + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 424 + }, + { + "localized_name": "text_embeds", + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 425 + }, + { + "localized_name": "samples", + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "localized_name": "feta_args", + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "localized_name": "context_options", + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "localized_name": "cache_args", + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "localized_name": "flowedit_args", + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "localized_name": "slg_args", + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "localized_name": "loop_args", + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "localized_name": "experimental_args", + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "localized_name": "sigmas", + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "localized_name": "unianimate_poses", + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "localized_name": "fantasytalking_embeds", + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "localized_name": "uni3c_embeds", + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "localized_name": "multitalk_embeds", + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "localized_name": "freeinit_args", + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + }, + { + "localized_name": "cfg", + "name": "cfg", + "type": "FLOAT", + "widget": { + "name": "cfg" + }, + "link": 426 + }, + { + "localized_name": "scheduler", + "name": "scheduler", + "type": "COMBO", + "widget": { + "name": "scheduler" + }, + "link": 427 + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 422 + ] + }, + { + "localized_name": "denoised_samples", + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "998a69cc0acbec503001b8b0ce0a5d5404420e1e", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 7, + 0, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + "" + ] + }, + { + "id": 258, + "type": "WanVideoEncode", + "pos": [ + 5808.149021594793, + -422.2708009425685 + ], + "size": [ + 270, + 242 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "localized_name": "vae", + "name": "vae", + "type": "WANVAE", + "link": 434 + }, + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 435 + }, + { + "localized_name": "mask", + "name": "mask", + "shape": 7, + "type": "MASK", + "link": null + } + ], + "outputs": [ + { + "localized_name": "samples", + "name": "samples", + "type": "LATENT", + "links": [ + 437 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "7ca221874e8a2cabfc766c51bb63774fde3c851b", + "Node name for S&R": "WanVideoEncode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + 0, + 1 + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 243, + "type": "ImageBatchExtendWithOverlap", + "pos": [ + 5810.843837432143, + -645.29589561116 + ], + "size": [ + 310.775, + 146 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "link": 420 + }, + { + "localized_name": "new_images", + "name": "new_images", + "shape": 7, + "type": "IMAGE", + "link": null + } + ], + "outputs": [ + { + "localized_name": "source_images", + "name": "source_images", + "type": "IMAGE", + "links": [ + 428 + ] + }, + { + "localized_name": "start_images", + "name": "start_images", + "type": "IMAGE", + "links": [ + 435 + ] + }, + { + "localized_name": "extended_images", + "name": "extended_images", + "type": "IMAGE", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "50e7dd34d3b6e6bbab1d41e8068e1ddd19bd4d1b", + "Node name for S&R": "ImageBatchExtendWithOverlap" + }, + "widgets_values": [ + 5, + "source", + "linear_blend" + ] + }, + { + "id": 260, + "type": "GetImageSizeAndCount", + "pos": [ + 5803.201936640504, + -835.3482135315888 + ], + "size": [ + 240.41265869140625, + 86 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "link": 444 + } + ], + "outputs": [ + { + "localized_name": "image", + "name": "image", + "type": "IMAGE", + "links": [ + 420 + ] + }, + { + "label": "480 width", + "localized_name": "width", + "name": "width", + "type": "INT", + "links": null + }, + { + "label": "832 height", + "localized_name": "height", + "name": "height", + "type": "INT", + "links": null + }, + { + "label": "233 count", + "localized_name": "count", + "name": "count", + "type": "INT", + "links": [ + 440 + ] + } + ], + "properties": { + "cnr_id": "comfyui-kjnodes", + "ver": "a6b867b63a29ca48ddb15c589e17a9f2d8530d57", + "Node name for S&R": "GetImageSizeAndCount" + } + } + ], + "groups": [], + "links": [ + { + "id": 437, + "origin_id": 258, + "origin_slot": 0, + "target_id": 261, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 440, + "origin_id": 260, + "origin_slot": 3, + "target_id": 261, + "target_slot": 4, + "type": "INT" + }, + { + "id": 422, + "origin_id": 248, + "origin_slot": 0, + "target_id": 247, + "target_slot": 1, + "type": "LATENT" + }, + { + "id": 428, + "origin_id": 243, + "origin_slot": 0, + "target_id": 249, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 429, + "origin_id": 247, + "origin_slot": 0, + "target_id": 249, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 431, + "origin_id": 261, + "origin_slot": 0, + "target_id": 251, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 442, + "origin_id": 261, + "origin_slot": 1, + "target_id": 251, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 424, + "origin_id": 251, + "origin_slot": 0, + "target_id": 248, + "target_slot": 1, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 435, + "origin_id": 243, + "origin_slot": 1, + "target_id": 258, + "target_slot": 1, + "type": "IMAGE" + }, + { + "id": 420, + "origin_id": 260, + "origin_slot": 0, + "target_id": 243, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 436, + "origin_id": -10, + "origin_slot": 0, + "target_id": 261, + "target_slot": 0, + "type": "WANVIDIMAGE_EMBEDS" + }, + { + "id": 438, + "origin_id": -10, + "origin_slot": 1, + "target_id": 261, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 439, + "origin_id": -10, + "origin_slot": 2, + "target_id": 261, + "target_slot": 3, + "type": "INT" + }, + { + "id": 443, + "origin_id": -10, + "origin_slot": 3, + "target_id": 247, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 434, + "origin_id": -10, + "origin_slot": 3, + "target_id": 258, + "target_slot": 0, + "type": "WANVAE" + }, + { + "id": 433, + "origin_id": -10, + "origin_slot": 4, + "target_id": 251, + "target_slot": 2, + "type": "IMAGE" + }, + { + "id": 423, + "origin_id": -10, + "origin_slot": 5, + "target_id": 248, + "target_slot": 0, + "type": "WANVIDEOMODEL" + }, + { + "id": 425, + "origin_id": -10, + "origin_slot": 6, + "target_id": 248, + "target_slot": 2, + "type": "WANVIDEOTEXTEMBEDS" + }, + { + "id": 426, + "origin_id": -10, + "origin_slot": 7, + "target_id": 248, + "target_slot": 17, + "type": "FLOAT" + }, + { + "id": 427, + "origin_id": -10, + "origin_slot": 8, + "target_id": 248, + "target_slot": 18, + "type": "COMBO" + }, + { + "id": 430, + "origin_id": 249, + "origin_slot": 2, + "target_id": -20, + "target_slot": 0, + "type": "IMAGE" + }, + { + "id": 444, + "origin_id": -10, + "origin_slot": 9, + "target_id": 260, + "target_slot": 0, + "type": "IMAGE" + } + ], + "extra": { + "workflowRendererVersion": "LG" + } + } + ] + }, + "config": {}, + "extra": { + "ds": { + "scale": 0.5054470284993341, + "offset": [ + -1010.2210634864717, + 1064.2147361793347 + ] + }, + "frontendVersion": "1.34.0", + "node_versions": { + "ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd", + "comfy-core": "0.3.26", + "ComfyUI-VideoHelperSuite": "0a75c7958fe320efcb052f1d9f8451fd20c730a8" + }, + "VHS_latentpreview": true, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true, + "workflowRendererVersion": "LG" + }, + "version": 0.4 +} \ No newline at end of file diff --git a/nodes_model_loading.py b/nodes_model_loading.py index ef624da..d6a1c04 100644 --- a/nodes_model_loading.py +++ b/nodes_model_loading.py @@ -6,7 +6,7 @@ import numpy as np from tqdm import tqdm import re -from .wanvideo.modules.model import WanModel, LoRALinearLayer +from .wanvideo.modules.model import WanModel, LoRALinearLayer, WanRMSNorm from .wanvideo.modules.t5 import T5EncoderModel from .wanvideo.modules.clip import CLIPModel from .wanvideo.wan_video_vae import WanVideoVAE, WanVideoVAE38 @@ -852,18 +852,18 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, total=param_count, leave=True): block_idx = vace_block_idx = None - if "vace_blocks." in name: + if name.startswith("vace_blocks."): try: vace_block_idx = int(name.split("vace_blocks.")[1].split(".")[0]) except Exception: vace_block_idx = None - elif "blocks." in name and "face" not in name: + elif name.startswith("blocks.") and "face" not in name: try: block_idx = int(name.split("blocks.")[1].split(".")[0]) except Exception: block_idx = None - if "loras" in name or "controlnet" in name: + if "loras" in name: continue # GGUF: skip GGUFParameter params @@ -1310,7 +1310,7 @@ class WanVideoModelLoader: if dim == 1536: model_variant = "1_3B" if dim == 3072: - log.info(f"5B model detected, no Teacache or MagCache coefficients available, consider using EasyCache for this model") + log.info("5B model detected, no Teacache or MagCache coefficients available, consider using EasyCache for this model") if "high" in model.lower() or "low" in model.lower(): if "i2v" in model.lower(): @@ -1375,7 +1375,7 @@ class WanVideoModelLoader: with init_empty_weights(): transformer.audio_model = WanModel(**TRANSFORMER_CONFIG).eval() - from .wanvideo.modules.model import WanLayerNorm, WanRMSNorm + from .wanvideo.modules.model import WanLayerNorm for block in transformer.blocks: block.cross_attn.k_fusion = nn.Linear(block.dim, block.dim) @@ -1477,16 +1477,12 @@ class WanVideoModelLoader: # Additional cond latents if "add_conv_in.weight" in sd: - def zero_module(module): - for p in module.parameters(): - torch.nn.init.zeros_(p) - return module inner_dim = sd["add_conv_in.weight"].shape[0] add_cond_in_dim = sd["add_conv_in.weight"].shape[1] attn_cond_in_dim = sd["attn_conv_in.weight"].shape[1] - transformer.add_conv_in = torch.nn.Conv3d(add_cond_in_dim, inner_dim, kernel_size=transformer.patch_size, stride=transformer.patch_size) - transformer.add_proj = zero_module(torch.nn.Linear(inner_dim, inner_dim)) - transformer.attn_conv_in = torch.nn.Conv3d(attn_cond_in_dim, inner_dim, kernel_size=transformer.patch_size, stride=transformer.patch_size) + transformer.add_conv_in = nn.Conv3d(add_cond_in_dim, inner_dim, kernel_size=transformer.patch_size, stride=transformer.patch_size) + transformer.add_proj = nn.Linear(inner_dim, inner_dim) + transformer.attn_conv_in = nn.Conv3d(attn_cond_in_dim, inner_dim, kernel_size=transformer.patch_size, stride=transformer.patch_size) # Bindweave text_projection if "text_projection.0.weight" in sd: @@ -1516,6 +1512,45 @@ class WanVideoModelLoader: transformer.condition_embedding_align = PoseRefNetNoBNV3(in_channels_x=16, in_channels_c=16, hidden_dim=128, num_heads=8) # Frame-wise Attention Alignment Unit + if "image_to_cond.conv_in.bias" in sd: + # One-to-all + from .onetoall.controlnet import MiniHunyuanEncoder, MiniEncoder2D + from .onetoall.refextractor_2d import WanRefextractor, WanAttentionBlock + log.info("One-to-all model detected, patching model...") + with init_empty_weights(): + transformer.image_to_cond = MiniEncoder2D( + in_channels = sd["image_to_cond.conv_in.bias"].shape[0], + out_channels = in_channels, + down_block_types= ("DownEncoderBlockInflated","DownEncoderBlockInflated","DownEncoderBlockInflated"), + block_out_channels=(16, 16, 16), + norm_num_groups = 4, + layers_per_block = 1, + spatial_compression_ratio=1 + ) + + transformer.input_hint_block = MiniHunyuanEncoder( + in_channels=3, + out_channels=in_channels, + block_out_channels=(16, 16, 16, 16), + norm_num_groups=4, + layers_per_block=1, + spatial_compression_ratio=16 + ) + + controlnet_layers = 1 + transformer.controlnet = nn.Module() + transformer.controlnet.blocks = nn.ModuleList([WanAttentionBlock(in_features, out_features, ffn_dim, ffn2_dim, num_heads) for _ in range(controlnet_layers)]) + transformer.controlnet_zero = nn.ModuleList([nn.Linear(in_features, out_features) for _ in range(controlnet_layers)]) + transformer.refextractor = WanRefextractor( + patch_size=(1, 2, 2), in_dim=sd["refextractor.patch_embedding.weight"].shape[1], + dim=dim, in_features=in_features, out_features=out_features, ffn_dim=ffn_dim, ffn2_dim=ffn2_dim, + num_heads=num_heads, num_layers=7) + + for block in transformer.blocks: + block.ref_attn_k_img = nn.Linear(in_features, out_features) + block.ref_attn_v_img = nn.Linear(in_features, out_features) + block.ref_attn_norm_k_img = WanRMSNorm(out_features, eps=1e-6) + comfy_model.diffusion_model = transformer comfy_model.load_device = transformer_load_device patcher = comfy.model_patcher.ModelPatcher(comfy_model, device, offload_device) @@ -1580,7 +1615,7 @@ class WanVideoModelLoader: if gguf: raise ValueError("GGUF models don't support vram management") from .diffsynth.vram_management import enable_vram_management, AutoWrappedModule, AutoWrappedLinear - from .wanvideo.modules.model import WanLayerNorm, WanRMSNorm + from .wanvideo.modules.model import WanLayerNorm total_params_in_model = sum(p.numel() for p in patcher.model.diffusion_model.parameters()) log.info(f"Total number of parameters in the loaded model: {total_params_in_model}") diff --git a/nodes_sampler.py b/nodes_sampler.py index c816fe0..5439ac3 100644 --- a/nodes_sampler.py +++ b/nodes_sampler.py @@ -1182,6 +1182,31 @@ class WanVideoSampler: sdancer_data = sdancer_embeds.copy() sdancer_data = dict_to_device(sdancer_data, device, dtype) + # One-to-all-Animation + one_to_all_embeds = image_embeds.get("one_to_all_embeds", None) + one_to_all_data = prev_latents = None + latents_to_not_step = 0 + if one_to_all_embeds is not None: + log.info("Using One-to-All embeddings:") + for k, v in one_to_all_embeds.items(): + log.info(f" {k}: {v.shape if isinstance(v, torch.Tensor) else v}") + one_to_all_data = one_to_all_embeds.copy() + one_to_all_data = dict_to_device(one_to_all_data, device, dtype) + if one_to_all_embeds.get("pose_images") is not None: + pose_images_in = one_to_all_data.pop("pose_images") + pose_images = transformer.input_hint_block(pose_images_in) + if one_to_all_embeds.get("ref_latent_pos") is not None: + pose_prefix_image = transformer.input_hint_block(one_to_all_data.pop("pose_prefix_image")) + pose_images = torch.cat([pose_prefix_image, pose_images],dim=2) + one_to_all_data["controlnet_tokens"] = pose_images.flatten(2).transpose(1, 2) + prev_latents = one_to_all_data.get("prev_latents", None) + if prev_latents is not None: + log.info(f"Using previous latents for One-to-All Animation with shape: {prev_latents.shape}") + latent[:, :prev_latents.shape[1]] = prev_latents.to(latent) + one_to_all_data["token_replace"] = True + latents_to_not_step = prev_latents.shape[1] + one_to_all_data["num_latent_frames_to_replace"] = latents_to_not_step + #region model pred def predict_with_cfg(z, cfg_scale, positive_embeds, negative_embeds, timestep, idx, image_cond=None, clip_fea=None, control_latents=None, vace_data=None, unianim_data=None, audio_proj=None, control_camera_latents=None, @@ -1477,6 +1502,7 @@ class WanVideoSampler: "flashvsr_strength": flashvsr_strength, # FlashVSR strength "num_cond_latents": len(all_indices) if transformer.is_longcat else None, "sdancer_input": sdancer_input, # SteadyDancer input + "one_to_all_input": one_to_all_data, # One-to-All input } batch_size = 1 @@ -3070,11 +3096,16 @@ class WanVideoSampler: new_latent.append(latent_slice[:, j:j+1]) latent = torch.cat(new_latent, dim=1) else: - latent = sample_scheduler.step( - noise_pred[:, :orig_noise_len].unsqueeze(0) if recammaster is not None or mocha_embeds is not None else noise_pred.unsqueeze(0), - timestep, - latent[:, :orig_noise_len].unsqueeze(0) if recammaster is not None or mocha_embeds is not None else latent.unsqueeze(0), - **scheduler_step_args)[0].squeeze(0) + if latents_to_not_step > 0: + raw_latent = latent[:, :latents_to_not_step] + noise_pred_in = noise_pred[:, latents_to_not_step:] + latent = latent[:, latents_to_not_step:] + elif recammaster is not None or mocha_embeds is not None: + noise_pred_in = noise_pred[:, :orig_noise_len] + latent = latent[:, :orig_noise_len] + else: + noise_pred_in = noise_pred + latent = sample_scheduler.step(noise_pred_in.unsqueeze(0), timestep, latent.unsqueeze(0), **scheduler_step_args)[0].squeeze(0) if noise_pred_flipped is not None: latent_backwards = sample_scheduler_flipped.step( noise_pred_flipped.unsqueeze(0), @@ -3083,6 +3114,8 @@ class WanVideoSampler: **scheduler_step_args)[0].squeeze(0) latent_backwards = torch.flip(latent_backwards, dims=[1]) latent = latent * 0.5 + latent_backwards * 0.5 + if latents_to_not_step > 0: + latent = torch.cat([raw_latent, latent], dim=1) if latent_ovi is not None: latent_ovi = sample_scheduler_ovi.step(noise_pred_ovi.unsqueeze(0), t, latent_ovi.to(device).unsqueeze(0), **scheduler_step_args)[0].squeeze(0) diff --git a/onetoall/controlnet.py b/onetoall/controlnet.py new file mode 100644 index 0000000..02725d1 --- /dev/null +++ b/onetoall/controlnet.py @@ -0,0 +1,440 @@ +import torch +from torch import nn +from torch.nn import functional as F +from einops import rearrange +import numpy as np +from typing import Tuple + +from .unet_causal_3d_blocks import get_down_block3d, CausalConv3d + +class ControlNetCausalConditioningEmbedding(nn.Module): + def __init__(self, conditioning_embedding_channels: int, conditioning_channels: int = 3, block_out_channels: Tuple[int, ...] = (16, 32, 96, 256)): + super().__init__() + self.conv_in = CausalConv3d(conditioning_channels, block_out_channels[0], kernel_size=3, padding=1) + self.blocks = nn.ModuleList([]) + + for i in range(len(block_out_channels) - 1): + channel_in = block_out_channels[i] + channel_out = block_out_channels[i + 1] + self.blocks.append(nn.Conv2d(channel_in, channel_in, kernel_size=3, padding=1)) + self.blocks.append(nn.Conv2d(channel_in, channel_out, kernel_size=3, padding=1, stride=2)) + + self.conv_out = nn.Conv2d(block_out_channels[-1], conditioning_embedding_channels, kernel_size=3, padding=1) + + def forward(self, conditioning): + embedding = self.conv_in(conditioning) + embedding = F.silu(embedding) + + for block in self.blocks: + embedding = block(embedding) + embedding = F.silu(embedding) + + embedding = self.conv_out(embedding) + + return embedding + +class MiniHunyuanEncoder(nn.Module): + ''' + a direct copy of hunyuan encoder + ''' + def __init__( + self, + in_channels = 3, + out_channels = 3, + down_block_types = ['DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D'], + block_out_channels = [128, 256, 512, 512], + layers_per_block = 2, + norm_num_groups = 32, + act_fn: str = "silu", + time_compression_ratio: int = 4, + spatial_compression_ratio: int = 8, + ): + super().__init__() + self.layers_per_block = layers_per_block + self.conv_in = CausalConv3d( + in_channels, block_out_channels[0], kernel_size=3, stride=1) + self.mid_block = None + self.down_blocks = nn.ModuleList([]) + + # down + output_channel = block_out_channels[0] + for i, down_block_type in enumerate(down_block_types): + input_channel = output_channel + output_channel = block_out_channels[i] + is_final_block = i == len(block_out_channels) - 1 + num_spatial_downsample_layers = int( + np.log2(spatial_compression_ratio)) + num_time_downsample_layers = int(np.log2(time_compression_ratio)) + + if time_compression_ratio == 4: + add_spatial_downsample = bool( + i < num_spatial_downsample_layers) + add_time_downsample = bool(i >= ( + len(block_out_channels) - 1 - num_time_downsample_layers) and not is_final_block) + elif time_compression_ratio == 8: + add_spatial_downsample = bool( + i < num_spatial_downsample_layers) + add_time_downsample = bool(i < num_time_downsample_layers) + else: + raise ValueError( + f"Unsupported time_compression_ratio: {time_compression_ratio}") + + downsample_stride_HW = (2, 2) if add_spatial_downsample else (1, 1) + downsample_stride_T = (2, ) if add_time_downsample else (1, ) + downsample_stride = tuple( + downsample_stride_T + downsample_stride_HW) + down_block = get_down_block3d( + down_block_type, + num_layers=self.layers_per_block, + in_channels=input_channel, + out_channels=output_channel, + add_downsample=bool( + add_spatial_downsample or add_time_downsample), + downsample_stride=downsample_stride, + resnet_eps=1e-6, + downsample_padding=0, + resnet_act_fn=act_fn, + resnet_groups=norm_num_groups, + ) + self.down_blocks.append(down_block) + + self.conv_out = CausalConv3d(block_out_channels[-1], out_channels, kernel_size=3) + + def forward(self, sample): + assert len(sample.shape) == 5, "The input tensor should have 5 dimensions" + sample = self.conv_in(sample) + # down + for down_block in self.down_blocks: + sample = down_block(sample) + sample = self.conv_out(sample) + return sample + + +class ControlNetConditioningEmbedding(nn.Module): + """ + Quoting from https://arxiv.org/abs/2302.05543: "Stable Diffusion uses a pre-processing method similar to VQ-GAN + [11] to convert the entire dataset of 512 × 512 images into smaller 64 × 64 “latent images” for stabilized + training. This requires ControlNets to convert image-based conditions to 64 × 64 feature space to match the + convolution size. We use a tiny network E(·) of four convolution layers with 4 × 4 kernels and 2 × 2 strides + (activated by ReLU, channels are 16, 32, 64, 128, initialized with Gaussian weights, trained jointly with the full + model) to encode image-space conditions ... into feature maps ..." + """ + + def __init__( + self, + conditioning_embedding_channels: int, + conditioning_channels: int = 3, + block_out_channels: Tuple[int, ...] = (16, 32, 96, 256), + ): + super().__init__() + + self.conv_in = nn.Conv2d(conditioning_channels, block_out_channels[0], kernel_size=3, padding=1) + + self.blocks = nn.ModuleList([]) + + for i in range(len(block_out_channels) - 1): + channel_in = block_out_channels[i] + channel_out = block_out_channels[i + 1] + self.blocks.append(nn.Conv2d(channel_in, channel_in, kernel_size=3, padding=1)) + self.blocks.append(nn.Conv2d(channel_in, channel_out, kernel_size=3, padding=1, stride=2)) + + self.conv_out = nn.Conv2d(block_out_channels[-1], conditioning_embedding_channels, kernel_size=3, padding=1) + + def forward(self, conditioning): + embedding = self.conv_in(conditioning) + embedding = F.silu(embedding) + + for block in self.blocks: + embedding = block(embedding) + embedding = F.silu(embedding) + + embedding = self.conv_out(embedding) + + return embedding + + +class InflatedGroupNorm(nn.GroupNorm): + def forward(self, x): + video_length = x.shape[2] + + x = rearrange(x, "b c f h w -> (b f) c h w") + x = super().forward(x) + x = rearrange(x, "(b f) c h w -> b c f h w", f=video_length) + + return x + +class InflatedConv3d(nn.Conv2d): + def forward(self, x): + video_length = x.shape[2] + + x = rearrange(x, "b c f h w -> (b f) c h w") + x = super().forward(x) + x = rearrange(x, "(b f) c h w -> b c f h w", f=video_length) + + return x + + +class ResnetBlockInflated(nn.Module): + def __init__(self, *, in_channels, out_channels=None, dropout=0.0, groups=32, groups_out=None, pre_norm=True, eps=1e-6, non_linearity="swish", output_scale_factor=1.0): + super().__init__() + self.pre_norm = pre_norm + self.pre_norm = True + self.in_channels = in_channels + out_channels = in_channels if out_channels is None else out_channels + self.out_channels = out_channels + self.output_scale_factor = output_scale_factor + + if groups_out is None: + groups_out = groups + + self.norm1 = InflatedGroupNorm(num_groups=groups, num_channels=in_channels, eps=eps, affine=True) + self.conv1 = InflatedConv3d(in_channels, out_channels, kernel_size=3, stride=1, padding=1) + self.norm2 = InflatedGroupNorm(num_groups=groups_out, num_channels=out_channels, eps=eps, affine=True) + self.dropout = torch.nn.Dropout(dropout) + self.conv2 = InflatedConv3d(out_channels, out_channels, kernel_size=3, stride=1, padding=1) + + if non_linearity == "swish": + self.nonlinearity = lambda x: F.silu(x) + elif non_linearity == "silu": + self.nonlinearity = nn.SiLU() + + def forward(self, input_tensor, temb): + if temb is not None: + print("Warning: temb is None in ResnetBlockInflated") + hidden_states = input_tensor + + hidden_states = self.norm1(hidden_states) + hidden_states = self.nonlinearity(hidden_states) + + hidden_states = self.conv1(hidden_states) + + if temb is not None: + hidden_states = hidden_states + temb + + hidden_states = self.norm2(hidden_states) + hidden_states = self.nonlinearity(hidden_states) + hidden_states = self.dropout(hidden_states) + hidden_states = self.conv2(hidden_states) + + output_tensor = (input_tensor + hidden_states) / self.output_scale_factor + + return output_tensor + +class DownEncoderBlockInflated(nn.Module): + def __init__(self, *, num_layers: int, in_channels: int, out_channels: int, add_downsample: bool, downsample_stride: tuple = (1, 2, 2), + resnet_eps: float = 1e-6, resnet_act_fn: str = "silu", resnet_groups: int = 32): + super().__init__() + + self.resnets = nn.ModuleList([ResnetBlockInflated( + in_channels=in_channels if i == 0 else out_channels, + out_channels=out_channels, + eps=resnet_eps, + non_linearity=resnet_act_fn, + groups=resnet_groups, + ) for i in range(num_layers)]) + + self.downsamplers = nn.ModuleList() + if add_downsample: + self.downsamplers.append( + InflatedConv3d( + out_channels, + out_channels, + kernel_size=3, + stride=2, + padding=1, + ) + ) + self.down_stride = downsample_stride + else: + self.down_stride = (1, 1, 1) + + def forward(self, x, temb=None): + for resnet in self.resnets: + x = resnet(x, temb) + + for down in self.downsamplers: + x = down(x) + return x + + +class SFT(nn.Module): # 2D SFT + def __init__( + self, in_channels, out_channels, intermediate_channels=128, groups=32, eps=1e-6): + super().__init__() + self.out_channels = out_channels + self.norm = InflatedGroupNorm(groups, out_channels, eps, affine=True) + self.mlp_shared = nn.Sequential(InflatedConv3d(in_channels, intermediate_channels, kernel_size=3, stride=1, padding=1), nn.SiLU()) + self.mlp_gamma = InflatedConv3d(intermediate_channels, out_channels, kernel_size=3, stride=1, padding=1) + self.mlp_beta = InflatedConv3d(intermediate_channels, out_channels, kernel_size=3, stride=1, padding=1) + + def forward(self, hidden_state, condition): + """ + hidden_state : (B, Cout, T, H, W) + condition : (B, Cin, 1, H, W) + """ + hidden_state = self.norm(hidden_state) #2D SFT 2D Norm + + actv = self.mlp_shared(condition) + gamma = self.mlp_gamma(actv) + beta = self.mlp_beta(actv) + + return torch.addcmul(beta, hidden_state, 1 + gamma) + + +class MiniEncoder2D(nn.Module): + + def __init__( + self, + in_channels: int = 3, + out_channels: int = 3, + down_block_types: list = ( + "DownEncoderBlockInflated", + "DownEncoderBlockInflated", + "DownEncoderBlockInflated", + "DownEncoderBlockInflated", + ), + block_out_channels: list = (128, 256, 512, 512), + layers_per_block: int = 2, + norm_num_groups: int = 32, + act_fn: str = "silu", + spatial_compression_ratio: int = 8, + ): + super().__init__() + + # ------------------------------------------------------------------- + # conv in + # ------------------------------------------------------------------- + self.conv_in = InflatedConv3d(in_channels, block_out_channels[0], kernel_size=3, stride=1, padding=1) + + self.down_blocks = nn.ModuleList() + output_channel = block_out_channels[0] + num_spatial_down_layers = int(np.log2(spatial_compression_ratio)) + + for i, block_type in enumerate(down_block_types): + input_channel = output_channel + output_channel = block_out_channels[i] + # is_final_block = i == len(block_out_channels) - 1 + + add_spatial_downsample = bool(i < num_spatial_down_layers) + + downsample_stride = (1, 2, 2) if add_spatial_downsample else (1, 1, 1) + + down_block = DownEncoderBlockInflated( + num_layers=layers_per_block, + in_channels=input_channel, + out_channels=output_channel, + add_downsample=add_spatial_downsample, + downsample_stride=downsample_stride, + resnet_eps=1e-6, + resnet_act_fn=act_fn, + resnet_groups=norm_num_groups, + ) + self.down_blocks.append(down_block) + + self.conv_out = InflatedConv3d(output_channel, out_channels, kernel_size=3, stride=1, padding=1) + + def forward(self, x): + # (B,C,1,H,W) + x = self.conv_in(x) + + for block in self.down_blocks: + x = block(x) + + return self.conv_out(x) + + +class Driven_Ref_PoseEncoder(nn.Module): + def __init__( + self, in_channels = 3, out_channels = 3, + down_block_types = ['DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D', 'DownEncoderBlockCausal3D'], + block_out_channels = [128, 256, 512, 512], layers_per_block = 2, norm_num_groups = 32, + act_fn: str = "silu", time_compression_ratio: int = 4, spatial_compression_ratio: int = 8, + ): + super().__init__() + self.layers_per_block = layers_per_block + self.conv_in = CausalConv3d(in_channels, block_out_channels[0], kernel_size=3, stride=1) + self.mid_block = None + self.down_blocks = nn.ModuleList([]) + + # down + output_channel = block_out_channels[0] + for i, down_block_type in enumerate(down_block_types): + input_channel = output_channel + output_channel = block_out_channels[i] + is_final_block = i == len(block_out_channels) - 1 + num_spatial_downsample_layers = int( + np.log2(spatial_compression_ratio)) + num_time_downsample_layers = int(np.log2(time_compression_ratio)) + + if time_compression_ratio == 4: + add_spatial_downsample = bool( + i < num_spatial_downsample_layers) + add_time_downsample = bool(i >= ( + len(block_out_channels) - 1 - num_time_downsample_layers) and not is_final_block) + elif time_compression_ratio == 8: + add_spatial_downsample = bool( + i < num_spatial_downsample_layers) + add_time_downsample = bool(i < num_time_downsample_layers) + else: + raise ValueError( + f"Unsupported time_compression_ratio: {time_compression_ratio}") + + downsample_stride_HW = (2, 2) if add_spatial_downsample else (1, 1) + downsample_stride_T = (2, ) if add_time_downsample else (1, ) + downsample_stride = tuple( + downsample_stride_T + downsample_stride_HW) + down_block = get_down_block3d( + down_block_type, + num_layers=self.layers_per_block, + in_channels=input_channel, + out_channels=output_channel, + add_downsample=bool( + add_spatial_downsample or add_time_downsample), + downsample_stride=downsample_stride, + resnet_eps=1e-6, + downsample_padding=0, + resnet_act_fn=act_fn, + resnet_groups=norm_num_groups, + attention_head_dim=output_channel, + ) + self.down_blocks.append(down_block) + + self.conv_out = CausalConv3d(block_out_channels[-1], out_channels, kernel_size=3) + + self.ref_pose_encoder = MiniEncoder2D( + in_channels = in_channels, + out_channels = out_channels, + block_out_channels = block_out_channels, + norm_num_groups = norm_num_groups, + layers_per_block = layers_per_block, + spatial_compression_ratio = spatial_compression_ratio, + ) + self.sft_layers = nn.ModuleList() + for i, ch in enumerate(block_out_channels): + if i == 0: # 0 层 (H/2,W/2) 不做 SFT + self.sft_layers.append(None) + else: # H/4、H/8、H/16 做 SFT + self.sft_layers.append( + SFT( + in_channels=ch, + out_channels=ch, + intermediate_channels=max(8, ch // 2), + groups=norm_num_groups, + ) + ) + + def forward(self, driven_pose, ref_pose): + # driven_pose b c t h w + # ref_pose b c 1 h w + ref_pose_cond, ref_feats = self.ref_pose_encoder(ref_pose) + + x = self.conv_in(driven_pose) + for i, down_block in enumerate(self.down_blocks): + x = down_block(x) + + if self.sft_layers[i] is not None: + cond_feat = ref_feats[i] + x = self.sft_layers[i](x, cond_feat) + + driven_pose_cond = self.conv_out(x) + return driven_pose_cond, ref_pose_cond diff --git a/onetoall/nodes.py b/onetoall/nodes.py new file mode 100644 index 0000000..519693f --- /dev/null +++ b/onetoall/nodes.py @@ -0,0 +1,150 @@ +import torch +from ..utils import log +import comfy.model_management as mm + +device = mm.get_torch_device() +offload_device = mm.unet_offload_device() + +class WanVideoAddOneToAllReferenceEmbeds: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "embeds": ("WANVIDIMAGE_EMBEDS",), + "vae": ("WANVAE", {"tooltip": "VAE model"}), + "ref_image": ("IMAGE",), + "strength": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 10.0, "step": 0.01, "tooltip": "Strength of the reference embedding"}), + "start_percent": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "Start percentage of the embedding application"}), + "end_percent": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "End percentage of the embedding application"}), + }, + "optional": { + "ref_mask": ("MASK",), + } + } + + RETURN_TYPES = ("WANVIDIMAGE_EMBEDS",) + RETURN_NAMES = ("image_embeds",) + FUNCTION = "add" + CATEGORY = "WanVideoWrapper" + + def add(self, embeds, vae, ref_image, strength, start_percent, end_percent, ref_mask=None): + updated = dict(embeds) + + ref_latent = ref_latent_empty = None + vae.to(device) + ref_image_in = (ref_image[..., :3].permute(3, 0, 1, 2) * 2 - 1).to(device, vae.dtype) + ref_latent = vae.encode([ref_image_in], device, tiled=False) + ref_mask_in = None + if ref_mask is not None: + ref_mask_in = (ref_mask.unsqueeze(0).repeat(3, 1, 1, 1) * 2 - 1.).to(device, vae.dtype) + else: + ref_mask_in = torch.zeros_like(ref_image_in)-1 + ref_mask_latent = vae.encode([ref_mask_in], device, tiled=False) + + if ref_mask is not None and not torch.all(ref_mask == 0): + ref_latent_empty = vae.encode([torch.zeros_like(ref_image_in)-1], device, tiled=False) + else: + ref_latent_empty = ref_mask_latent + + vae.to(offload_device) + + updated.setdefault("one_to_all_embeds", {}) + updated["one_to_all_embeds"]["ref_latent_pos"] = torch.cat([ref_latent, ref_latent_empty], dim=1) + updated["one_to_all_embeds"]["ref_latent_neg"] = torch.cat([ref_latent_empty, ref_latent_empty], dim=1) + updated["one_to_all_embeds"]["ref_strength"] = strength + updated["one_to_all_embeds"]["ref_start_percent"] = start_percent + updated["one_to_all_embeds"]["ref_end_percent"] = end_percent + + return (updated,) + +class WanVideoAddOneToAllPoseEmbeds: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "embeds": ("WANVIDIMAGE_EMBEDS",), + "pose_images": ("IMAGE", {"tooltip": "Pose images for the entire video"}), + "strength": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 10.0, "step": 0.01, "tooltip": "Strength of the pose control"}), + "start_percent": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "Start percentage of the pose control application"}), + "end_percent": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01, "tooltip": "End percentage of the pose control application"}), + }, + "optional": { + "pose_prefix_image": ("IMAGE",), + } + } + + RETURN_TYPES = ("WANVIDIMAGE_EMBEDS",) + RETURN_NAMES = ("image_embeds",) + FUNCTION = "add" + CATEGORY = "WanVideoWrapper" + + def add(self, embeds, pose_images, strength, pose_prefix_image=None, start_percent=0.0, end_percent=1.0): + updated = dict(embeds) + updated.setdefault("one_to_all_embeds", {}) + pose_images_in = pose_images[..., :3].unsqueeze(0).permute(0, 4, 1, 2, 3) * 2 - 1 # 1 B H W C -> B C 1 H W + updated["one_to_all_embeds"]["pose_images"] = pose_images_in + if pose_prefix_image is not None: + updated["one_to_all_embeds"]["pose_prefix_image"] = pose_prefix_image.unsqueeze(0).permute(0, 4, 1, 2, 3) * 2 - 1 # 1 B H W C -> B C 1 H W + else: + updated["one_to_all_embeds"]["pose_prefix_image"] = pose_images_in[:, :, :1] + + updated["one_to_all_embeds"]["controlnet_strength"] = strength + updated["one_to_all_embeds"]["controlnet_start_percent"] = start_percent + updated["one_to_all_embeds"]["controlnet_end_percent"] = end_percent + + return (updated,) + +class WanVideoAddOneToAllExtendEmbeds: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "embeds": ("WANVIDIMAGE_EMBEDS",), + "prev_latents": ("LATENT", {"tooltip": "Previous latents to be used to continue generation"}), + "window_size": ("INT", {"default": 81, "min": 1, "max": 256, "step": 1, "tooltip": "Number of new frames to generate" }), + "overlap": ("INT", {"default": 5, "min": 0, "max": 64, "step": 1, "tooltip": "Number of overlapping frames between previous and new frames" }), + "frames_processed": ("INT", {"default": 0, "min": 0, "max": 10000, "step": 1, "tooltip": "Number of frames already processed in the video" }), + "if_not_enough_frames": (["pad_with_last", "error"], {"default": "pad_with_last", "tooltip": "What to do if there are not enough frames in pose_images for the window"}), + }, + "optional": { + "pose_images": ("IMAGE", {"tooltip": "Pose images for the entire video"}), + } + } + + RETURN_TYPES = ("WANVIDIMAGE_EMBEDS", "IMAGE",) + RETURN_NAMES = ("image_embeds", "pose_slice",) + FUNCTION = "add" + CATEGORY = "WanVideoWrapper" + + def add(self, embeds, prev_latents, if_not_enough_frames, window_size=81, overlap=5, frames_processed=0, pose_images=None): + updated = dict(embeds) + updated.setdefault("one_to_all_embeds", {}) + updated["one_to_all_embeds"]["prev_latents"] = prev_latents["samples"][0] + if pose_images is not None: + pose_images_in = pose_images.clone()[..., :3] + start = max(0, frames_processed - overlap) + end = start + window_size + log.info(f"Extracting pose images from {start} to {end}") + if start >= pose_images_in.shape[0]: + raise ValueError(f"start index {start} exceeds pose images length {pose_images_in.shape[0]}") + if end > pose_images_in.shape[0]: + if if_not_enough_frames == "pad_with_last": + padding_needed = end - pose_images_in.shape[0] + pose_images_in = torch.cat([pose_images_in, pose_images_in[-1:].repeat(padding_needed, 1, 1, 1)], dim=0) + log.info(f"Not enough frames, padding with {padding_needed} frames to reach {end} total frames") + else: + raise ValueError(f"end index {end} exceeds pose images length {pose_images.shape[0]}") + pose_slice = pose_images_in[start:end] + else: + pose_slice = torch.zeros((1, 64, 64, 3)) + + return (updated, pose_slice) + + +NODE_CLASS_MAPPINGS = { + "WanVideoAddOneToAllReferenceEmbeds": WanVideoAddOneToAllReferenceEmbeds, + "WanVideoAddOneToAllPoseEmbeds": WanVideoAddOneToAllPoseEmbeds, + "WanVideoAddOneToAllExtendEmbeds": WanVideoAddOneToAllExtendEmbeds, + } +NODE_DISPLAY_NAME_MAPPINGS = { + "WanVideoAddOneToAllReferenceEmbeds": "WanVideo Add OneToAll Reference Embeds", + "WanVideoAddOneToAllPoseEmbeds": "WanVideo Add OneToAll Pose Embeds", + "WanVideoAddOneToAllExtendEmbeds": "WanVideo Add OneToAll Extend Embeds", + } \ No newline at end of file diff --git a/onetoall/refextractor_2d.py b/onetoall/refextractor_2d.py new file mode 100644 index 0000000..dcdb9b9 --- /dev/null +++ b/onetoall/refextractor_2d.py @@ -0,0 +1,220 @@ +from typing import Dict, Union + +import torch +import torch.nn as nn + +from ..wanvideo.modules.model import WanLayerNorm, WanSelfAttention, EmbedND_RifleX, sinusoidal_embedding_1d, apply_rotary_emb_split, apply_rope_comfy1 + +class WanAttentionBlock(nn.Module): + def __init__(self, in_features, out_features, ffn_dim, ffn2_dim, num_heads, qk_norm=True, cross_attn_norm=False, eps=1e-6, attention_mode="sdpa", rope_func="comfy", rms_norm_function="default"): + super().__init__() + self.dim = out_features + self.ffn_dim = ffn_dim + self.num_heads = num_heads + self.head_dim = out_features // num_heads + self.qk_norm = qk_norm + self.cross_attn_norm = cross_attn_norm + self.eps = eps + self.attention_mode = attention_mode + self.rope_func = rope_func + + # layers + self.norm1 = WanLayerNorm(self.dim, eps) + self.self_attn = WanSelfAttention(in_features, out_features, num_heads, qk_norm, eps, self.attention_mode, rms_norm_function=rms_norm_function, head_norm=False) + self.norm2 = WanLayerNorm(self.dim, eps) + self.ffn = nn.Sequential(nn.Linear(in_features, ffn_dim), nn.GELU(approximate='tanh'), nn.Linear(ffn2_dim, out_features)) + + self.modulation = nn.Parameter(torch.randn(1, 6, out_features) / in_features**0.5) + + + def get_mod(self, e, modulation): + if e.dim() == 3: + if e.shape[-1] == 512: + e = self.modulation(e) + return e.unsqueeze(2).chunk(6, dim=-1) + return (modulation + e).chunk(6, dim=1) # 1, 6, dim + elif e.dim() == 4: + e_mod = modulation.unsqueeze(2) + e + return [ei.squeeze(1) for ei in e_mod.unbind(dim=1)] + + + def modulate(self, norm_x, shift_msa, scale_msa): + return torch.addcmul(shift_msa, norm_x, 1 + scale_msa) + + def ffn_chunked(self, mod_x, num_chunks=4): + seq_len = mod_x.shape[1] + if seq_len <= 8192 or num_chunks <= 1: + return self.ffn(mod_x) + return torch.cat([self.ffn(chunk.contiguous()) for chunk in mod_x.chunk(num_chunks, dim=1)], dim=1) + + #region attention forward + def forward(self, x, e, seq_lens, freqs, split_rope=True, e_tr=None, tr_start=0, tr_num=0): + + use_token_replace = False + if e_tr is not None and tr_num > 0: + tr_shift_msa, tr_scale_msa, tr_gate_msa, tr_shift_mlp, tr_scale_mlp, tr_gate_mlp = self.get_mod(e_tr.to(x.device), self.modulation) + use_token_replace = True + tr_start = tr_start or 0 + tr_end = tr_start + (tr_num or 0) + + shift_msa, scale_msa, gate_msa, shift_mlp, scale_mlp, gate_mlp = self.get_mod(e.to(x.device), self.modulation) + del e + input_dtype = x.dtype + + if use_token_replace: + norm_x = self.norm1(x.to(shift_msa.dtype)) + input_x = torch.cat([ + torch.addcmul(shift_msa, norm_x[:, :tr_start], 1 + scale_msa), # before replace → T + torch.addcmul(tr_shift_msa, norm_x[:, tr_start:tr_end], 1 + tr_scale_msa), # replace segment → t=0 + torch.addcmul(shift_msa, norm_x[:, tr_end:], 1 + scale_msa) # after replace → T + ], dim=1).to(input_dtype) + else: + input_x = self.modulate(self.norm1(x.to(shift_msa.dtype)), shift_msa, scale_msa).to(input_dtype) + del shift_msa, scale_msa + + b, s, n, d = *x.shape[:2], self.self_attn.num_heads, self.self_attn.head_dim + h_dim = w_dim = 2 * (self.head_dim // 6) + t_dim = self.head_dim - h_dim - w_dim + + q = self.self_attn.norm_q(self.self_attn.q(input_x)).to(self.self_attn.norm_q.weight.dtype).view(b, s, n, d) + if split_rope: + q = apply_rotary_emb_split(q, freqs, t_dim) # Apply split rotary embedding (only to H/W dimensions, leaving T unchanged) + else: + q = apply_rope_comfy1(q, freqs) + + k = self.self_attn.norm_k(self.self_attn.k(input_x).to(self.self_attn.norm_k.weight.dtype)).to(input_x.dtype).view(b, s, n, d) + if split_rope: + k = apply_rotary_emb_split(k, freqs, t_dim) + else: + k = apply_rope_comfy1(k, freqs) + + v = self.self_attn.v(input_x).view(b, s, n, d) + del input_x + + y = self.self_attn.forward(q, k, v, seq_lens) + del q, k, v + if use_token_replace: + x = x + torch.cat([ + y[:, :tr_start] * gate_msa, + y[:, tr_start:tr_end] * tr_gate_msa, + y[:, tr_end:] * gate_msa + ], dim=1).to(input_dtype) + else: + x = x.addcmul(y, gate_msa) + del y, gate_msa + + # ffn + if use_token_replace: + norm2_x = self.norm2(x.to(shift_mlp.dtype)) + mod_x = torch.cat([ + torch.addcmul(shift_mlp, norm2_x[:, :tr_start], 1 + scale_mlp), + torch.addcmul(tr_shift_mlp, norm2_x[:, tr_start:tr_end], 1 + tr_scale_mlp), + torch.addcmul(shift_mlp, norm2_x[:, tr_end:], 1 + scale_mlp) + ], dim=1) + else: + mod_x = torch.addcmul(shift_mlp, self.norm2(x.to(shift_mlp.dtype)), 1 + scale_mlp) + del shift_mlp, scale_mlp + x_ffn = self.ffn_chunked(mod_x.to(input_dtype), num_chunks=1) + del mod_x + + # gate_mlp + if use_token_replace: + x = x + torch.cat([ + x_ffn[:, :tr_start] * gate_mlp, + x_ffn[:, tr_start:tr_end] * tr_gate_mlp, + x_ffn[:, tr_end:] * gate_mlp + ], dim=1).to(input_dtype) + else: + x = x.addcmul(x_ffn.to(gate_mlp.dtype), gate_mlp).to(input_dtype) + del gate_mlp + + return x + + +class WanRefextractor(nn.Module): + def __init__(self, patch_size=(1, 2, 2), in_dim=16, dim=5120, in_features=5120, out_features=5120, ffn_dim=8192, ffn2_dim=8192, + freq_dim=256, num_heads=16, num_layers=32, eps=1e-6, + qk_norm=True, cross_attn_norm=True, + attention_mode='sdpa', rope_func='comfy', rms_norm_function='default', + main_device=torch.device('cuda'), offload_device=torch.device('cpu'), dtype=torch.float16): + super().__init__() + self.patch_size = patch_size + self.freq_dim = freq_dim + self.dim = dim + self.main_device = main_device + self.base_dtype = dtype + self.attention_mode = attention_mode + self.patch_embedding = nn.Conv3d(in_dim, dim, kernel_size=patch_size, stride=patch_size) + self.time_embedding = nn.Sequential(nn.Linear(freq_dim, dim), nn.SiLU(), nn.Linear(dim, dim)) + self.time_projection = nn.Sequential(nn.SiLU(), nn.Linear(dim, dim * 6)) + + self.blocks = nn.ModuleList([ + WanAttentionBlock(in_features, out_features, ffn_dim, ffn2_dim, num_heads, + qk_norm, cross_attn_norm, eps, attention_mode="sdpa", rope_func=rope_func, rms_norm_function=rms_norm_function) + for i in range(num_layers) + ]) + + self.ref_blocks = nn.ModuleList([]) + for _ in range(len(self.blocks)+1): + self.ref_blocks.append(nn.Linear(in_features, out_features)) + + d = dim // num_heads + self.rope_embedder = EmbedND_RifleX(d,10000.0, [d - 4 * (d // 6), 2 * (d // 6), 2 * (d // 6)], num_frames=1, k=0) + + def rope_encode_comfy(self, t, h, w, freq_offset=0, t_start=0, steps_t=None, steps_h=None, steps_w=None, ntk_alphas=[1,1,1], device=None, dtype=None): + patch_size = self.patch_size + t_len = ((t + (patch_size[0] // 2)) // patch_size[0]) + h_len = ((h + (patch_size[1] // 2)) // patch_size[1]) + w_len = ((w + (patch_size[2] // 2)) // patch_size[2]) + + if steps_t is None: + steps_t = t_len + if steps_h is None: + steps_h = h_len + if steps_w is None: + steps_w = w_len + + img_ids = torch.zeros((steps_t, steps_h, steps_w, 3), device=device, dtype=dtype) + img_ids[:, :, :, 0] = img_ids[:, :, :, 0] + torch.linspace(t_start+freq_offset, t_start + (t_len - 1), steps=steps_t, device=device, dtype=dtype).reshape(-1, 1, 1) + img_ids[:, :, :, 1] = img_ids[:, :, :, 1] + torch.linspace(freq_offset, h_len - 1, steps=steps_h, device=device, dtype=dtype).reshape(1, -1, 1) + img_ids[:, :, :, 2] = img_ids[:, :, :, 2] + torch.linspace(freq_offset, w_len - 1, steps=steps_w, device=device, dtype=dtype).reshape(1, 1, -1) + img_ids = img_ids.reshape(1, -1, img_ids.shape[-1]) + + freqs = self.rope_embedder(img_ids, ntk_alphas).movedim(1, 2) + return freqs + + def forward( + self, + x: torch.Tensor, + timestep: torch.LongTensor, + ) -> Union[torch.Tensor, Dict[str, torch.Tensor]]: + B, C, F, H, W = x.shape + + freqs = self.rope_encode_comfy(F, H, W, device=x.device, dtype=x.dtype) + + self.patch_embedding.to(self.main_device) + x = self.patch_embedding(x.float()).to(x.dtype).flatten(2).transpose(1, 2).to(self.base_dtype) + + seq_lens = torch.tensor([u.size(1) for u in x], dtype=torch.int32) + + time_embed_dtype = self.time_embedding[0].weight.dtype + if time_embed_dtype not in [torch.float16, torch.bfloat16, torch.float32]: + time_embed_dtype = self.base_dtype + e = self.time_embedding(sinusoidal_embedding_1d(self.freq_dim, timestep.flatten()).to(time_embed_dtype)) # b, dim + e0 = self.time_projection(e).unflatten(1, (6, self.dim)).to(self.base_dtype) # b, 6, dim + del e + + # 4. Transformer blocks + block_samples = () + for block in self.blocks: + block_samples = block_samples + (x, ) + x = block(x, e0, seq_lens, freqs) + + block_samples = block_samples + (x, ) + + ref_block_samples = () + for block_sample, ref_block in zip(block_samples, self.ref_blocks): + block_sample = ref_block(block_sample) + ref_block_samples = ref_block_samples + (block_sample, ) + + return ref_block_samples, freqs diff --git a/onetoall/unet_causal_3d_blocks.py b/onetoall/unet_causal_3d_blocks.py new file mode 100644 index 0000000..6982e2c --- /dev/null +++ b/onetoall/unet_causal_3d_blocks.py @@ -0,0 +1,144 @@ +# Copyright 2024 The HuggingFace Team. All rights reserved. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ============================================================================== +# +# Modified from diffusers==0.29.2 +# +# ============================================================================== + +from typing import Optional +import torch +import torch.nn.functional as F +from torch import nn + +import comfy.ops +ops = comfy.ops.disable_weight_init + +def prepare_causal_attention_mask(n_frame: int, n_hw: int, dtype, device, batch_size: int = None): + seq_len = n_frame * n_hw + mask = torch.full((seq_len, seq_len), float( + "-inf"), dtype=dtype, device=device) + for i in range(seq_len): + i_frame = i // n_hw + mask[i, : (i_frame + 1) * n_hw] = 0 + if batch_size is not None: + mask = mask.unsqueeze(0).expand(batch_size, -1, -1) + return mask + + +class CausalConv3d(nn.Module): + def __init__(self, chan_in, chan_out, kernel_size, stride = 1, dilation = 1, pad_mode='replicate', **kwargs): + super().__init__() + self.pad_mode = pad_mode + padding = (kernel_size // 2, kernel_size // 2, kernel_size // 2, kernel_size // 2, kernel_size - 1, 0) # W, H, T + self.time_causal_padding = padding + self.conv = ops.Conv3d(chan_in, chan_out, kernel_size, stride=stride, dilation=dilation, **kwargs) + + def forward(self, x): + x = F.pad(x, self.time_causal_padding, mode=self.pad_mode) + return self.conv(x) + + +class DownsampleCausal3D(nn.Module): + def __init__(self, channels, use_conv=False, out_channels=None, padding=1, name="conv", kernel_size=3, bias=True, stride=2): + super().__init__() + self.channels, self.out_channels, self.use_conv, self.padding, self.name = channels, out_channels or channels, use_conv, padding, name + self.conv = CausalConv3d(self.channels, self.out_channels, kernel_size=kernel_size, stride=stride, bias=bias) + + def forward(self, x, scale=1.0): + return self.conv(x) + + +class ResnetBlockCausal3D(nn.Module): + def __init__(self, *, in_channels: int, out_channels: Optional[int] = None, groups: int = 32, eps: float = 1e-6, conv_3d_out_channels: Optional[int] = None): + super().__init__() + self.in_channels = in_channels + out_channels = in_channels if out_channels is None else out_channels + self.out_channels = out_channels + self.norm1 = torch.nn.GroupNorm(num_groups=groups, num_channels=in_channels, eps=eps, affine=True) + self.norm2 = torch.nn.GroupNorm(num_groups=groups, num_channels=out_channels, eps=eps, affine=True) + self.conv1 = CausalConv3d(in_channels, out_channels, kernel_size=3, stride=1) + conv_3d_out_channels = conv_3d_out_channels or out_channels + self.conv2 = CausalConv3d(out_channels, conv_3d_out_channels, kernel_size=3, stride=1) + + def forward(self, input_tensor: torch.FloatTensor, temb: torch.FloatTensor, scale: float = 1.0) -> torch.FloatTensor: + hidden_states = input_tensor + hidden_states = self.conv1(nn.SiLU()(self.norm1(hidden_states))) + if temb is not None: + hidden_states = hidden_states + temb + hidden_states = self.conv2(nn.SiLU()(self.norm2(hidden_states))) + return input_tensor + hidden_states + + +def get_down_block3d(down_block_type: str, num_layers: int, in_channels: int, out_channels: int, + add_downsample: bool, downsample_stride: int, resnet_eps: float, resnet_act_fn: str, resnet_groups: Optional[int] = None, + downsample_padding: Optional[int] = None, **kwargs): + + down_block_type = down_block_type[7:] if down_block_type.startswith( + "UNetRes") else down_block_type + if down_block_type == "DownEncoderBlockCausal3D": + return DownEncoderBlockCausal3D( + num_layers=num_layers, + in_channels=in_channels, + out_channels=out_channels, + add_downsample=add_downsample, + downsample_stride=downsample_stride, + resnet_eps=resnet_eps, + resnet_act_fn=resnet_act_fn, + resnet_groups=resnet_groups, + downsample_padding=downsample_padding, + ) + raise ValueError(f"{down_block_type} does not exist.") + + +class DownEncoderBlockCausal3D(nn.Module): + def __init__(self, in_channels: int, out_channels: int, num_layers: int = 1, resnet_eps: float = 1e-6, + resnet_groups: int = 32, add_downsample: bool = True, downsample_stride: int = 2, downsample_padding: int = 1, **kwargs): + super().__init__() + + resnets = [] + for i in range(num_layers): + in_channels = in_channels if i == 0 else out_channels + resnets.append( + ResnetBlockCausal3D( + in_channels=in_channels, + out_channels=out_channels, + eps=resnet_eps, + groups=resnet_groups, + ) + ) + + self.resnets = nn.ModuleList(resnets) + + if add_downsample: + self.downsamplers = nn.ModuleList([DownsampleCausal3D( + out_channels, + use_conv=True, + out_channels=out_channels, + padding=downsample_padding, + name="op", + stride=downsample_stride, + )]) + else: + self.downsamplers = None + + def forward(self, hidden_states: torch.FloatTensor, scale: float = 1.0) -> torch.FloatTensor: + for resnet in self.resnets: + hidden_states = resnet(hidden_states, temb=None, scale=scale) + + if self.downsamplers is not None: + for downsampler in self.downsamplers: + hidden_states = downsampler(hidden_states, scale) + + return hidden_states diff --git a/uni3c/controlnet.py b/uni3c/controlnet.py index b8b3934..dd2e713 100644 --- a/uni3c/controlnet.py +++ b/uni3c/controlnet.py @@ -115,16 +115,9 @@ class WanRotaryPosEmbed(nn.Module): freqs_w = freqs[2][:ppw].view(1, 1, ppw, -1).expand(ppf, pph, ppw, -1) freqs = torch.cat([freqs_f, freqs_h, freqs_w], dim=-1).reshape(1, 1, ppf * pph * ppw, -1) return freqs - + from ..wanvideo.modules.attention import sageattn_func -def zero_module(module): - # Zero out the parameters of a module and return it. - for p in module.parameters(): - p.detach().zero_() - return module - - class SimpleAttnProcessor2_0: def __init__(self, attention_mode): self.attention_mode = attention_mode @@ -278,7 +271,7 @@ class MaskCamEmbed(nn.Module): mid_channels = controlnet_cfg.get("mid_channels", 64) self.mask_proj = nn.Sequential(nn.Conv3d(add_channels, mid_channels, kernel_size=(4, 8, 8), stride=(4, 8, 8)), nn.GroupNorm(mid_channels // 8, mid_channels), nn.SiLU()) - self.mask_zero_proj = zero_module(nn.Conv3d(mid_channels, controlnet_cfg["conv_out_dim"], kernel_size=(1, 2, 2), stride=(1, 2, 2))) + self.mask_zero_proj = nn.Conv3d(mid_channels, controlnet_cfg["conv_out_dim"], kernel_size=(1, 2, 2), stride=(1, 2, 2)) def forward(self, add_inputs: torch.Tensor): # render_mask.shape [b,c,f,h,w] @@ -321,7 +314,7 @@ class WanControlNet(ModelMixin): ) self.proj_out = nn.ModuleList( [ - zero_module(nn.Linear(self.dim, 5120)) + nn.Linear(self.dim, 5120) for _ in range(controlnet_cfg["num_layers"]) ] ) diff --git a/wanvideo/modules/model.py b/wanvideo/modules/model.py index e94b31e..b162f06 100644 --- a/wanvideo/modules/model.py +++ b/wanvideo/modules/model.py @@ -29,6 +29,18 @@ from comfy import model_management as mm __all__ = ['WanModel'] +def apply_rotary_emb_split(hidden_states, freqs_cis, t_dim): + """Apply rotary embedding only to the spatial (H/W) dimensions, leaving temporal (T) unchanged.""" + t_part, hw_part = torch.split(hidden_states, [t_dim, hidden_states.shape[-1] - t_dim], dim=-1) + hw_freqs = freqs_cis[..., t_dim//2:, :, :] + + x_ = hw_part.to(dtype=hw_freqs.dtype).reshape(*hw_part.shape[:-1], -1, 1, 2) + x_out = hw_freqs[..., 0] * x_[..., 0] + x_out.addcmul_(hw_freqs[..., 1], x_[..., 1]) + out_hw = x_out.reshape(*hw_part.shape).type_as(hidden_states) + + return torch.cat([t_part, out_hw], dim=-1) + class AdaLayerNorm(nn.Module): def __init__(self, embedding_dim, output_dim=None, norm_elementwise_affine=False, norm_eps=1e-5): super().__init__() @@ -100,11 +112,6 @@ class FramePackMotioner(nn.Module):#from comfy.ldm.wan.model rope = torch.cat([rope_post, rope_2x, rope_4x], dim=1) return motion_lat, rope -def zero_module(module): - for p in module.parameters(): - p.detach().zero_() - return module - def torch_dfs(model: nn.Module, parent_name='root'): module_names, modules = [], [] current_name = parent_name if parent_name else 'root' @@ -450,7 +457,7 @@ class WanSelfAttention(nn.Module): def qkv_fn_v(self, x): b, s, n, d = *x.shape[:2], self.num_heads, self.head_dim return self.v(x).view(b, s, n, d) - + def qkv_fn_ip(self, x): b, s, n, d = *x.shape[:2], self.num_heads, self.head_dim q = self.norm_q(self.q(x) + self.q_loras(x).to(self.norm_q.weight.dtype)).to(x.dtype).view(b, s, n, d) @@ -458,7 +465,7 @@ class WanSelfAttention(nn.Module): v = (self.v(x) + self.v_loras(x)).view(b, s, n, d) return q, k, v - def forward(self, q, k, v, seq_lens, lynx_ref_feature=None, lynx_ref_scale=1.0, attention_mode_override=None): + def forward(self, q, k, v, seq_lens, lynx_ref_feature=None, lynx_ref_scale=1.0, attention_mode_override=None, onetoall_ref=None, onetoall_ref_scale=1.0): r""" Args: x(Tensor): Shape [B, L, num_heads, C / num_heads] @@ -478,9 +485,12 @@ class WanSelfAttention(nn.Module): if self.ref_adapter is not None and lynx_ref_feature is not None: x = x.add(ref_x, alpha=lynx_ref_scale) + if onetoall_ref is not None: + x = x.add(onetoall_ref, alpha=onetoall_ref_scale) + # output return self.o(x.flatten(2)) - + def forward_ip(self, q, k, v, q_ip, k_ip, v_ip, seq_lens, attention_mode_override=None): attention_mode = self.attention_mode if attention_mode_override is not None: @@ -883,6 +893,8 @@ class WanAttentionBlock(nn.Module): self.kv_cache = None self.use_motion_attn = use_motion_attn self.has_face_fuser_block = face_fuser_block + self.ref_attn_k_img = None + self.ref_attn_v_img = None # layers self.norm1 = WanLayerNorm(self.dim, eps) @@ -965,7 +977,7 @@ class WanAttentionBlock(nn.Module): return norm_x else: return torch.addcmul(shift_msa, norm_x, 1 + scale_msa) - + def ffn_chunked(self, mod_x, num_chunks=4): seq_len = mod_x.shape[1] if seq_len <= 8192 or num_chunks <= 1: @@ -996,6 +1008,8 @@ class WanAttentionBlock(nn.Module): lynx_x_ip=None, lynx_ref_feature=None, lynx_ip_scale=1.0, lynx_ref_scale=1.0, #lynx x_ovi=None, e_ovi=None, freqs_ovi=None, context_ovi=None, seq_lens_ovi=None, grid_sizes_ovi=None, num_cond_latents=None, #longcat image cond amount + x_onetoall_ref=None, onetoall_freqs=None, onetoall_ref=None, onetoall_ref_scale=1.0, #one-to-all + e_tr=None, tr_num=0, tr_start=0, #token replacement ): r""" Args: @@ -1012,6 +1026,13 @@ class WanAttentionBlock(nn.Module): self.seg_idx = [0, self.seg_idx, x.size(1)] e = e[0] + use_token_replace = False + if e_tr is not None and tr_num > 0: + tr_shift_msa, tr_scale_msa, tr_gate_msa, tr_shift_mlp, tr_scale_mlp, tr_gate_mlp = self.get_mod(e_tr.to(x.device), self.modulation) + use_token_replace = True + tr_start = tr_start or 0 + tr_end = tr_start + (tr_num or 0) + shift_msa, scale_msa, gate_msa, shift_mlp, scale_mlp, gate_mlp = self.get_mod(e.to(x.device), self.modulation) del e input_dtype = x.dtype @@ -1020,6 +1041,13 @@ class WanAttentionBlock(nn.Module): is_longcat = C == 4096 if is_longcat: input_x = self.modulate(self.norm1(x.view(B, T, -1, C).to(shift_msa.dtype)), shift_msa, scale_msa, seg_idx=self.seg_idx).to(input_dtype).view(B, N, C) + elif use_token_replace: + norm_x = self.norm1(x.to(shift_msa.dtype)) + input_x = torch.cat([ + torch.addcmul(shift_msa, norm_x[:, :tr_start], 1 + scale_msa), # before replace → T + torch.addcmul(tr_shift_msa, norm_x[:, tr_start:tr_end], 1 + tr_scale_msa), # replace segment → t=0 + torch.addcmul(shift_msa, norm_x[:, tr_end:], 1 + scale_msa) # after replace → T + ], dim=1).to(input_dtype) else: input_x = self.modulate(self.norm1(x.to(shift_msa.dtype)), shift_msa, scale_msa, seg_idx=self.seg_idx).to(input_dtype) @@ -1053,6 +1081,23 @@ class WanAttentionBlock(nn.Module): if lynx_ref_feature is None and self.self_attn.ref_adapter is not None: lynx_ref_feature = input_x + onetoall_ref = None + if x_onetoall_ref is not None: + b, s, n, d = *x_onetoall_ref.shape[:2], self.self_attn.num_heads, self.self_attn.head_dim + h_dim = w_dim = 2 * (self.head_dim // 6) + t_dim = self.head_dim - h_dim - w_dim + + q_ref = self.self_attn.norm_q(self.self_attn.q(input_x)).to(input_x.dtype).view(b, N, n, d) + q_ref = apply_rotary_emb_split(q_ref, freqs, t_dim) # Apply split rotary embedding (only to H/W dimensions, leaving T unchanged) + + k_ref = self.ref_attn_norm_k_img(self.ref_attn_k_img(x_onetoall_ref).to(self.ref_attn_norm_k_img.weight.dtype)).to(x_onetoall_ref.dtype).view(b, s, n, d) + k_ref = apply_rotary_emb_split(k_ref, onetoall_freqs, t_dim) + + v_ref = self.ref_attn_v_img(x_onetoall_ref).view(b, s, n, d) + + onetoall_ref = attention(q_ref, k_ref, v_ref, k_lens=seq_lens, attention_mode=self.attention_mode) + del q_ref, k_ref, v_ref + #RoPE and QKV computation if inner_t is not None: #query, key, value @@ -1104,10 +1149,10 @@ class WanAttentionBlock(nn.Module): # FETA if enhance_enabled: feta_scores = get_feta_scores(q, k) - + #self-attention - split_attn = (context is not None - and (context.shape[0] > 1 or (clip_embed is not None and clip_embed.shape[0] > 1)) + split_attn = (context is not None + and (context.shape[0] > 1 or (clip_embed is not None and clip_embed.shape[0] > 1)) and x.shape[0] == 1 and inner_t is None and x_ip is None # Don't split when using IP-Adapter @@ -1144,18 +1189,18 @@ class WanAttentionBlock(nn.Module): num_cond_latents_thw = num_cond_latents * (N // num_latent_frames) # process the condition tokens x_cond = self.self_attn.forward( - q[:, :num_cond_latents_thw].contiguous(), - k[:, :num_cond_latents_thw].contiguous(), - v[:, :num_cond_latents_thw].contiguous(), + q[:, :num_cond_latents_thw].contiguous(), + k[:, :num_cond_latents_thw].contiguous(), + v[:, :num_cond_latents_thw].contiguous(), seq_lens) # process the noise tokens x_noise = self.self_attn.forward(q[:, num_cond_latents_thw:].contiguous(), k, v, seq_lens) # merge x_cond and x_noise y = torch.cat([x_cond, x_noise], dim=1).contiguous() else: - y = self.self_attn.forward(q, k, v, seq_lens, lynx_ref_feature=lynx_ref_feature, lynx_ref_scale=lynx_ref_scale) + y = self.self_attn.forward(q, k, v, seq_lens, lynx_ref_feature=lynx_ref_feature, lynx_ref_scale=lynx_ref_scale, onetoall_ref=onetoall_ref, onetoall_ref_scale=onetoall_ref_scale) - del q, k, v + del q, k, v, # FETA if enhance_enabled: @@ -1173,17 +1218,23 @@ class WanAttentionBlock(nn.Module): ) # S2V - if zero_timestep: + if zero_timestep: z = [] for i in range(2): z.append(y[:, self.seg_idx[i]:self.seg_idx[i + 1]] * gate_msa[:, i:i + 1]) y = torch.cat(z, dim=1) x = x.add(y) else: - if not is_longcat: - x = x.addcmul(y, gate_msa) - else: + if is_longcat: x = x + (y.view(B, -1, N//T, C).float() * gate_msa).to(input_dtype).view(B, -1, C) + elif use_token_replace: + x = x + torch.cat([ + y[:, :tr_start] * gate_msa, + y[:, tr_start:tr_end] * tr_gate_msa, + y[:, tr_end:] * gate_msa + ], dim=1).to(input_dtype) + else: + x = x.addcmul(y, gate_msa) del y, gate_msa # cross-attention & ffn function @@ -1225,7 +1276,7 @@ class WanAttentionBlock(nn.Module): x = x.add(x_audio, alpha=audio_scale) # MTV-Crafter Motion Attention - if self.use_motion_attn and mtv_motion_tokens is not None and mtv_motion_rotary_emb is not None: + if self.use_motion_attn and mtv_motion_tokens is not None and mtv_motion_rotary_emb is not None: x_motion = self.motion_attn(self.norm4(x), mtv_motion_tokens, mtv_motion_rotary_emb, grid_sizes, mtv_freqs) x = x.add(x_motion, alpha=mtv_strength) @@ -1248,14 +1299,22 @@ class WanAttentionBlock(nn.Module): norm2_x = torch.cat(parts, dim=1) x_ffn = self.ffn(norm2_x) else: - if not is_longcat: - mod_x = torch.addcmul(shift_mlp, self.norm2(x.to(shift_mlp.dtype)), 1 + scale_mlp) - else: + if is_longcat: mod_x = torch.addcmul(shift_mlp, self.norm2(x.view(B, -1, N//T, C).float()), 1 + scale_mlp).view(B, -1, C) + elif use_token_replace: + norm2_x = self.norm2(x.to(shift_mlp.dtype)) + mod_x = torch.cat([ + torch.addcmul(shift_mlp, norm2_x[:, :tr_start], 1 + scale_mlp), + torch.addcmul(tr_shift_mlp, norm2_x[:, tr_start:tr_end], 1 + tr_scale_mlp), + torch.addcmul(shift_mlp, norm2_x[:, tr_end:], 1 + scale_mlp) + ], dim=1) + else: + mod_x = torch.addcmul(shift_mlp, self.norm2(x.to(shift_mlp.dtype)), 1 + scale_mlp) + del shift_mlp, scale_mlp x_ffn = self.ffn_chunked(mod_x.to(input_dtype), num_chunks=1) del mod_x - + # gate_mlp if zero_timestep: z = [] @@ -1264,10 +1323,16 @@ class WanAttentionBlock(nn.Module): x_ffn = torch.cat(z, dim=1) x = x.add(x_ffn) else: - if not is_longcat: - x = x.addcmul(x_ffn.to(gate_mlp.dtype), gate_mlp).to(input_dtype) - else: + if is_longcat: x = x + (gate_mlp * x_ffn.view(B, -1, N//T, C).float()).to(input_dtype).view(B, -1, C) + elif use_token_replace: + x = x + torch.cat([ + x_ffn[:, :tr_start] * gate_mlp, + x_ffn[:, tr_start:tr_end] * tr_gate_mlp, + x_ffn[:, tr_end:] * gate_mlp + ], dim=1).to(input_dtype) + else: + x = x.addcmul(x_ffn.to(gate_mlp.dtype), gate_mlp).to(input_dtype) del gate_mlp if x_ip is not None: #stand-in @@ -1389,7 +1454,7 @@ class BaseWanAttentionBlock(WanAttentionBlock): x, x_ip, lynx_ref_feature, x_ovi = super().forward(x, **kwargs) if vace_hints is None: return x, x_ip, lynx_ref_feature, x_ovi - + if self.block_id is not None: for i in range(len(vace_hints)): x.add_(vace_hints[i][self.block_id].to(x.device), alpha=vace_context_scale[i]) @@ -1419,17 +1484,26 @@ class Head(nn.Module): e = (self.modulation.unsqueeze(2) + e.unsqueeze(1)).chunk(2, dim=1) return [ei.squeeze(1) for ei in e] - def forward(self, x, e, **kwargs): + def forward(self, x, e, e_tr=None, tr_start=0, tr_num=0, **kwargs): r""" Args: x(Tensor): Shape [B, L1, C] e(Tensor): Shape [B, C] """ - e = self.get_mod(e.to(x.device)) - x = self.head(self.norm(x.float()).to(x.dtype).mul_(1 + e[1]).add_(e[0])) + if tr_num > 0 and e_tr is not None: + e_tr = self.get_mod(e_tr.to(x.device)) + tr_end = tr_start + tr_num + norm_x = self.norm(x.float()).to(x.dtype) + x = self.head(torch.cat([ + norm_x[:, :tr_start].mul(1 + e[1]).add(e[0]), + norm_x[:, tr_start:tr_end].mul(1 + e_tr[1]).add(e_tr[0]), + norm_x[:, tr_end:].mul(1 + e[1]).add(e[0]) + ], dim=1)) + else: + x = self.head(self.norm(x.float()).to(x.dtype).mul_(1 + e[1]).add_(e[0])) return x - + class Head_adaLN(nn.Module): def __init__(self, dim, out_dim, patch_size, eps=1e-6, adaln_tembed_dim=512): @@ -1459,7 +1533,7 @@ class Head_adaLN(nn.Module): self.modulation.to(torch.float32) shift, scale = self.modulation(e).unsqueeze(2).chunk(2, dim=-1) # [B, T, 1, C] return self.head(self.norm(x.view(B, T, -1, C).float()).mul_(1 + scale).add_(shift).view(B, N, C).to(x.dtype)) - + class MLPProj(torch.nn.Module): @@ -1732,7 +1806,7 @@ class WanModel(torch.nn.Module): nn.SiLU(), ConvMLP(dim, dim * 4, kernel_size=7, padding=3), ) - + self.original_patch_embedding = self.patch_embedding self.expanded_patch_embedding = self.patch_embedding @@ -1803,18 +1877,12 @@ class WanModel(torch.nn.Module): self.head = Head_adaLN(dim, out_dim, patch_size, eps, adaln_tembed_dim=512) d = self.dim // self.num_heads - self.rope_embedder = EmbedND_RifleX( - d, - 10000.0, - [d - 4 * (d // 6), 2 * (d // 6), 2 * (d // 6)], - num_frames=None, - k=None, - ) + self.rope_embedder = EmbedND_RifleX(d, 10000.0, [d - 4 * (d // 6), 2 * (d // 6), 2 * (d // 6)], num_frames=None, k=None) self.cached_freqs = self.cached_shape = self.cached_cond = None # buffers (don't use register_buffer otherwise dtype will be changed in to()) assert (dim % num_heads) == 0 and (dim // num_heads) % 2 == 0 - + if model_type == 'i2v' or model_type == 'fl2v': self.img_emb = MLPProj(1280, dim, fl_pos_emb=model_type == 'fl2v') @@ -2149,7 +2217,8 @@ class WanModel(torch.nn.Module): flashvsr_LQ_latent=None, flashvsr_strength=1.0, num_cond_latents=None, add_text_emb=None, - sdancer_input=None # SteadyDancer + sdancer_input=None, # SteadyDancer + one_to_all_input=None, # One-to-All ): r""" Forward pass through the diffusion model @@ -2173,9 +2242,9 @@ class WanModel(torch.nn.Module): List of denoised video tensors with original input shapes [C_out, F, H / 8, W / 8] """ # Stand-In only used on first positive pass, then cached in kv_cache - if is_uncond or current_step > 0: + if is_uncond or current_step > 0: standin_input = None - + # MTV Crafter motion projection if mtv_motion_tokens is not None: bs, motion_seq_len = mtv_motion_tokens.shape[0], mtv_motion_tokens.shape[1] @@ -2213,14 +2282,14 @@ class WanModel(torch.nn.Module): lynx_ip_scale = lynx_embeds.get("ip_scale", 1.0) lynx_ref_scale = lynx_embeds.get("ref_scale", 1.0) - + #s2v if self.model_type == 's2v' and s2v_audio_input is not None: if is_uncond: s2v_audio_input = s2v_audio_input * 0 # to match original code s2v_audio_input = torch.cat([s2v_audio_input[..., 0:1].repeat(1, 1, 1, s2v_motion_frames[0]), s2v_audio_input], dim=-1) - + audio_emb_res = self.casual_audio_encoder(s2v_audio_input) if self.enable_adain: audio_emb_global, audio_emb = audio_emb_res @@ -2241,7 +2310,7 @@ class WanModel(torch.nn.Module): if sdancer_input is not None and sdancer_input['start_percent'] <= current_step_percentage <= sdancer_input['end_percent']: sdancer_enabled = True x_noise_clone = torch.stack(x) - + # I2V if y is not None: if hasattr(self, "randomref_embedding_pose") and unianim_data is not None: @@ -2251,6 +2320,40 @@ class WanModel(torch.nn.Module): y[0].add_(random_ref_emb, alpha=unianim_data["strength"]) x = [torch.cat([u, v], dim=0) for u, v in zip(x, y)] + suffix_frames = x[0].shape[1] + prefix_frames = 0 + + # One-to-all-Animation + onetoall_ref_block_samples = onetoall_freqs = prev_x = prev_control = None + onetoall_ref_scale = 1.0 + onetoall_control_enabled = use_token_replace = False + e0_token_replace = token_replace_start = None + replace_token_num = token_replace_start = 0 + if one_to_all_input is not None: + # reference condition + ref_cond_latent = one_to_all_input.get("ref_latent_pos", None) if not is_uncond else one_to_all_input.get("ref_latent_neg", None) + if ref_cond_latent is not None and one_to_all_input['ref_start_percent'] <= current_step_percentage <= one_to_all_input['ref_end_percent']: + onetoall_ref_scale = one_to_all_input.get("ref_strength", 1.0) + image_cond = self.image_to_cond(ref_cond_latent.to(self.main_device, self.base_dtype))[0] + x = [torch.cat([v, u], dim=1) for v, u in zip([image_cond], x)] + seq_len += math.ceil((image_cond.shape[-1] * image_cond.shape[-2]) / 4 * image_cond.shape[-3]) + F += 1 + prefix_frames = 1 + suffix_frames += 1 + onetoall_ref_block_samples, onetoall_freqs = self.refextractor(ref_cond_latent, timestep=t) + # pose controlnet + controlnet_tokens = one_to_all_input.get("controlnet_tokens", None) + if not is_uncond and controlnet_tokens is not None and one_to_all_input['controlnet_start_percent'] <= current_step_percentage <= one_to_all_input['controlnet_end_percent']: + onetoall_control_enabled = True + onetoall_control_strength = one_to_all_input.get("controlnet_strength", 1.0) + # token replace + if one_to_all_input.get("token_replace", False): + use_token_replace = True + num_latent_frames_to_replace = one_to_all_input.get("num_latent_frames_to_replace", 2) + t_token_replace = torch.zeros_like(t) + token_replace_start = (H // self.patch_size[1]) * (W // self.patch_size[2]) # skip first (ref) frame + replace_token_num = num_latent_frames_to_replace * token_replace_start # zero next frames + #uni3c controlnet if uni3c_data is not None: render_latent = uni3c_data["render_latent"].to(self.base_dtype) @@ -2278,15 +2381,13 @@ class WanModel(torch.nn.Module): else: self.original_patch_embedding.to(self.main_device) x = [self.original_patch_embedding(u.unsqueeze(0).to(torch.float32)).to(x[0].dtype) for u in x] - - orig_frames = x[0].shape[1] # ovi audio model if self.audio_model is not None: x_ovi = [self.audio_model.original_patch_embedding(u.unsqueeze(0).to(torch.float32)).to(x_ovi[0].dtype) for u in x_ovi] grid_sizes_ovi = torch.stack([torch.tensor(u.shape[1:2], dtype=torch.long) for u in x_ovi]) seq_lens_ovi = torch.tensor([u.size(1) for u in x_ovi], dtype=torch.int32) - x_ovi = torch.cat([torch.cat([u, u.new_zeros(1, seq_len_ovi - u.size(1), u.size(2))], dim=1) for u in x_ovi]) + x_ovi = torch.cat([torch.cat([u, u.new_zeros(1, seq_len_ovi - u.size(1), u.size(2))], dim=1) for u in x_ovi]) d = self.dim // self.num_heads freqs_ovi = rope_params(1024, d - 4 * (d // 6), freqs_scaling=0.19676).to(self.main_device) x_ovi = x_ovi.to(self.main_device, self.base_dtype) @@ -2375,7 +2476,7 @@ class WanModel(torch.nn.Module): seq_len += end_ref_latent_seq_len x = [torch.cat([u, end_ref_latent.unsqueeze(0)], dim=1) for end_ref_latent, u in zip(end_ref_latent, x)] - + x = torch.cat([torch.cat([u, u.new_zeros(1, seq_len - u.size(1), u.size(2))], dim=1) for u in x]) if self.trainable_cond_mask is not None: @@ -2397,11 +2498,11 @@ class WanModel(torch.nn.Module): if freqs is None and "comfy" in self.rope_func: #comfy rope current_shape = (F, H, W) - + has_cond = attn_cond is not None - if (self.cached_freqs is not None and - self.cached_shape == current_shape and + if (self.cached_freqs is not None and + self.cached_shape == current_shape and self.cached_cond == has_cond and self.cached_rope_k == self.rope_embedder.k and self.cached_ntk_alphas == ntk_alphas @@ -2411,9 +2512,9 @@ class WanModel(torch.nn.Module): freqs = self.rope_encode_comfy(F, H, W, freq_offset=freq_offset, ntk_alphas=ntk_alphas, attn_cond_shape=attn_cond_shape, device=x.device, dtype=x.dtype) if s2v_ref_latent is not None: freqs_ref = self.rope_encode_comfy( - s2v_ref_latent.shape[2], - s2v_ref_latent.shape[3], - s2v_ref_latent.shape[4], + s2v_ref_latent.shape[2], + s2v_ref_latent.shape[3], + s2v_ref_latent.shape[4], t_start=max(30, F + 9), device=x.device, dtype=x.dtype) freqs = torch.cat([freqs, freqs_ref], dim=1) @@ -2463,6 +2564,9 @@ class WanModel(torch.nn.Module): time_embed_dtype = self.base_dtype e = self.time_embedding(sinusoidal_embedding_1d(self.freq_dim, t.flatten()).to(time_embed_dtype)) # b, dim e0 = self.time_projection(e).unflatten(1, (6, self.dim)) # b, 6, dim + if use_token_replace: + e_token_replace = self.time_embedding(sinusoidal_embedding_1d(self.freq_dim, t_token_replace.flatten()).to(time_embed_dtype)) # b, dim + e0_token_replace = self.time_projection(e_token_replace).unflatten(1, (6, self.dim)) # b, 6, dim else: time_embed_dtype = self.time_embedding.mlp[0].weight.dtype if time_embed_dtype not in [torch.float16, torch.bfloat16, torch.float32]: @@ -2480,7 +2584,7 @@ class WanModel(torch.nn.Module): last_timestep = t[:, -1:] padding = last_timestep.expand(t.size(0), seq_len_ovi - t.size(1)) t_ovi = torch.cat([t, padding], dim=1) - + e_ovi = self.audio_model.time_embedding(sinusoidal_embedding_1d(self.audio_model.freq_dim, t_ovi.flatten()).to(time_embed_dtype)).unsqueeze(0) # b, dim e0_ovi = self.audio_model.time_projection(e_ovi).unflatten(2, (6, self.dim)).movedim(1, 2) # B, seq_len, 6, dim else: @@ -2516,14 +2620,14 @@ class WanModel(torch.nn.Module): if expanded_timesteps: e = e.view(b, f, 1, 1, self.dim).expand(b, f, grid_sizes[0][1], grid_sizes[0][2], self.dim) e0 = e0.view(b, f, 1, 1, 6, self.dim).expand(b, f, grid_sizes[0][1], grid_sizes[0][2], 6, self.dim) - + e = e.flatten(1, 3) e0 = e0.flatten(1, 3) - + e0 = e0.transpose(1, 2) if not e0.is_contiguous(): e0 = e0.contiguous() - + e = e.to(self.offload_device, non_blocking=self.use_non_blocking) # clip vision embedding @@ -2580,7 +2684,7 @@ class WanModel(torch.nn.Module): [u, u.new_zeros(self.text_len - u.size(0), u.size(1))]) for u in nag_context ]).to(text_embed_dtype)) - + if self.offload_txt_emb: self.text_embedding.to(self.offload_device, non_blocking=self.use_non_blocking) @@ -2637,7 +2741,7 @@ class WanModel(torch.nn.Module): accumulated_rel_l1_distance = torch.tensor(0.0, dtype=torch.float32, device=device) if pred_id is None: pred_id = self.teacache_state.new_prediction(cache_device=self.cache_device) - should_calc = True + should_calc = True else: previous_modulated_input = self.teacache_state.get(pred_id)['previous_modulated_input'] previous_modulated_input = previous_modulated_input.to(device) @@ -2665,7 +2769,7 @@ class WanModel(torch.nn.Module): accumulated_rel_l1_distance = accumulated_rel_l1_distance.to(self.cache_device) previous_modulated_input = e.to(self.cache_device).clone() if (self.teacache_use_coefficients and self.teacache_mode == 'e') else e0.to(self.cache_device).clone() - + if not should_calc: x = x.to(previous_residual.dtype) + previous_residual.to(x.device) self.teacache_state.update( @@ -2803,7 +2907,11 @@ class WanModel(torch.nn.Module): lynx_x_ip=lynx_x_ip, lynx_ip_scale=lynx_ip_scale, lynx_ref_scale=lynx_ref_scale, - num_cond_latents=num_cond_latents + num_cond_latents=num_cond_latents, + onetoall_ref_scale=onetoall_ref_scale, + e_tr=e0_token_replace if use_token_replace else None, + tr_start=token_replace_start, + tr_num=replace_token_num, ) if self.audio_model is not None: kwargs['e_ovi'] = e0_ovi.to(self.base_dtype) @@ -2812,7 +2920,7 @@ class WanModel(torch.nn.Module): kwargs['seq_lens_ovi'] = seq_lens_ovi kwargs['freqs_ovi'] = freqs_ovi - + if vace_data is not None: vace_hint_list = [] vace_scale_list = [] @@ -2828,7 +2936,7 @@ class WanModel(torch.nn.Module): vace_hints = self.forward_vace(x, vace_data, seq_len, kwargs) vace_hint_list.append(vace_hints) vace_scale_list.append(1.0) - + kwargs['vace_hints'] = vace_hint_list kwargs['vace_context_scale'] = vace_scale_list @@ -2898,7 +3006,17 @@ class WanModel(torch.nn.Module): if b in self.slg_blocks and is_uncond: if self.slg_start_percent <= current_step_percentage <= self.slg_end_percent: continue - x, x_ip, lynx_ref_feature, x_ovi = block(x, x_ip=x_ip, lynx_ref_feature=lynx_ref_feature, x_ovi=x_ovi, **kwargs) #run block + + x_onetoall_ref = None + if onetoall_ref_block_samples is not None: + interval_ref = len(self.blocks) / len(onetoall_ref_block_samples) + interval_ref = int(np.ceil(interval_ref)) + x_onetoall_ref = onetoall_ref_block_samples[b // interval_ref] + + # ---run block----# + x, x_ip, lynx_ref_feature, x_ovi = block(x, x_ip=x_ip, lynx_ref_feature=lynx_ref_feature, x_ovi=x_ovi, x_onetoall_ref=x_onetoall_ref, onetoall_freqs=onetoall_freqs, **kwargs) + # ---post block----# + if self.audio_injector is not None and s2v_audio_input is not None: x = self.audio_injector_forward(b, x, merged_audio_emb, scale=s2v_audio_scale) #s2v if block.has_face_fuser_block and motion_vec is not None: @@ -2924,6 +3042,31 @@ class WanModel(torch.nn.Module): #controlnet if (controlnet is not None) and (b % controlnet["controlnet_stride"] == 0) and (b // controlnet["controlnet_stride"] < len(controlnet["controlnet_states"])): x[:, :self.original_seq_len] += controlnet["controlnet_states"][b // controlnet["controlnet_stride"]].to(x) * controlnet["controlnet_weight"] + # One-to-All-Animation controlnet + if onetoall_control_enabled: + if prev_x is not None and (b - 1) < len(self.controlnet.blocks): + tqdm.write(f"Applying One-to-All ControlNet at block {b}") + if b == 1: + ctrl_in = prev_x + controlnet_tokens + elif prev_control is not None: + ctrl_in = prev_control + + self.controlnet.blocks[b - 1].to(self.main_device) + control_out = self.controlnet.blocks[b - 1](ctrl_in, e0, seq_lens, freqs, e_tr=e0_token_replace, tr_num=replace_token_num,tr_start=token_replace_start, split_rope=False) + self.controlnet.blocks[b - 1].to(self.offload_device, non_blocking=self.use_non_blocking) + prev_control = control_out + + control_out_proj = self.controlnet_zero[b - 1](control_out) + x = x + control_out_proj * onetoall_control_strength + if b < len(self.controlnet.blocks): # Store prev_x only while controlnet is active + prev_x = x + elif b == len(self.controlnet.blocks): # Controlnet done, free memory + prev_x = None + prev_control = None + if controlnet_tokens is not None: + del controlnet_tokens + controlnet_tokens = None + mm.soft_empty_cache() if lynx_ref_feature_extractor: return lynx_ref_buffer @@ -2953,20 +3096,20 @@ class WanModel(torch.nn.Module): accumulated_error = 0.0, cache_ovi = x_ovi.clone().to(original_x.device) - original_x_ovi if x_ovi is not None else None ) - - + + if self.enable_easycache and (self.easycache_start_step <= current_step <= self.easycache_end_step) and pred_id is not None: self.easycache_state.update( pred_id, previous_raw_output=x.clone(), ) - + if self.ref_conv is not None and fun_ref is not None: fun_ref_length = fun_ref.size(1) x = x[:, fun_ref_length:] grid_sizes = torch.stack([torch.tensor([u[0] - 1, u[1], u[2]]) for u in grid_sizes]).to(grid_sizes.device) - + if end_ref_latent is not None: end_ref_latent_length = end_ref_latent.size(1) x = x[:, :-end_ref_latent_length] @@ -2976,10 +3119,11 @@ class WanModel(torch.nn.Module): x = x[:, :self.original_seq_len] grid_sizes = torch.stack([torch.tensor([u[0] - 1, u[1], u[2]]) for u in grid_sizes]).to(grid_sizes.device) - + x = x[:, :self.original_seq_len] - x = self.head(x, e.to(x.device), temp_length=F) + x = self.head(x, e.to(x.device), temp_length=F, + e_tr=e_token_replace.to(x.device) if use_token_replace else None, tr_start=token_replace_start, tr_num=replace_token_num) if x_ovi is not None: x_ovi = self.audio_model.head(x_ovi, e_ovi.to(x_ovi.device)) @@ -2987,9 +3131,9 @@ class WanModel(torch.nn.Module): assert len(x) == len(grid_sizes_ovi) x_ovi = [u[:gs] for u, gs in zip(x_ovi, grid_sizes_ovi)] x_ovi = [u.float() for u in x_ovi] - - x = self.unpatchify(x, original_grid_sizes) # type: ignore[arg-type] - x = [u[:, :orig_frames, ...].float() for u in x] + + x = self.unpatchify(x, original_grid_sizes) + x = [u[:, prefix_frames:suffix_frames, ...].float() for u in x] return (x, x_ovi, pred_id) if pred_id is not None else (x, x_ovi, None) def unpatchify(self, x, grid_sizes):