diff --git a/example_workflows/wanvideo_WanAnimate_example_01.json b/example_workflows/wanvideo_WanAnimate_example_01.json index 9e69f54..dcc67c2 100644 --- a/example_workflows/wanvideo_WanAnimate_example_01.json +++ b/example_workflows/wanvideo_WanAnimate_example_01.json @@ -730,8 +730,7 @@ "name": "INT", "type": "INT", "links": [ - 268, - 284 + 268 ] } ], @@ -1824,13 +1823,13 @@ "hidden": false, "paused": false, "params": { - "filename": "WanVideo2_1_T2V_00011.mp4", + "filename": "WanVideo2_1_T2V_00020.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "WanVideo2_1_T2V_00011.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00011.mp4" + "workflow": "WanVideo2_1_T2V_00020.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00020.mp4" } } } @@ -2407,7 +2406,7 @@ "links": null }, { - "label": "109 count", + "label": "154 count", "name": "count", "type": "INT", "links": null @@ -2498,159 +2497,6 @@ null ] }, - { - "id": 27, - "type": "WanVideoSampler", - "pos": [ - 1626.8492431640625, - -523.2383422851562 - ], - "size": [ - 315, - 874.1923217773438 - ], - "flags": {}, - "order": 50, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "WANVIDEOMODEL", - "link": 63 - }, - { - "name": "image_embeds", - "type": "WANVIDIMAGE_EMBEDS", - "link": 84 - }, - { - "name": "text_embeds", - "shape": 7, - "type": "WANVIDEOTEXTEMBEDS", - "link": 281 - }, - { - "name": "samples", - "shape": 7, - "type": "LATENT", - "link": null - }, - { - "name": "feta_args", - "shape": 7, - "type": "FETAARGS", - "link": null - }, - { - "name": "context_options", - "shape": 7, - "type": "WANVIDCONTEXT", - "link": 283 - }, - { - "name": "cache_args", - "shape": 7, - "type": "CACHEARGS", - "link": null - }, - { - "name": "flowedit_args", - "shape": 7, - "type": "FLOWEDITARGS", - "link": null - }, - { - "name": "slg_args", - "shape": 7, - "type": "SLGARGS", - "link": null - }, - { - "name": "loop_args", - "shape": 7, - "type": "LOOPARGS", - "link": null - }, - { - "name": "experimental_args", - "shape": 7, - "type": "EXPERIMENTALARGS", - "link": null - }, - { - "name": "sigmas", - "shape": 7, - "type": "SIGMAS", - "link": null - }, - { - "name": "unianimate_poses", - "shape": 7, - "type": "UNIANIMATE_POSE", - "link": null - }, - { - "name": "fantasytalking_embeds", - "shape": 7, - "type": "FANTASYTALKING_EMBEDS", - "link": null - }, - { - "name": "uni3c_embeds", - "shape": 7, - "type": "UNI3C_EMBEDS", - "link": null - }, - { - "name": "multitalk_embeds", - "shape": 7, - "type": "MULTITALK_EMBEDS", - "link": null - }, - { - "name": "freeinit_args", - "shape": 7, - "type": "FREEINITARGS", - "link": null - } - ], - "outputs": [ - { - "name": "samples", - "type": "LATENT", - "slot_index": 0, - "links": [ - 33 - ] - }, - { - "name": "denoised_samples", - "type": "LATENT", - "links": null - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", - "Node name for S&R": "WanVideoSampler" - }, - "widgets_values": [ - 6, - 1, - 5, - 42, - "fixed", - true, - "dpm++_sde", - 0, - 1, - "", - "comfy", - 0, - -1, - false - ] - }, { "id": 110, "type": "WanVideoContextOptions", @@ -2677,9 +2523,7 @@ { "name": "context_options", "type": "WANVIDCONTEXT", - "links": [ - 283 - ] + "links": [] } ], "properties": { @@ -2697,174 +2541,6 @@ "linear" ] }, - { - "id": 62, - "type": "WanVideoAnimateEmbeds", - "pos": [ - 1086.8922119140625, - -380.2015380859375 - ], - "size": [ - 274.2164001464844, - 370 - ], - "flags": {}, - "order": 43, - "mode": 0, - "inputs": [ - { - "name": "vae", - "type": "WANVAE", - "link": 280 - }, - { - "name": "clip_embeds", - "shape": 7, - "type": "WANVIDIMAGE_CLIPEMBEDS", - "link": 96 - }, - { - "name": "ref_images", - "shape": 7, - "type": "IMAGE", - "link": 282 - }, - { - "name": "pose_images", - "shape": 7, - "type": "IMAGE", - "link": 244 - }, - { - "name": "face_images", - "shape": 7, - "type": "IMAGE", - "link": 240 - }, - { - "name": "bg_images", - "shape": 7, - "type": "IMAGE", - "link": 234 - }, - { - "name": "mask", - "shape": 7, - "type": "MASK", - "link": 247 - }, - { - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": 265 - }, - { - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": 266 - }, - { - "name": "num_frames", - "type": "INT", - "widget": { - "name": "num_frames" - }, - "link": 268 - }, - { - "name": "frame_window_size", - "type": "INT", - "widget": { - "name": "frame_window_size" - }, - "link": 284 - } - ], - "outputs": [ - { - "name": "image_embeds", - "type": "WANVIDIMAGE_EMBEDS", - "links": [ - 84 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0", - "Node name for S&R": "WanVideoAnimateEmbeds" - }, - "widgets_values": [ - 832, - 480, - 501, - false, - 77, - "disabled", - 1, - 1, - false - ], - "color": "#322", - "bgcolor": "#533" - }, - { - "id": 49, - "type": "WanVideoLoraSelect", - "pos": [ - -530.6770629882812, - -794.5624389648438 - ], - "size": [ - 589.3995971679688, - 150 - ], - "flags": {}, - "order": 27, - "mode": 0, - "inputs": [ - { - "name": "prev_lora", - "shape": 7, - "type": "WANVIDLORA", - "link": null - }, - { - "name": "blocks", - "shape": 7, - "type": "SELECTEDBLOCKS", - "link": null - } - ], - "outputs": [ - { - "name": "lora", - "type": "WANVIDLORA", - "links": [ - 61 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", - "Node name for S&R": "WanVideoLoraSelect" - }, - "widgets_values": [ - "WanVideo\\Lightx2v\\lightx2v_T2V_14B_cfg_step_distill_v2_lora_rank64_bf16_.safetensors", - 1, - false, - false - ], - "color": "#223", - "bgcolor": "#335" - }, { "id": 30, "type": "VHS_VideoCombine", @@ -2931,13 +2607,13 @@ "hidden": false, "paused": false, "params": { - "filename": "Wanimate_00007-audio.mp4", + "filename": "Wanimate_00015-audio.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "Wanimate_00007.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00007-audio.mp4" + "workflow": "Wanimate_00015.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00015-audio.mp4" } } } @@ -2954,7 +2630,7 @@ 362.56817626953125 ], "flags": {}, - "order": 28, + "order": 27, "mode": 0, "inputs": [ { @@ -3066,17 +2742,353 @@ "hidden": false, "paused": false, "params": { - "filename": "WanVideo2_1_T2V_00008.mp4", + "filename": "WanVideo2_1_T2V_00018.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "WanVideo2_1_T2V_00008.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00008.mp4" + "workflow": "WanVideo2_1_T2V_00018.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00018.mp4" } } } }, + { + "id": 169, + "type": "Note", + "pos": [ + 802.1256103515625, + -809.209228515625 + ], + "size": [ + 484.7705383300781, + 194.427978515625 + ], + "flags": {}, + "order": 28, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 49, + "type": "WanVideoLoraSelect", + "pos": [ + -530.6770629882812, + -794.5624389648438 + ], + "size": [ + 589.3995971679688, + 150 + ], + "flags": {}, + "order": 29, + "mode": 0, + "inputs": [ + { + "name": "prev_lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "blocks", + "shape": 7, + "type": "SELECTEDBLOCKS", + "link": null + } + ], + "outputs": [ + { + "name": "lora", + "type": "WANVIDLORA", + "links": [ + 61 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", + "Node name for S&R": "WanVideoLoraSelect" + }, + "widgets_values": [ + "WanVideo\\Lightx2v\\lightx2v_I2V_14B_480p_cfg_step_distill_rank64_bf16.safetensors", + 1, + false, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1626.8492431640625, + -523.2383422851562 + ], + "size": [ + 315, + 874.1923217773438 + ], + "flags": {}, + "order": 50, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 63 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 84 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 281 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 33 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 5, + 42, + "fixed", + true, + "dpm++_sde", + 0, + 1, + "", + "comfy", + 0, + -1, + false + ] + }, + { + "id": 62, + "type": "WanVideoAnimateEmbeds", + "pos": [ + 1086.8922119140625, + -380.2015380859375 + ], + "size": [ + 274.2164001464844, + 370 + ], + "flags": {}, + "order": 43, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 280 + }, + { + "name": "clip_embeds", + "shape": 7, + "type": "WANVIDIMAGE_CLIPEMBEDS", + "link": 96 + }, + { + "name": "ref_images", + "shape": 7, + "type": "IMAGE", + "link": 282 + }, + { + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 244 + }, + { + "name": "face_images", + "shape": 7, + "type": "IMAGE", + "link": 240 + }, + { + "name": "bg_images", + "shape": 7, + "type": "IMAGE", + "link": 234 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": 247 + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 265 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 266 + }, + { + "name": "num_frames", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "link": 268 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 84 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0", + "Node name for S&R": "WanVideoAnimateEmbeds" + }, + "widgets_values": [ + 832, + 480, + 501, + false, + 77, + "disabled", + 1, + 1, + false + ], + "color": "#322", + "bgcolor": "#533" + }, { "id": 22, "type": "WanVideoModelLoader", @@ -3157,7 +3169,7 @@ "Node name for S&R": "WanVideoModelLoader" }, "widgets_values": [ - "WanVideo\\2_2\\Wan2_2_Animate\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors", + "WanVideo\\2_2\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors", "fp16_fast", "disabled", "offload_device", @@ -3165,29 +3177,6 @@ ], "color": "#223", "bgcolor": "#335" - }, - { - "id": 169, - "type": "Note", - "pos": [ - 802.1256103515625, - -809.209228515625 - ], - "size": [ - 484.7705383300781, - 194.427978515625 - ], - "flags": {}, - "order": 29, - "mode": 0, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips." - ], - "color": "#432", - "bgcolor": "#653" } ], "links": [ @@ -3702,22 +3691,6 @@ 62, 2, "IMAGE" - ], - [ - 283, - 110, - 0, - 27, - 5, - "WANVIDCONTEXT" - ], - [ - 284, - 158, - 0, - 62, - 10, - "INT" ] ], "groups": [ @@ -3790,10 +3763,10 @@ "config": {}, "extra": { "ds": { - "scale": 0.43056764313424933, + "scale": 0.5730855330116861, "offset": [ - 2217.6235348644855, - 2515.973484920164 + 194.58444325255581, + 1398.253756932944 ] }, "frontendVersion": "1.27.4", diff --git a/nodes_model_loading.py b/nodes_model_loading.py index f8315e3..81d1827 100644 --- a/nodes_model_loading.py +++ b/nodes_model_loading.py @@ -761,7 +761,7 @@ class WanVideoSetLoRAs: def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, transformer_load_device=None, block_swap_args=None, gguf=False, reader=None, patcher=None): params_to_keep = {"time_in", "patch_embedding", "time_", "modulation", "text_embedding", - "adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "motion_encoder"} + "adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "face_encoder"} param_count = sum(1 for _ in transformer.named_parameters()) pbar = ProgressBar(param_count) cnt = 0 @@ -845,13 +845,13 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, continue if gguf: - dtype_to_use = torch.float32 if "patch_embedding" in name else base_dtype + dtype_to_use = torch.float32 if "patch_embedding" in name or "motion_encoder" in name else base_dtype else: dtype_to_use = base_dtype if any(keyword in name for keyword in params_to_keep) else weight_dtype dtype_to_use = weight_dtype if sd[name.replace("_orig_mod.", "")].dtype == weight_dtype else dtype_to_use if "modulation" in name or "norm" in name or "bias" in name or "img_emb" in name: dtype_to_use = base_dtype - if "patch_embedding" in name or "motion_encoder" in name or "face_encoder" in name: + if "patch_embedding" in name or "motion_encoder" in name: dtype_to_use = torch.float32 load_device = transformer_load_device diff --git a/wanvideo/modules/model.py b/wanvideo/modules/model.py index d5befae..790cbc8 100644 --- a/wanvideo/modules/model.py +++ b/wanvideo/modules/model.py @@ -1736,6 +1736,7 @@ class WanModel(torch.nn.Module): in_dim=motion_encoder_dim, out_dim=self.dim, num_heads=4, + dtype=dtype ) def block_swap(self, blocks_to_swap, offload_txt_emb=False, offload_img_emb=False, vace_blocks_to_swap=None, prefetch_blocks=0, block_swap_debug=False): @@ -1894,7 +1895,7 @@ class WanModel(torch.nn.Module): motion_vec = rearrange(torch.cat(face_pixel_values_tmp), "(b t) c -> b t c", t=T) del face_pixel_values_tmp self.face_encoder.to(self.main_device) - motion_vec = self.face_encoder(motion_vec) + motion_vec = self.face_encoder(motion_vec.to(self.face_encoder.dtype)) self.face_encoder.to(self.offload_device) B, L, H, C = motion_vec.shape diff --git a/wanvideo/modules/wananimate/face_blocks.py b/wanvideo/modules/wananimate/face_blocks.py index 0627fa8..ead9856 100644 --- a/wanvideo/modules/wananimate/face_blocks.py +++ b/wanvideo/modules/wananimate/face_blocks.py @@ -22,6 +22,8 @@ class CausalConv1d(nn.Module): class FaceEncoder(nn.Module): def __init__(self, in_dim: int, out_dim: int, num_heads: int, dtype=None, device=None): super().__init__() + self.dtype = dtype + self.device = device self.num_heads = num_heads self.conv1_local = CausalConv1d(in_dim, 1024 * num_heads, 3, stride=1)