From b06562823bf70f6504fbca2c8d9e85b0446fd893 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Fri, 19 Sep 2025 11:35:14 +0300 Subject: [PATCH] Fix GGUF with WanAnimate --- .../wanvideo_WanAnimate_example_01.json | 733 +++++++++--------- nodes_model_loading.py | 6 +- wanvideo/modules/model.py | 3 +- wanvideo/modules/wananimate/face_blocks.py | 2 + 4 files changed, 360 insertions(+), 384 deletions(-) diff --git a/example_workflows/wanvideo_WanAnimate_example_01.json b/example_workflows/wanvideo_WanAnimate_example_01.json index 9e69f54..dcc67c2 100644 --- a/example_workflows/wanvideo_WanAnimate_example_01.json +++ b/example_workflows/wanvideo_WanAnimate_example_01.json @@ -730,8 +730,7 @@ "name": "INT", "type": "INT", "links": [ - 268, - 284 + 268 ] } ], @@ -1824,13 +1823,13 @@ "hidden": false, "paused": false, "params": { - "filename": "WanVideo2_1_T2V_00011.mp4", + "filename": "WanVideo2_1_T2V_00020.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "WanVideo2_1_T2V_00011.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00011.mp4" + "workflow": "WanVideo2_1_T2V_00020.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00020.mp4" } } } @@ -2407,7 +2406,7 @@ "links": null }, { - "label": "109 count", + "label": "154 count", "name": "count", "type": "INT", "links": null @@ -2498,159 +2497,6 @@ null ] }, - { - "id": 27, - "type": "WanVideoSampler", - "pos": [ - 1626.8492431640625, - -523.2383422851562 - ], - "size": [ - 315, - 874.1923217773438 - ], - "flags": {}, - "order": 50, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "WANVIDEOMODEL", - "link": 63 - }, - { - "name": "image_embeds", - "type": "WANVIDIMAGE_EMBEDS", - "link": 84 - }, - { - "name": "text_embeds", - "shape": 7, - "type": "WANVIDEOTEXTEMBEDS", - "link": 281 - }, - { - "name": "samples", - "shape": 7, - "type": "LATENT", - "link": null - }, - { - "name": "feta_args", - "shape": 7, - "type": "FETAARGS", - "link": null - }, - { - "name": "context_options", - "shape": 7, - "type": "WANVIDCONTEXT", - "link": 283 - }, - { - "name": "cache_args", - "shape": 7, - "type": "CACHEARGS", - "link": null - }, - { - "name": "flowedit_args", - "shape": 7, - "type": "FLOWEDITARGS", - "link": null - }, - { - "name": "slg_args", - "shape": 7, - "type": "SLGARGS", - "link": null - }, - { - "name": "loop_args", - "shape": 7, - "type": "LOOPARGS", - "link": null - }, - { - "name": "experimental_args", - "shape": 7, - "type": "EXPERIMENTALARGS", - "link": null - }, - { - "name": "sigmas", - "shape": 7, - "type": "SIGMAS", - "link": null - }, - { - "name": "unianimate_poses", - "shape": 7, - "type": "UNIANIMATE_POSE", - "link": null - }, - { - "name": "fantasytalking_embeds", - "shape": 7, - "type": "FANTASYTALKING_EMBEDS", - "link": null - }, - { - "name": "uni3c_embeds", - "shape": 7, - "type": "UNI3C_EMBEDS", - "link": null - }, - { - "name": "multitalk_embeds", - "shape": 7, - "type": "MULTITALK_EMBEDS", - "link": null - }, - { - "name": "freeinit_args", - "shape": 7, - "type": "FREEINITARGS", - "link": null - } - ], - "outputs": [ - { - "name": "samples", - "type": "LATENT", - "slot_index": 0, - "links": [ - 33 - ] - }, - { - "name": "denoised_samples", - "type": "LATENT", - "links": null - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", - "Node name for S&R": "WanVideoSampler" - }, - "widgets_values": [ - 6, - 1, - 5, - 42, - "fixed", - true, - "dpm++_sde", - 0, - 1, - "", - "comfy", - 0, - -1, - false - ] - }, { "id": 110, "type": "WanVideoContextOptions", @@ -2677,9 +2523,7 @@ { "name": "context_options", "type": "WANVIDCONTEXT", - "links": [ - 283 - ] + "links": [] } ], "properties": { @@ -2697,174 +2541,6 @@ "linear" ] }, - { - "id": 62, - "type": "WanVideoAnimateEmbeds", - "pos": [ - 1086.8922119140625, - -380.2015380859375 - ], - "size": [ - 274.2164001464844, - 370 - ], - "flags": {}, - "order": 43, - "mode": 0, - "inputs": [ - { - "name": "vae", - "type": "WANVAE", - "link": 280 - }, - { - "name": "clip_embeds", - "shape": 7, - "type": "WANVIDIMAGE_CLIPEMBEDS", - "link": 96 - }, - { - "name": "ref_images", - "shape": 7, - "type": "IMAGE", - "link": 282 - }, - { - "name": "pose_images", - "shape": 7, - "type": "IMAGE", - "link": 244 - }, - { - "name": "face_images", - "shape": 7, - "type": "IMAGE", - "link": 240 - }, - { - "name": "bg_images", - "shape": 7, - "type": "IMAGE", - "link": 234 - }, - { - "name": "mask", - "shape": 7, - "type": "MASK", - "link": 247 - }, - { - "name": "width", - "type": "INT", - "widget": { - "name": "width" - }, - "link": 265 - }, - { - "name": "height", - "type": "INT", - "widget": { - "name": "height" - }, - "link": 266 - }, - { - "name": "num_frames", - "type": "INT", - "widget": { - "name": "num_frames" - }, - "link": 268 - }, - { - "name": "frame_window_size", - "type": "INT", - "widget": { - "name": "frame_window_size" - }, - "link": 284 - } - ], - "outputs": [ - { - "name": "image_embeds", - "type": "WANVIDIMAGE_EMBEDS", - "links": [ - 84 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0", - "Node name for S&R": "WanVideoAnimateEmbeds" - }, - "widgets_values": [ - 832, - 480, - 501, - false, - 77, - "disabled", - 1, - 1, - false - ], - "color": "#322", - "bgcolor": "#533" - }, - { - "id": 49, - "type": "WanVideoLoraSelect", - "pos": [ - -530.6770629882812, - -794.5624389648438 - ], - "size": [ - 589.3995971679688, - 150 - ], - "flags": {}, - "order": 27, - "mode": 0, - "inputs": [ - { - "name": "prev_lora", - "shape": 7, - "type": "WANVIDLORA", - "link": null - }, - { - "name": "blocks", - "shape": 7, - "type": "SELECTEDBLOCKS", - "link": null - } - ], - "outputs": [ - { - "name": "lora", - "type": "WANVIDLORA", - "links": [ - 61 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", - "Node name for S&R": "WanVideoLoraSelect" - }, - "widgets_values": [ - "WanVideo\\Lightx2v\\lightx2v_T2V_14B_cfg_step_distill_v2_lora_rank64_bf16_.safetensors", - 1, - false, - false - ], - "color": "#223", - "bgcolor": "#335" - }, { "id": 30, "type": "VHS_VideoCombine", @@ -2931,13 +2607,13 @@ "hidden": false, "paused": false, "params": { - "filename": "Wanimate_00007-audio.mp4", + "filename": "Wanimate_00015-audio.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "Wanimate_00007.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00007-audio.mp4" + "workflow": "Wanimate_00015.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00015-audio.mp4" } } } @@ -2954,7 +2630,7 @@ 362.56817626953125 ], "flags": {}, - "order": 28, + "order": 27, "mode": 0, "inputs": [ { @@ -3066,17 +2742,353 @@ "hidden": false, "paused": false, "params": { - "filename": "WanVideo2_1_T2V_00008.mp4", + "filename": "WanVideo2_1_T2V_00018.mp4", "subfolder": "", "type": "temp", "format": "video/h264-mp4", "frame_rate": 16, - "workflow": "WanVideo2_1_T2V_00008.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00008.mp4" + "workflow": "WanVideo2_1_T2V_00018.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00018.mp4" } } } }, + { + "id": 169, + "type": "Note", + "pos": [ + 802.1256103515625, + -809.209228515625 + ], + "size": [ + 484.7705383300781, + 194.427978515625 + ], + "flags": {}, + "order": 28, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 49, + "type": "WanVideoLoraSelect", + "pos": [ + -530.6770629882812, + -794.5624389648438 + ], + "size": [ + 589.3995971679688, + 150 + ], + "flags": {}, + "order": 29, + "mode": 0, + "inputs": [ + { + "name": "prev_lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "blocks", + "shape": 7, + "type": "SELECTEDBLOCKS", + "link": null + } + ], + "outputs": [ + { + "name": "lora", + "type": "WANVIDLORA", + "links": [ + 61 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", + "Node name for S&R": "WanVideoLoraSelect" + }, + "widgets_values": [ + "WanVideo\\Lightx2v\\lightx2v_I2V_14B_480p_cfg_step_distill_rank64_bf16.safetensors", + 1, + false, + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1626.8492431640625, + -523.2383422851562 + ], + "size": [ + 315, + 874.1923217773438 + ], + "flags": {}, + "order": 50, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 63 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 84 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 281 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": null + }, + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 33 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "e5ef9752a7e846b232fc05fd993327a2e870a788", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 6, + 1, + 5, + 42, + "fixed", + true, + "dpm++_sde", + 0, + 1, + "", + "comfy", + 0, + -1, + false + ] + }, + { + "id": 62, + "type": "WanVideoAnimateEmbeds", + "pos": [ + 1086.8922119140625, + -380.2015380859375 + ], + "size": [ + 274.2164001464844, + 370 + ], + "flags": {}, + "order": 43, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 280 + }, + { + "name": "clip_embeds", + "shape": 7, + "type": "WANVIDIMAGE_CLIPEMBEDS", + "link": 96 + }, + { + "name": "ref_images", + "shape": 7, + "type": "IMAGE", + "link": 282 + }, + { + "name": "pose_images", + "shape": 7, + "type": "IMAGE", + "link": 244 + }, + { + "name": "face_images", + "shape": 7, + "type": "IMAGE", + "link": 240 + }, + { + "name": "bg_images", + "shape": 7, + "type": "IMAGE", + "link": 234 + }, + { + "name": "mask", + "shape": 7, + "type": "MASK", + "link": 247 + }, + { + "name": "width", + "type": "INT", + "widget": { + "name": "width" + }, + "link": 265 + }, + { + "name": "height", + "type": "INT", + "widget": { + "name": "height" + }, + "link": 266 + }, + { + "name": "num_frames", + "type": "INT", + "widget": { + "name": "num_frames" + }, + "link": 268 + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 84 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0", + "Node name for S&R": "WanVideoAnimateEmbeds" + }, + "widgets_values": [ + 832, + 480, + 501, + false, + 77, + "disabled", + 1, + 1, + false + ], + "color": "#322", + "bgcolor": "#533" + }, { "id": 22, "type": "WanVideoModelLoader", @@ -3157,7 +3169,7 @@ "Node name for S&R": "WanVideoModelLoader" }, "widgets_values": [ - "WanVideo\\2_2\\Wan2_2_Animate\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors", + "WanVideo\\2_2\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors", "fp16_fast", "disabled", "offload_device", @@ -3165,29 +3177,6 @@ ], "color": "#223", "bgcolor": "#335" - }, - { - "id": 169, - "type": "Note", - "pos": [ - 802.1256103515625, - -809.209228515625 - ], - "size": [ - 484.7705383300781, - 194.427978515625 - ], - "flags": {}, - "order": 29, - "mode": 0, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips." - ], - "color": "#432", - "bgcolor": "#653" } ], "links": [ @@ -3702,22 +3691,6 @@ 62, 2, "IMAGE" - ], - [ - 283, - 110, - 0, - 27, - 5, - "WANVIDCONTEXT" - ], - [ - 284, - 158, - 0, - 62, - 10, - "INT" ] ], "groups": [ @@ -3790,10 +3763,10 @@ "config": {}, "extra": { "ds": { - "scale": 0.43056764313424933, + "scale": 0.5730855330116861, "offset": [ - 2217.6235348644855, - 2515.973484920164 + 194.58444325255581, + 1398.253756932944 ] }, "frontendVersion": "1.27.4", diff --git a/nodes_model_loading.py b/nodes_model_loading.py index f8315e3..81d1827 100644 --- a/nodes_model_loading.py +++ b/nodes_model_loading.py @@ -761,7 +761,7 @@ class WanVideoSetLoRAs: def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, transformer_load_device=None, block_swap_args=None, gguf=False, reader=None, patcher=None): params_to_keep = {"time_in", "patch_embedding", "time_", "modulation", "text_embedding", - "adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "motion_encoder"} + "adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "face_encoder"} param_count = sum(1 for _ in transformer.named_parameters()) pbar = ProgressBar(param_count) cnt = 0 @@ -845,13 +845,13 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None, continue if gguf: - dtype_to_use = torch.float32 if "patch_embedding" in name else base_dtype + dtype_to_use = torch.float32 if "patch_embedding" in name or "motion_encoder" in name else base_dtype else: dtype_to_use = base_dtype if any(keyword in name for keyword in params_to_keep) else weight_dtype dtype_to_use = weight_dtype if sd[name.replace("_orig_mod.", "")].dtype == weight_dtype else dtype_to_use if "modulation" in name or "norm" in name or "bias" in name or "img_emb" in name: dtype_to_use = base_dtype - if "patch_embedding" in name or "motion_encoder" in name or "face_encoder" in name: + if "patch_embedding" in name or "motion_encoder" in name: dtype_to_use = torch.float32 load_device = transformer_load_device diff --git a/wanvideo/modules/model.py b/wanvideo/modules/model.py index d5befae..790cbc8 100644 --- a/wanvideo/modules/model.py +++ b/wanvideo/modules/model.py @@ -1736,6 +1736,7 @@ class WanModel(torch.nn.Module): in_dim=motion_encoder_dim, out_dim=self.dim, num_heads=4, + dtype=dtype ) def block_swap(self, blocks_to_swap, offload_txt_emb=False, offload_img_emb=False, vace_blocks_to_swap=None, prefetch_blocks=0, block_swap_debug=False): @@ -1894,7 +1895,7 @@ class WanModel(torch.nn.Module): motion_vec = rearrange(torch.cat(face_pixel_values_tmp), "(b t) c -> b t c", t=T) del face_pixel_values_tmp self.face_encoder.to(self.main_device) - motion_vec = self.face_encoder(motion_vec) + motion_vec = self.face_encoder(motion_vec.to(self.face_encoder.dtype)) self.face_encoder.to(self.offload_device) B, L, H, C = motion_vec.shape diff --git a/wanvideo/modules/wananimate/face_blocks.py b/wanvideo/modules/wananimate/face_blocks.py index 0627fa8..ead9856 100644 --- a/wanvideo/modules/wananimate/face_blocks.py +++ b/wanvideo/modules/wananimate/face_blocks.py @@ -22,6 +22,8 @@ class CausalConv1d(nn.Module): class FaceEncoder(nn.Module): def __init__(self, in_dim: int, out_dim: int, num_heads: int, dtype=None, device=None): super().__init__() + self.dtype = dtype + self.device = device self.num_heads = num_heads self.conv1_local = CausalConv1d(in_dim, 1024 * num_heads, 3, stride=1)