Fix GGUF with WanAnimate

This commit is contained in:
kijai
2025-09-19 11:35:14 +03:00
parent 8bc1d6651a
commit b06562823b
4 changed files with 360 additions and 384 deletions
@@ -730,8 +730,7 @@
"name": "INT",
"type": "INT",
"links": [
268,
284
268
]
}
],
@@ -1824,13 +1823,13 @@
"hidden": false,
"paused": false,
"params": {
"filename": "WanVideo2_1_T2V_00011.mp4",
"filename": "WanVideo2_1_T2V_00020.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4",
"frame_rate": 16,
"workflow": "WanVideo2_1_T2V_00011.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00011.mp4"
"workflow": "WanVideo2_1_T2V_00020.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00020.mp4"
}
}
}
@@ -2407,7 +2406,7 @@
"links": null
},
{
"label": "109 count",
"label": "154 count",
"name": "count",
"type": "INT",
"links": null
@@ -2498,159 +2497,6 @@
null
]
},
{
"id": 27,
"type": "WanVideoSampler",
"pos": [
1626.8492431640625,
-523.2383422851562
],
"size": [
315,
874.1923217773438
],
"flags": {},
"order": 50,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "WANVIDEOMODEL",
"link": 63
},
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"link": 84
},
{
"name": "text_embeds",
"shape": 7,
"type": "WANVIDEOTEXTEMBEDS",
"link": 281
},
{
"name": "samples",
"shape": 7,
"type": "LATENT",
"link": null
},
{
"name": "feta_args",
"shape": 7,
"type": "FETAARGS",
"link": null
},
{
"name": "context_options",
"shape": 7,
"type": "WANVIDCONTEXT",
"link": 283
},
{
"name": "cache_args",
"shape": 7,
"type": "CACHEARGS",
"link": null
},
{
"name": "flowedit_args",
"shape": 7,
"type": "FLOWEDITARGS",
"link": null
},
{
"name": "slg_args",
"shape": 7,
"type": "SLGARGS",
"link": null
},
{
"name": "loop_args",
"shape": 7,
"type": "LOOPARGS",
"link": null
},
{
"name": "experimental_args",
"shape": 7,
"type": "EXPERIMENTALARGS",
"link": null
},
{
"name": "sigmas",
"shape": 7,
"type": "SIGMAS",
"link": null
},
{
"name": "unianimate_poses",
"shape": 7,
"type": "UNIANIMATE_POSE",
"link": null
},
{
"name": "fantasytalking_embeds",
"shape": 7,
"type": "FANTASYTALKING_EMBEDS",
"link": null
},
{
"name": "uni3c_embeds",
"shape": 7,
"type": "UNI3C_EMBEDS",
"link": null
},
{
"name": "multitalk_embeds",
"shape": 7,
"type": "MULTITALK_EMBEDS",
"link": null
},
{
"name": "freeinit_args",
"shape": 7,
"type": "FREEINITARGS",
"link": null
}
],
"outputs": [
{
"name": "samples",
"type": "LATENT",
"slot_index": 0,
"links": [
33
]
},
{
"name": "denoised_samples",
"type": "LATENT",
"links": null
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "e5ef9752a7e846b232fc05fd993327a2e870a788",
"Node name for S&R": "WanVideoSampler"
},
"widgets_values": [
6,
1,
5,
42,
"fixed",
true,
"dpm++_sde",
0,
1,
"",
"comfy",
0,
-1,
false
]
},
{
"id": 110,
"type": "WanVideoContextOptions",
@@ -2677,9 +2523,7 @@
{
"name": "context_options",
"type": "WANVIDCONTEXT",
"links": [
283
]
"links": []
}
],
"properties": {
@@ -2697,174 +2541,6 @@
"linear"
]
},
{
"id": 62,
"type": "WanVideoAnimateEmbeds",
"pos": [
1086.8922119140625,
-380.2015380859375
],
"size": [
274.2164001464844,
370
],
"flags": {},
"order": 43,
"mode": 0,
"inputs": [
{
"name": "vae",
"type": "WANVAE",
"link": 280
},
{
"name": "clip_embeds",
"shape": 7,
"type": "WANVIDIMAGE_CLIPEMBEDS",
"link": 96
},
{
"name": "ref_images",
"shape": 7,
"type": "IMAGE",
"link": 282
},
{
"name": "pose_images",
"shape": 7,
"type": "IMAGE",
"link": 244
},
{
"name": "face_images",
"shape": 7,
"type": "IMAGE",
"link": 240
},
{
"name": "bg_images",
"shape": 7,
"type": "IMAGE",
"link": 234
},
{
"name": "mask",
"shape": 7,
"type": "MASK",
"link": 247
},
{
"name": "width",
"type": "INT",
"widget": {
"name": "width"
},
"link": 265
},
{
"name": "height",
"type": "INT",
"widget": {
"name": "height"
},
"link": 266
},
{
"name": "num_frames",
"type": "INT",
"widget": {
"name": "num_frames"
},
"link": 268
},
{
"name": "frame_window_size",
"type": "INT",
"widget": {
"name": "frame_window_size"
},
"link": 284
}
],
"outputs": [
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"links": [
84
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0",
"Node name for S&R": "WanVideoAnimateEmbeds"
},
"widgets_values": [
832,
480,
501,
false,
77,
"disabled",
1,
1,
false
],
"color": "#322",
"bgcolor": "#533"
},
{
"id": 49,
"type": "WanVideoLoraSelect",
"pos": [
-530.6770629882812,
-794.5624389648438
],
"size": [
589.3995971679688,
150
],
"flags": {},
"order": 27,
"mode": 0,
"inputs": [
{
"name": "prev_lora",
"shape": 7,
"type": "WANVIDLORA",
"link": null
},
{
"name": "blocks",
"shape": 7,
"type": "SELECTEDBLOCKS",
"link": null
}
],
"outputs": [
{
"name": "lora",
"type": "WANVIDLORA",
"links": [
61
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "e5ef9752a7e846b232fc05fd993327a2e870a788",
"Node name for S&R": "WanVideoLoraSelect"
},
"widgets_values": [
"WanVideo\\Lightx2v\\lightx2v_T2V_14B_cfg_step_distill_v2_lora_rank64_bf16_.safetensors",
1,
false,
false
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 30,
"type": "VHS_VideoCombine",
@@ -2931,13 +2607,13 @@
"hidden": false,
"paused": false,
"params": {
"filename": "Wanimate_00007-audio.mp4",
"filename": "Wanimate_00015-audio.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4",
"frame_rate": 16,
"workflow": "Wanimate_00007.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00007-audio.mp4"
"workflow": "Wanimate_00015.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\Wanimate_00015-audio.mp4"
}
}
}
@@ -2954,7 +2630,7 @@
362.56817626953125
],
"flags": {},
"order": 28,
"order": 27,
"mode": 0,
"inputs": [
{
@@ -3066,17 +2742,353 @@
"hidden": false,
"paused": false,
"params": {
"filename": "WanVideo2_1_T2V_00008.mp4",
"filename": "WanVideo2_1_T2V_00018.mp4",
"subfolder": "",
"type": "temp",
"format": "video/h264-mp4",
"frame_rate": 16,
"workflow": "WanVideo2_1_T2V_00008.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00008.mp4"
"workflow": "WanVideo2_1_T2V_00018.png",
"fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideo2_1_T2V_00018.mp4"
}
}
}
},
{
"id": 169,
"type": "Note",
"pos": [
802.1256103515625,
-809.209228515625
],
"size": [
484.7705383300781,
194.427978515625
],
"flags": {},
"order": 28,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips."
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 49,
"type": "WanVideoLoraSelect",
"pos": [
-530.6770629882812,
-794.5624389648438
],
"size": [
589.3995971679688,
150
],
"flags": {},
"order": 29,
"mode": 0,
"inputs": [
{
"name": "prev_lora",
"shape": 7,
"type": "WANVIDLORA",
"link": null
},
{
"name": "blocks",
"shape": 7,
"type": "SELECTEDBLOCKS",
"link": null
}
],
"outputs": [
{
"name": "lora",
"type": "WANVIDLORA",
"links": [
61
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "e5ef9752a7e846b232fc05fd993327a2e870a788",
"Node name for S&R": "WanVideoLoraSelect"
},
"widgets_values": [
"WanVideo\\Lightx2v\\lightx2v_I2V_14B_480p_cfg_step_distill_rank64_bf16.safetensors",
1,
false,
false
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 27,
"type": "WanVideoSampler",
"pos": [
1626.8492431640625,
-523.2383422851562
],
"size": [
315,
874.1923217773438
],
"flags": {},
"order": 50,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "WANVIDEOMODEL",
"link": 63
},
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"link": 84
},
{
"name": "text_embeds",
"shape": 7,
"type": "WANVIDEOTEXTEMBEDS",
"link": 281
},
{
"name": "samples",
"shape": 7,
"type": "LATENT",
"link": null
},
{
"name": "feta_args",
"shape": 7,
"type": "FETAARGS",
"link": null
},
{
"name": "context_options",
"shape": 7,
"type": "WANVIDCONTEXT",
"link": null
},
{
"name": "cache_args",
"shape": 7,
"type": "CACHEARGS",
"link": null
},
{
"name": "flowedit_args",
"shape": 7,
"type": "FLOWEDITARGS",
"link": null
},
{
"name": "slg_args",
"shape": 7,
"type": "SLGARGS",
"link": null
},
{
"name": "loop_args",
"shape": 7,
"type": "LOOPARGS",
"link": null
},
{
"name": "experimental_args",
"shape": 7,
"type": "EXPERIMENTALARGS",
"link": null
},
{
"name": "sigmas",
"shape": 7,
"type": "SIGMAS",
"link": null
},
{
"name": "unianimate_poses",
"shape": 7,
"type": "UNIANIMATE_POSE",
"link": null
},
{
"name": "fantasytalking_embeds",
"shape": 7,
"type": "FANTASYTALKING_EMBEDS",
"link": null
},
{
"name": "uni3c_embeds",
"shape": 7,
"type": "UNI3C_EMBEDS",
"link": null
},
{
"name": "multitalk_embeds",
"shape": 7,
"type": "MULTITALK_EMBEDS",
"link": null
},
{
"name": "freeinit_args",
"shape": 7,
"type": "FREEINITARGS",
"link": null
}
],
"outputs": [
{
"name": "samples",
"type": "LATENT",
"slot_index": 0,
"links": [
33
]
},
{
"name": "denoised_samples",
"type": "LATENT",
"links": null
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "e5ef9752a7e846b232fc05fd993327a2e870a788",
"Node name for S&R": "WanVideoSampler"
},
"widgets_values": [
6,
1,
5,
42,
"fixed",
true,
"dpm++_sde",
0,
1,
"",
"comfy",
0,
-1,
false
]
},
{
"id": 62,
"type": "WanVideoAnimateEmbeds",
"pos": [
1086.8922119140625,
-380.2015380859375
],
"size": [
274.2164001464844,
370
],
"flags": {},
"order": 43,
"mode": 0,
"inputs": [
{
"name": "vae",
"type": "WANVAE",
"link": 280
},
{
"name": "clip_embeds",
"shape": 7,
"type": "WANVIDIMAGE_CLIPEMBEDS",
"link": 96
},
{
"name": "ref_images",
"shape": 7,
"type": "IMAGE",
"link": 282
},
{
"name": "pose_images",
"shape": 7,
"type": "IMAGE",
"link": 244
},
{
"name": "face_images",
"shape": 7,
"type": "IMAGE",
"link": 240
},
{
"name": "bg_images",
"shape": 7,
"type": "IMAGE",
"link": 234
},
{
"name": "mask",
"shape": 7,
"type": "MASK",
"link": 247
},
{
"name": "width",
"type": "INT",
"widget": {
"name": "width"
},
"link": 265
},
{
"name": "height",
"type": "INT",
"widget": {
"name": "height"
},
"link": 266
},
{
"name": "num_frames",
"type": "INT",
"widget": {
"name": "num_frames"
},
"link": 268
}
],
"outputs": [
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"links": [
84
]
}
],
"properties": {
"cnr_id": "ComfyUI-WanVideoWrapper",
"ver": "761b1d191e50d589465e31dc0d40ff7c59b1b7b0",
"Node name for S&R": "WanVideoAnimateEmbeds"
},
"widgets_values": [
832,
480,
501,
false,
77,
"disabled",
1,
1,
false
],
"color": "#322",
"bgcolor": "#533"
},
{
"id": 22,
"type": "WanVideoModelLoader",
@@ -3157,7 +3169,7 @@
"Node name for S&R": "WanVideoModelLoader"
},
"widgets_values": [
"WanVideo\\2_2\\Wan2_2_Animate\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors",
"WanVideo\\2_2\\Wan2_2-Animate-14B_fp8_e4m3fn_scaled_KJ.safetensors",
"fp16_fast",
"disabled",
"offload_device",
@@ -3165,29 +3177,6 @@
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 169,
"type": "Note",
"pos": [
802.1256103515625,
-809.209228515625
],
"size": [
484.7705383300781,
194.427978515625
],
"flags": {},
"order": 29,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"You can use either context options, or the original long gen method.\n\nWhen using context options, set num_frames and frame_window_size to be equal'\n\nOtherwise set frame_window_size to something the model is capbable of doing normally, original default is 77, 81 seems to work too.\n\nOriginal method is probably better for motion etc, and is faster, but the benefit of context windows is that it doesn't degrade over time on long clips."
],
"color": "#432",
"bgcolor": "#653"
}
],
"links": [
@@ -3702,22 +3691,6 @@
62,
2,
"IMAGE"
],
[
283,
110,
0,
27,
5,
"WANVIDCONTEXT"
],
[
284,
158,
0,
62,
10,
"INT"
]
],
"groups": [
@@ -3790,10 +3763,10 @@
"config": {},
"extra": {
"ds": {
"scale": 0.43056764313424933,
"scale": 0.5730855330116861,
"offset": [
2217.6235348644855,
2515.973484920164
194.58444325255581,
1398.253756932944
]
},
"frontendVersion": "1.27.4",
+3 -3
View File
@@ -761,7 +761,7 @@ class WanVideoSetLoRAs:
def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None,
transformer_load_device=None, block_swap_args=None, gguf=False, reader=None, patcher=None):
params_to_keep = {"time_in", "patch_embedding", "time_", "modulation", "text_embedding",
"adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "motion_encoder"}
"adapter", "add", "ref_conv", "casual_audio_encoder", "cond_encoder", "frame_packer", "audio_proj_glob", "face_encoder"}
param_count = sum(1 for _ in transformer.named_parameters())
pbar = ProgressBar(param_count)
cnt = 0
@@ -845,13 +845,13 @@ def load_weights(transformer, sd=None, weight_dtype=None, base_dtype=None,
continue
if gguf:
dtype_to_use = torch.float32 if "patch_embedding" in name else base_dtype
dtype_to_use = torch.float32 if "patch_embedding" in name or "motion_encoder" in name else base_dtype
else:
dtype_to_use = base_dtype if any(keyword in name for keyword in params_to_keep) else weight_dtype
dtype_to_use = weight_dtype if sd[name.replace("_orig_mod.", "")].dtype == weight_dtype else dtype_to_use
if "modulation" in name or "norm" in name or "bias" in name or "img_emb" in name:
dtype_to_use = base_dtype
if "patch_embedding" in name or "motion_encoder" in name or "face_encoder" in name:
if "patch_embedding" in name or "motion_encoder" in name:
dtype_to_use = torch.float32
load_device = transformer_load_device
+2 -1
View File
@@ -1736,6 +1736,7 @@ class WanModel(torch.nn.Module):
in_dim=motion_encoder_dim,
out_dim=self.dim,
num_heads=4,
dtype=dtype
)
def block_swap(self, blocks_to_swap, offload_txt_emb=False, offload_img_emb=False, vace_blocks_to_swap=None, prefetch_blocks=0, block_swap_debug=False):
@@ -1894,7 +1895,7 @@ class WanModel(torch.nn.Module):
motion_vec = rearrange(torch.cat(face_pixel_values_tmp), "(b t) c -> b t c", t=T)
del face_pixel_values_tmp
self.face_encoder.to(self.main_device)
motion_vec = self.face_encoder(motion_vec)
motion_vec = self.face_encoder(motion_vec.to(self.face_encoder.dtype))
self.face_encoder.to(self.offload_device)
B, L, H, C = motion_vec.shape
@@ -22,6 +22,8 @@ class CausalConv1d(nn.Module):
class FaceEncoder(nn.Module):
def __init__(self, in_dim: int, out_dim: int, num_heads: int, dtype=None, device=None):
super().__init__()
self.dtype = dtype
self.device = device
self.num_heads = num_heads
self.conv1_local = CausalConv1d(in_dim, 1024 * num_heads, 3, stride=1)