Much improved context window blending

Co-Authored-By: Benjamin Paine <57536852+painebenjamin@users.noreply.github.com>
This commit is contained in:
kijai
2025-03-01 02:52:02 +02:00
co-authored by Benjamin Paine
parent 889efe7371
commit e5a6cc872a
2 changed files with 690 additions and 2 deletions
@@ -0,0 +1,672 @@
{
"last_node_id": 44,
"last_link_id": 57,
"nodes": [
{
"id": 33,
"type": "Note",
"pos": [
227.3764190673828,
-205.28524780273438
],
"size": [
351.70458984375,
60
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"Models:\nhttps://huggingface.co/Kijai/WanVideo_comfy/tree/main"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 11,
"type": "LoadWanVideoT5TextEncoder",
"pos": [
224.15325927734375,
-34.481563568115234
],
"size": [
377.1661376953125,
130
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "wan_t5_model",
"type": "WANTEXTENCODER",
"links": [
15
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadWanVideoT5TextEncoder"
},
"widgets_values": [
"umt5-xxl-enc-bf16.safetensors",
"bf16",
"offload_device",
"disabled"
]
},
{
"id": 28,
"type": "WanVideoDecode",
"pos": [
1692.973876953125,
-404.8614501953125
],
"size": [
315,
174
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "vae",
"type": "WANVAE",
"link": 43
},
{
"name": "samples",
"type": "LATENT",
"link": 33
}
],
"outputs": [
{
"name": "images",
"type": "IMAGE",
"links": [
48
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoDecode"
},
"widgets_values": [
true,
272,
272,
144,
128
]
},
{
"id": 22,
"type": "WanVideoModelLoader",
"pos": [
620.3950805664062,
-357.8426818847656
],
"size": [
477.4410095214844,
226.43276977539062
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "compile_args",
"type": "WANCOMPILEARGS",
"shape": 7,
"link": 54
},
{
"name": "block_swap_args",
"type": "BLOCKSWAPARGS",
"shape": 7,
"link": null
},
{
"name": "lora",
"type": "WANVIDLORA",
"shape": 7,
"link": null
}
],
"outputs": [
{
"name": "model",
"type": "WANVIDEOMODEL",
"links": [
29
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoModelLoader"
},
"widgets_values": [
"WanVideo\\Wan2_1-T2V-1_3B_fp32.safetensors",
"fp16",
"disabled",
"offload_device",
"sageattn"
]
},
{
"id": 38,
"type": "WanVideoVAELoader",
"pos": [
1687.4093017578125,
-582.2750854492188
],
"size": [
416.25482177734375,
82
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "vae",
"type": "WANVAE",
"links": [
43
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoVAELoader"
},
"widgets_values": [
"wanvideo\\Wan2_1_VAE_bf16.safetensors",
"bf16"
]
},
{
"id": 42,
"type": "GetImageSizeAndCount",
"pos": [
1708.7301025390625,
-140.99705505371094
],
"size": [
277.20001220703125,
86
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 48
}
],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": [
56
],
"slot_index": 0
},
{
"name": "832 width",
"type": "INT",
"links": null
},
{
"name": "480 height",
"type": "INT",
"links": null
},
{
"name": "129 count",
"type": "INT",
"links": null
}
],
"properties": {
"Node name for S&R": "GetImageSizeAndCount"
},
"widgets_values": []
},
{
"id": 16,
"type": "WanVideoTextEncode",
"pos": [
675.8850708007812,
-36.032100677490234
],
"size": [
420.30511474609375,
261.5306701660156
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "t5",
"type": "WANTEXTENCODER",
"link": 15
}
],
"outputs": [
{
"name": "text_embeds",
"type": "WANVIDEOTEXTEMBEDS",
"links": [
30
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoTextEncode"
},
"widgets_values": [
"high quality nature video featuring a red panda balancing on a bamboo stem while a bird lands on it's head, on the background there is a waterfall",
"色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走",
true
]
},
{
"id": 27,
"type": "WanVideoSampler",
"pos": [
1315.2401123046875,
-401.48028564453125
],
"size": [
315,
534.1923217773438
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "WANVIDEOMODEL",
"link": 29
},
{
"name": "text_embeds",
"type": "WANVIDEOTEXTEMBEDS",
"link": 30
},
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"link": 42
},
{
"name": "samples",
"type": "LATENT",
"shape": 7,
"link": null
},
{
"name": "feta_args",
"type": "FETAARGS",
"shape": 7,
"link": null
},
{
"name": "context_options",
"type": "WANVIDCONTEXT",
"shape": 7,
"link": 57
}
],
"outputs": [
{
"name": "samples",
"type": "LATENT",
"links": [
33
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoSampler"
},
"widgets_values": [
30,
6,
5,
1057359483639288,
"fixed",
true,
"dpm++",
0,
1,
""
]
},
{
"id": 36,
"type": "Note",
"pos": [
796.0189208984375,
-521.5020751953125
],
"size": [
298.2554016113281,
108.62744140625
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [],
"properties": {},
"widgets_values": [
"sdpa should work too, haven't tested flaash\n\nfp8_fast seems to cause huge quality degradation"
],
"color": "#432",
"bgcolor": "#653"
},
{
"id": 30,
"type": "VHS_VideoCombine",
"pos": [
2127.120849609375,
-511.9014587402344
],
"size": [
873.2135620117188,
840.2385864257812
],
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 56
},
{
"name": "audio",
"type": "AUDIO",
"shape": 7,
"link": null
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"shape": 7,
"link": null
},
{
"name": "vae",
"type": "VAE",
"shape": 7,
"link": null
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 16,
"loop_count": 0,
"filename_prefix": "WanVideo2_1_T2V",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "WanVideo2_1_T2V_00125.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 16,
"workflow": "WanVideo2_1_T2V_00125.png",
"fullpath": "N:\\AI\\ComfyUI\\output\\WanVideo2_1_T2V_00125.mp4"
}
}
}
},
{
"id": 37,
"type": "WanVideoEmptyEmbeds",
"pos": [
1305.26708984375,
-571.7843627929688
],
"size": [
315,
106
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "image_embeds",
"type": "WANVIDIMAGE_EMBEDS",
"links": [
42
]
}
],
"properties": {
"Node name for S&R": "WanVideoEmptyEmbeds"
},
"widgets_values": [
832,
480,
257
]
},
{
"id": 43,
"type": "WanVideoContextOptions",
"pos": [
1307.9541015625,
-776.4462280273438
],
"size": [
315,
154
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "context_options",
"type": "WANVIDCONTEXT",
"links": [
57
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoContextOptions"
},
"widgets_values": [
"uniform_standard",
81,
4,
16,
true
]
},
{
"id": 35,
"type": "WanVideoTorchCompileSettings",
"pos": [
193.47103881835938,
-614.6900024414062
],
"size": [
390.5999755859375,
178
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "torch_compile_args",
"type": "WANCOMPILEARGS",
"links": [
54
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "WanVideoTorchCompileSettings"
},
"widgets_values": [
"inductor",
false,
"default",
false,
64,
true
]
}
],
"links": [
[
15,
11,
0,
16,
0,
"WANTEXTENCODER"
],
[
29,
22,
0,
27,
0,
"WANVIDEOMODEL"
],
[
30,
16,
0,
27,
1,
"WANVIDEOTEXTEMBEDS"
],
[
33,
27,
0,
28,
1,
"LATENT"
],
[
42,
37,
0,
27,
2,
"WANVIDIMAGE_EMBEDS"
],
[
43,
38,
0,
28,
0,
"VAE"
],
[
48,
28,
0,
42,
0,
"IMAGE"
],
[
54,
35,
0,
22,
0,
"WANCOMPILEARGS"
],
[
56,
42,
0,
30,
0,
"IMAGE"
],
[
57,
43,
0,
27,
5,
"WANVIDCONTEXT"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.7400249944258609,
"offset": [
-125.57518289152554,
940.6273495482347
]
},
"node_versions": {
"ComfyUI-WanVideoWrapper": "dfe8000e63aaa961e1e4c71d14ce47ed22a419bc",
"ComfyUI-KJNodes": "dc482957d814a5a78000a3452b6c623a48fbd992",
"ComfyUI-VideoHelperSuite": "2c25b8b53835aaeb63f831b3137c705cf9f85dce"
},
"VHS_latentpreview": true,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+18 -2
View File
@@ -1112,11 +1112,13 @@ class WanVideoSampler:
disable_enhance()
if context_options is not None:
counter = torch.zeros_like(latent_model_input[0], device=offload_device)
noise_pred = torch.zeros_like(latent_model_input[0], device=offload_device)
context_queue = list(context(
i, steps, latent_video_length, context_frames, context_stride, context_overlap,
))
for c in context_queue:
partial_latent_model_input = [latent_model_input[0][:, c, :, :]]
# Model inference - returns [frames, channels, height, width]
@@ -1130,8 +1132,22 @@ class WanVideoSampler:
noise_pred_cond - noise_pred_uncond)
else:
noise_pred_context = noise_pred_cond
noise_pred[:, c, :, :] += noise_pred_context
counter[:, c, :, :] += 1
window_mask = torch.ones_like(noise_pred_context)
# Apply left-side blending for all except first chunk
if min(c) > 0:
ramp_up = torch.linspace(0, 1, context_overlap, device=noise_pred.device)
ramp_up = ramp_up.view(1, -1, 1, 1)
window_mask[:, :context_overlap] = ramp_up
# Apply right-side blending for all except last chunk
if max(c) < latent_video_length - 1:
ramp_down = torch.linspace(1, 0, context_overlap, device=noise_pred.device)
ramp_down = ramp_down.view(1, -1, 1, 1)
window_mask[:, -context_overlap:] = ramp_down
# Apply masked prediction
noise_pred[:, c, :, :] += noise_pred_context * window_mask
counter[:, c, :, :] += window_mask
#model inference end
noise_pred /= counter
else: