diff --git a/example_workflows/wanvideo_long_T2V_example_01.json b/example_workflows/wanvideo_long_T2V_example_01.json new file mode 100644 index 0000000..2025da2 --- /dev/null +++ b/example_workflows/wanvideo_long_T2V_example_01.json @@ -0,0 +1,672 @@ +{ + "last_node_id": 44, + "last_link_id": 57, + "nodes": [ + { + "id": 33, + "type": "Note", + "pos": [ + 227.3764190673828, + -205.28524780273438 + ], + "size": [ + 351.70458984375, + 60 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Models:\nhttps://huggingface.co/Kijai/WanVideo_comfy/tree/main" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 11, + "type": "LoadWanVideoT5TextEncoder", + "pos": [ + 224.15325927734375, + -34.481563568115234 + ], + "size": [ + 377.1661376953125, + 130 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "wan_t5_model", + "type": "WANTEXTENCODER", + "links": [ + 15 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LoadWanVideoT5TextEncoder" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "offload_device", + "disabled" + ] + }, + { + "id": 28, + "type": "WanVideoDecode", + "pos": [ + 1692.973876953125, + -404.8614501953125 + ], + "size": [ + 315, + 174 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 43 + }, + { + "name": "samples", + "type": "LATENT", + "link": 33 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "links": [ + 48 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + true, + 272, + 272, + 144, + 128 + ] + }, + { + "id": 22, + "type": "WanVideoModelLoader", + "pos": [ + 620.3950805664062, + -357.8426818847656 + ], + "size": [ + 477.4410095214844, + 226.43276977539062 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "type": "WANCOMPILEARGS", + "shape": 7, + "link": 54 + }, + { + "name": "block_swap_args", + "type": "BLOCKSWAPARGS", + "shape": 7, + "link": null + }, + { + "name": "lora", + "type": "WANVIDLORA", + "shape": 7, + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 29 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoModelLoader" + }, + "widgets_values": [ + "WanVideo\\Wan2_1-T2V-1_3B_fp32.safetensors", + "fp16", + "disabled", + "offload_device", + "sageattn" + ] + }, + { + "id": 38, + "type": "WanVideoVAELoader", + "pos": [ + 1687.4093017578125, + -582.2750854492188 + ], + "size": [ + 416.25482177734375, + 82 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "vae", + "type": "WANVAE", + "links": [ + 43 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoVAELoader" + }, + "widgets_values": [ + "wanvideo\\Wan2_1_VAE_bf16.safetensors", + "bf16" + ] + }, + { + "id": 42, + "type": "GetImageSizeAndCount", + "pos": [ + 1708.7301025390625, + -140.99705505371094 + ], + "size": [ + 277.20001220703125, + 86 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 48 + } + ], + "outputs": [ + { + "name": "image", + "type": "IMAGE", + "links": [ + 56 + ], + "slot_index": 0 + }, + { + "name": "832 width", + "type": "INT", + "links": null + }, + { + "name": "480 height", + "type": "INT", + "links": null + }, + { + "name": "129 count", + "type": "INT", + "links": null + } + ], + "properties": { + "Node name for S&R": "GetImageSizeAndCount" + }, + "widgets_values": [] + }, + { + "id": 16, + "type": "WanVideoTextEncode", + "pos": [ + 675.8850708007812, + -36.032100677490234 + ], + "size": [ + 420.30511474609375, + 261.5306701660156 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "t5", + "type": "WANTEXTENCODER", + "link": 15 + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": [ + 30 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoTextEncode" + }, + "widgets_values": [ + "high quality nature video featuring a red panda balancing on a bamboo stem while a bird lands on it's head, on the background there is a waterfall", + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走", + true + ] + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1315.2401123046875, + -401.48028564453125 + ], + "size": [ + 315, + 534.1923217773438 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 29 + }, + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "link": 30 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 42 + }, + { + "name": "samples", + "type": "LATENT", + "shape": 7, + "link": null + }, + { + "name": "feta_args", + "type": "FETAARGS", + "shape": 7, + "link": null + }, + { + "name": "context_options", + "type": "WANVIDCONTEXT", + "shape": 7, + "link": 57 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 33 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 30, + 6, + 5, + 1057359483639288, + "fixed", + true, + "dpm++", + 0, + 1, + "" + ] + }, + { + "id": 36, + "type": "Note", + "pos": [ + 796.0189208984375, + -521.5020751953125 + ], + "size": [ + 298.2554016113281, + 108.62744140625 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "sdpa should work too, haven't tested flaash\n\nfp8_fast seems to cause huge quality degradation" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 30, + "type": "VHS_VideoCombine", + "pos": [ + 2127.120849609375, + -511.9014587402344 + ], + "size": [ + 873.2135620117188, + 840.2385864257812 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 56 + }, + { + "name": "audio", + "type": "AUDIO", + "shape": 7, + "link": null + }, + { + "name": "meta_batch", + "type": "VHS_BatchManager", + "shape": 7, + "link": null + }, + { + "name": "vae", + "type": "VAE", + "shape": 7, + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo2_1_T2V", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo2_1_T2V_00125.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo2_1_T2V_00125.png", + "fullpath": "N:\\AI\\ComfyUI\\output\\WanVideo2_1_T2V_00125.mp4" + } + } + } + }, + { + "id": 37, + "type": "WanVideoEmptyEmbeds", + "pos": [ + 1305.26708984375, + -571.7843627929688 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 42 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoEmptyEmbeds" + }, + "widgets_values": [ + 832, + 480, + 257 + ] + }, + { + "id": 43, + "type": "WanVideoContextOptions", + "pos": [ + 1307.9541015625, + -776.4462280273438 + ], + "size": [ + 315, + 154 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "context_options", + "type": "WANVIDCONTEXT", + "links": [ + 57 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoContextOptions" + }, + "widgets_values": [ + "uniform_standard", + 81, + 4, + 16, + true + ] + }, + { + "id": 35, + "type": "WanVideoTorchCompileSettings", + "pos": [ + 193.47103881835938, + -614.6900024414062 + ], + "size": [ + 390.5999755859375, + 178 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "torch_compile_args", + "type": "WANCOMPILEARGS", + "links": [ + 54 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "WanVideoTorchCompileSettings" + }, + "widgets_values": [ + "inductor", + false, + "default", + false, + 64, + true + ] + } + ], + "links": [ + [ + 15, + 11, + 0, + 16, + 0, + "WANTEXTENCODER" + ], + [ + 29, + 22, + 0, + 27, + 0, + "WANVIDEOMODEL" + ], + [ + 30, + 16, + 0, + 27, + 1, + "WANVIDEOTEXTEMBEDS" + ], + [ + 33, + 27, + 0, + 28, + 1, + "LATENT" + ], + [ + 42, + 37, + 0, + 27, + 2, + "WANVIDIMAGE_EMBEDS" + ], + [ + 43, + 38, + 0, + 28, + 0, + "VAE" + ], + [ + 48, + 28, + 0, + 42, + 0, + "IMAGE" + ], + [ + 54, + 35, + 0, + 22, + 0, + "WANCOMPILEARGS" + ], + [ + 56, + 42, + 0, + 30, + 0, + "IMAGE" + ], + [ + 57, + 43, + 0, + 27, + 5, + "WANVIDCONTEXT" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.7400249944258609, + "offset": [ + -125.57518289152554, + 940.6273495482347 + ] + }, + "node_versions": { + "ComfyUI-WanVideoWrapper": "dfe8000e63aaa961e1e4c71d14ce47ed22a419bc", + "ComfyUI-KJNodes": "dc482957d814a5a78000a3452b6c623a48fbd992", + "ComfyUI-VideoHelperSuite": "2c25b8b53835aaeb63f831b3137c705cf9f85dce" + }, + "VHS_latentpreview": true, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/nodes.py b/nodes.py index d9e45db..0ded5ca 100644 --- a/nodes.py +++ b/nodes.py @@ -1112,11 +1112,13 @@ class WanVideoSampler: disable_enhance() if context_options is not None: + counter = torch.zeros_like(latent_model_input[0], device=offload_device) noise_pred = torch.zeros_like(latent_model_input[0], device=offload_device) context_queue = list(context( i, steps, latent_video_length, context_frames, context_stride, context_overlap, )) + for c in context_queue: partial_latent_model_input = [latent_model_input[0][:, c, :, :]] # Model inference - returns [frames, channels, height, width] @@ -1130,8 +1132,22 @@ class WanVideoSampler: noise_pred_cond - noise_pred_uncond) else: noise_pred_context = noise_pred_cond - noise_pred[:, c, :, :] += noise_pred_context - counter[:, c, :, :] += 1 + + window_mask = torch.ones_like(noise_pred_context) + # Apply left-side blending for all except first chunk + if min(c) > 0: + ramp_up = torch.linspace(0, 1, context_overlap, device=noise_pred.device) + ramp_up = ramp_up.view(1, -1, 1, 1) + window_mask[:, :context_overlap] = ramp_up + + # Apply right-side blending for all except last chunk + if max(c) < latent_video_length - 1: + ramp_down = torch.linspace(1, 0, context_overlap, device=noise_pred.device) + ramp_down = ramp_down.view(1, -1, 1, 1) + window_mask[:, -context_overlap:] = ramp_down + # Apply masked prediction + noise_pred[:, c, :, :] += noise_pred_context * window_mask + counter[:, c, :, :] += window_mask #model inference end noise_pred /= counter else: