diff --git a/echoshot/echoshot.py b/echoshot/echoshot.py index 49857df..659387c 100644 --- a/echoshot/echoshot.py +++ b/echoshot/echoshot.py @@ -60,6 +60,45 @@ def rope_apply_c(x, freqs, inner_c, shift=6): # apply rotary embedding x_i = torch.view_as_real(x_i * freqs_i).flatten(2) + # append to collection + output.append(x_i) + return torch.stack(output).float() + +@torch.autocast(device_type=get_autocast_device(get_torch_device()), enabled=False) +@torch.compiler.disable() +def rope_apply_echoshot(x, grid_sizes, freqs, inner_t, shift=4): + n, c = x.size(2), x.size(3) // 2 + + # split freqs + freqs = freqs.split([c - 2 * (c // 3), c // 3, c // 3], dim=1) + + # loop over samples + output = [] + for i, (f, h, w) in enumerate(grid_sizes.tolist()): + seq_len = f * h * w + + # precompute multipliers + x_i = torch.view_as_complex( + x[i, :seq_len].to(torch.float64).reshape(seq_len, n, -1, 2) + ) + start_ind = [sum(inner_t[i][:_]) for _ in range(len(inner_t[i]))] + end_ind = [sum(inner_t[i][:_+1]) for _ in range(len(inner_t[i]))] + freq_select = [] + for shot_ind, (s, e) in enumerate(zip(start_ind, end_ind)): + freq_select += list(range(shot_ind * shift + s, shot_ind * shift + e)) + t_freqs = freqs[0][freq_select] + + freqs_i = torch.cat([ + # freqs[0][:f].view(f, 1, 1, -1).expand(f, h, w, -1), + t_freqs.view(f, 1, 1, -1).expand(f, h, w, -1), ### + freqs[1][:h].view(1, h, 1, -1).expand(f, h, w, -1), + freqs[2][:w].view(1, 1, w, -1).expand(f, h, w, -1) + ], dim=-1).reshape(seq_len, 1, -1) + + # apply rotary embedding + x_i = torch.view_as_real(x_i * freqs_i).flatten(2) + x_i = torch.cat([x_i, x[i, seq_len:]]) + # append to collection output.append(x_i) return torch.stack(output).float() \ No newline at end of file diff --git a/example_workflows/wanvideo_1_3B_EchoShot_example.json b/example_workflows/wanvideo_1_3B_EchoShot_example.json index e121d28..b3c3f9c 100644 --- a/example_workflows/wanvideo_1_3B_EchoShot_example.json +++ b/example_workflows/wanvideo_1_3B_EchoShot_example.json @@ -1,8 +1,8 @@ { "id": "206247b6-9fec-4ed2-8927-e4f388c674d4", "revision": 0, - "last_node_id": 101, - "last_link_id": 160, + "last_node_id": 112, + "last_link_id": 179, "nodes": [ { "id": 11, @@ -25,7 +25,8 @@ "type": "WANTEXTENCODER", "slot_index": 0, "links": [ - 15 + 162, + 176 ] } ], @@ -43,58 +44,6 @@ "color": "#332922", "bgcolor": "#593930" }, - { - "id": 28, - "type": "WanVideoDecode", - "pos": [ - 1688.0194091796875, - -647.6461791992188 - ], - "size": [ - 315, - 198 - ], - "flags": {}, - "order": 12, - "mode": 0, - "inputs": [ - { - "name": "vae", - "type": "WANVAE", - "link": 159 - }, - { - "name": "samples", - "type": "LATENT", - "link": 117 - } - ], - "outputs": [ - { - "name": "images", - "type": "IMAGE", - "slot_index": 0, - "links": [ - 36 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", - "Node name for S&R": "WanVideoDecode" - }, - "widgets_values": [ - false, - 272, - 272, - 144, - 128, - "default" - ], - "color": "#322", - "bgcolor": "#533" - }, { "id": 35, "type": "WanVideoTorchCompileSettings", @@ -115,9 +64,7 @@ "name": "torch_compile_args", "type": "WANCOMPILEARGS", "slot_index": 0, - "links": [ - 150 - ] + "links": [] } ], "properties": { @@ -151,14 +98,22 @@ "flags": {}, "order": 2, "mode": 0, - "inputs": [], + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + } + ], "outputs": [ { "name": "vae", "type": "WANVAE", "slot_index": 0, "links": [ - 159 + 159, + 169 ] } ], @@ -174,191 +129,6 @@ "color": "#322", "bgcolor": "#533" }, - { - "id": 16, - "type": "WanVideoTextEncode", - "pos": [ - 707.3203125, - -114.53802490234375 - ], - "size": [ - 522.08447265625, - 636.5288696289062 - ], - "flags": {}, - "order": 10, - "mode": 0, - "inputs": [ - { - "name": "t5", - "shape": 7, - "type": "WANTEXTENCODER", - "link": 15 - }, - { - "name": "model_to_offload", - "shape": 7, - "type": "WANVIDEOMODEL", - "link": 79 - } - ], - "outputs": [ - { - "name": "text_embeds", - "type": "WANVIDEOTEXTEMBEDS", - "slot_index": 0, - "links": [ - 30 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", - "Node name for S&R": "WanVideoTextEncode" - }, - "widgets_values": [ - "[1] A red panda with soft, reddish-brown fur and a bushy striped tail is perched on a wooden bench in the heart of a bustling city. The panda is joyfully eating a large, colorful ice cream cone, its small paws gripping the treat as it licks it with delight. Around the panda, tall glass skyscrapers reflect the afternoon sunlight, and busy city dwellers walk past on the sidewalk. Street vendors line the avenue, and cars and buses move along the road, creating a lively urban atmosphere.\n\n[2] In a vibrant, sun-dappled forest, a red panda with bright, expressive eyes is running swiftly along a narrow dirt path. The forest is dense with towering trees whose green leaves form a thick canopy overhead, allowing rays of sunlight to filter through and create shifting patterns on the ground. Ferns, wildflowers, and mossy rocks line the path, and birds can be seen flitting between the branches. The air is fresh and filled with the sounds of rustling leaves and distant animal calls, making the scene feel alive and natural.\n\n[3] Inside a cozy, warmly lit bathroom, a red panda is taking a relaxing bath in a classic white porcelain bathtub filled with fluffy bubbles. The panda leans back, its fur slightly damp, and playfully splashes water with its paws. The bathroom walls are decorated with pastel-colored tiles, and a small window lets in soft morning light. On the edge of the tub sits a yellow rubber duck, and a potted plant adds a touch of greenery to the scene, creating a peaceful and inviting atmosphere.", - "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards\"", - true, - false, - "gpu" - ], - "color": "#332922", - "bgcolor": "#593930" - }, - { - "id": 22, - "type": "WanVideoModelLoader", - "pos": [ - 157.20700073242188, - -839.4575805664062 - ], - "size": [ - 624.1026611328125, - 274 - ], - "flags": {}, - "order": 9, - "mode": 0, - "inputs": [ - { - "name": "compile_args", - "shape": 7, - "type": "WANCOMPILEARGS", - "link": 150 - }, - { - "name": "block_swap_args", - "shape": 7, - "type": "BLOCKSWAPARGS", - "link": null - }, - { - "name": "lora", - "shape": 7, - "type": "WANVIDLORA", - "link": 152 - }, - { - "name": "vram_management_args", - "shape": 7, - "type": "VRAM_MANAGEMENTARGS", - "link": null - }, - { - "name": "vace_model", - "shape": 7, - "type": "VACEPATH", - "link": null - }, - { - "name": "fantasytalking_model", - "shape": 7, - "type": "FANTASYTALKINGMODEL", - "link": null - }, - { - "name": "multitalk_model", - "shape": 7, - "type": "MULTITALKMODEL", - "link": null - } - ], - "outputs": [ - { - "name": "model", - "type": "WANVIDEOMODEL", - "slot_index": 0, - "links": [ - 79, - 144 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", - "Node name for S&R": "WanVideoModelLoader" - }, - "widgets_values": [ - "WanVideo\\EchoShot\\Wan2_1-T2V-1-3B-EchoShot_fp16.safetensors", - "fp16_fast", - "disabled", - "offload_device", - "sageattn" - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 78, - "type": "WanVideoEmptyEmbeds", - "pos": [ - 869.359375, - -317.0496520996094 - ], - "size": [ - 272.431640625, - 126 - ], - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "name": "control_embeds", - "shape": 7, - "type": "WANVIDIMAGE_EMBEDS", - "link": null - }, - { - "name": "extra_latents", - "shape": 7, - "type": "LATENT", - "link": null - } - ], - "outputs": [ - { - "name": "image_embeds", - "type": "WANVIDIMAGE_EMBEDS", - "links": [ - 102 - ] - } - ], - "properties": { - "cnr_id": "ComfyUI-WanVideoWrapper", - "ver": "6bc53b771d5d2af316801cb69e2ee10dbf7d18b1", - "Node name for S&R": "WanVideoEmptyEmbeds" - }, - "widgets_values": [ - 832, - 480, - 149 - ] - }, { "id": 68, "type": "WanVideoLoraSelect", @@ -371,7 +141,7 @@ 164.70230102539062 ], "flags": {}, - "order": 8, + "order": 11, "mode": 0, "inputs": [ { @@ -420,16 +190,14 @@ 106 ], "flags": {}, - "order": 4, + "order": 3, "mode": 0, "inputs": [], "outputs": [ { "name": "feta_args", "type": "FETAARGS", - "links": [ - 157 - ] + "links": [] } ], "properties": { @@ -455,7 +223,7 @@ 200 ], "flags": {}, - "order": 7, + "order": 10, "mode": 0, "inputs": [ { @@ -492,83 +260,6 @@ true ] }, - { - "id": 30, - "type": "VHS_VideoCombine", - "pos": [ - 1684.1597900390625, - -394.2595520019531 - ], - "size": [ - 951.5560913085938, - 334 - ], - "flags": {}, - "order": 13, - "mode": 0, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 36 - }, - { - "name": "audio", - "shape": 7, - "type": "AUDIO", - "link": null - }, - { - "name": "meta_batch", - "shape": 7, - "type": "VHS_BatchManager", - "link": null - }, - { - "name": "vae", - "shape": 7, - "type": "VAE", - "link": null - } - ], - "outputs": [ - { - "name": "Filenames", - "type": "VHS_FILENAMES", - "links": null - } - ], - "properties": { - "cnr_id": "comfyui-videohelpersuite", - "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", - "Node name for S&R": "VHS_VideoCombine" - }, - "widgets_values": { - "frame_rate": 16, - "loop_count": 0, - "filename_prefix": "WanVideoWrapper_EchoShot", - "format": "video/h264-mp4", - "pix_fmt": "yuv420p", - "crf": 19, - "save_metadata": true, - "trim_to_audio": false, - "pingpong": false, - "save_output": false, - "videopreview": { - "hidden": false, - "paused": false, - "params": { - "filename": "WanVideoWrapper_EchoShot_00003.mp4", - "subfolder": "", - "type": "temp", - "format": "video/h264-mp4", - "frame_rate": 16, - "workflow": "WanVideoWrapper_EchoShot_00003.png", - "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideoWrapper_EchoShot_00003.mp4" - } - } - } - }, { "id": 101, "type": "WanVideoLoraSelect", @@ -581,7 +272,7 @@ 200 ], "flags": {}, - "order": 5, + "order": 4, "mode": 0, "inputs": [ { @@ -630,7 +321,7 @@ 197.84388732910156 ], "flags": {}, - "order": 6, + "order": 5, "mode": 0, "inputs": [], "outputs": [], @@ -642,18 +333,409 @@ "bgcolor": "#653" }, { - "id": 27, - "type": "WanVideoSampler", + "id": 22, + "type": "WanVideoModelLoader", "pos": [ - 1315.2401123046875, - -401.48028564453125 + 157.20700073242188, + -839.4575805664062 + ], + "size": [ + 624.1026611328125, + 274 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": null + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": 152 + }, + { + "name": "vram_management_args", + "shape": 7, + "type": "VRAM_MANAGEMENTARGS", + "link": null + }, + { + "name": "vace_model", + "shape": 7, + "type": "VACEPATH", + "link": null + }, + { + "name": "fantasytalking_model", + "shape": 7, + "type": "FANTASYTALKINGMODEL", + "link": null + }, + { + "name": "multitalk_model", + "shape": 7, + "type": "MULTITALKMODEL", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "slot_index": 0, + "links": [ + 144, + 163, + 171, + 177 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoModelLoader" + }, + "widgets_values": [ + "WanVideo\\EchoShot\\Wan2_1-T2V-1-3B-EchoShot_fp16.safetensors", + "fp16_fast", + "disabled", + "offload_device", + "sageattn" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 78, + "type": "WanVideoEmptyEmbeds", + "pos": [ + 869.359375, + -317.0496520996094 + ], + "size": [ + 272.431640625, + 126 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "control_embeds", + "shape": 7, + "type": "WANVIDIMAGE_EMBEDS", + "link": null + }, + { + "name": "extra_latents", + "shape": 7, + "type": "LATENT", + "link": null + } + ], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 102, + 172 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "6bc53b771d5d2af316801cb69e2ee10dbf7d18b1", + "Node name for S&R": "WanVideoEmptyEmbeds" + }, + "widgets_values": [ + 832, + 480, + 93 + ] + }, + { + "id": 28, + "type": "WanVideoDecode", + "pos": [ + 1688.0194091796875, + -647.6461791992188 ], "size": [ 315, - 802.1923217773438 + 198 ], "flags": {}, - "order": 11, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 159 + }, + { + "name": "samples", + "type": "LATENT", + "link": 117 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 36 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 106, + "type": "WanVideoDecode", + "pos": [ + 1669.4420166015625, + 42.821956634521484 + ], + "size": [ + 315, + 198 + ], + "flags": {}, + "order": 18, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 169 + }, + { + "name": "samples", + "type": "LATENT", + "link": 174 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 175 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + false, + 272, + 272, + 144, + 128, + "default" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 104, + "type": "WanVideoTextEncode", + "pos": [ + 379.3588562011719, + -66.9586410522461 + ], + "size": [ + 543.7053833007812, + 759.9481811523438 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "t5", + "shape": 7, + "type": "WANTEXTENCODER", + "link": 162 + }, + { + "name": "model_to_offload", + "shape": 7, + "type": "WANVIDEOMODEL", + "link": 163 + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "slot_index": 0, + "links": [ + 166 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoTextEncode" + }, + "widgets_values": [ + "[1]这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为静止站立,身体略微前倾,头部轻微转动,似乎在与人交谈。视频场景为:背景是一个金属结构的墙面,墙面上有许多圆形灯光,灯光呈模糊状态。墙面结构具有重复的线条和几何形状。整体场景显得简洁而现代。视频打光为整体光线柔和,主要光源来自前方,照亮了该人物的面部和上半身\n\n[2]这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为坐在咖啡馆的靠窗位置,手里端着一杯咖啡,轻轻地品尝着。视频场景为一个充满情调的咖啡馆,空气中弥漫着咖啡的香气,柔和的灯光照亮着桌椅,墙上挂着艺术画作,营造出一种温馨浪漫的氛围。视频打光为室内灯光与室外光线相结合,整体光线柔和,突出了咖啡和人物的惬意\n\n[3]这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为在海边缓缓散步,时不时停下脚步,眺望远方。视频场景为开阔的海滩,海浪轻轻拍打着沙滩,海风吹拂着他的头发和衣角,远处海天一色,偶尔有海鸥飞过,显得宁静而广阔。视频打光为自然光,阳光洒在海面和人物身上,营造出一种温暖而自由的氛围。", + "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards\"", + true, + false, + "gpu" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 109, + "type": "WanVideoTextEncode", + "pos": [ + 396.4079895019531, + 762.1580810546875 + ], + "size": [ + 543.7053833007812, + 759.9481811523438 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "t5", + "shape": 7, + "type": "WANTEXTENCODER", + "link": 176 + }, + { + "name": "model_to_offload", + "shape": 7, + "type": "WANVIDEOMODEL", + "link": 177 + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "slot_index": 0, + "links": [ + 178 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoTextEncode" + }, + "widgets_values": [ + "这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为静止站立,身体略微前倾,头部轻微转动,似乎在与人交谈。视频场景为:背景是一个金属结构的墙面,墙面上有许多圆形灯光,灯光呈模糊状态。墙面结构具有重复的线条和几何形状。整体场景显得简洁而现代。视频打光为整体光线柔和,主要光源来自前方,照亮了该人物的面部和上半身。\n\n|\n\n这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为坐在咖啡馆的靠窗位置,手里端着一杯咖啡,轻轻地品尝着。视频场景为一个充满情调的咖啡馆,空气中弥漫着咖啡的香气,柔和的灯光照亮着桌椅,墙上挂着艺术画作,营造出一种温馨浪漫的氛围。视频打光为室内灯光与室外光线相结合,整体光线柔和,突出了咖啡和人物的惬意。\n\n|\n\n这个片段展示了一个白种人,男性,老年。该人物发型为短发,发色为灰白色,头发蓬松,略微向后梳。该人物服饰为身穿一件浅绿色衬衫,外面套着一件深灰色马甲。马甲上有口袋和拉链细节。该人物表情为平静,略带微笑,眼神专注,似乎在倾听或思考。该人物动作为在海边缓缓散步,时不时停下脚步,眺望远方。视频场景为开阔的海滩,海浪轻轻拍打着沙滩,海风吹拂着他的头发和衣角,远处海天一色,偶尔有海鸥飞过,显得宁静而广阔。视频打光为自然光,阳光洒在海面和人物身上,营造出一种温暖而自由的氛围。", + "Bright tones, overexposed, static, blurred details, subtitles, style, works, paintings, images, static, overall gray, worst quality, low quality, JPEG compression residue, ugly, incomplete, extra fingers, poorly drawn hands, poorly drawn faces, deformed, disfigured, misshapen limbs, fused fingers, still picture, messy background, three legs, many people in the background, walking backwards\"", + true, + false, + "gpu" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 110, + "type": "Note", + "pos": [ + 34.62555694580078, + 57.70677185058594 + ], + "size": [ + 312.06622314453125, + 126.66446685791016 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Original method, might be implemented wrong..." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 111, + "type": "Note", + "pos": [ + 48.427650451660156, + 786.52001953125 + ], + "size": [ + 312.06622314453125, + 126.66446685791016 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "My method of splitting crossattention" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1275.601806640625, + -586.78955078125 + ], + "size": [ + 315, + 873.39453125 + ], + "flags": {}, + "order": 15, "mode": 0, "inputs": [ { @@ -670,7 +752,7 @@ "name": "text_embeds", "shape": 7, "type": "WANVIDEOTEXTEMBEDS", - "link": 30 + "link": 166 }, { "name": "samples", @@ -682,7 +764,7 @@ "name": "feta_args", "shape": 7, "type": "FETAARGS", - "link": 157 + "link": null }, { "name": "context_options", @@ -781,37 +863,371 @@ 8, 1, 5, - 67, + 68, "fixed", true, - "dpm++_sde", + "euler", 0, 1, - "", + false, "comfy", 0, -1, + false, + "" + ] + }, + { + "id": 112, + "type": "WanVideoExperimentalArgs", + "pos": [ + 957.8191528320312, + 819.0532836914062 + ], + "size": [ + 308.767578125, + 250 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "exp_args", + "type": "EXPERIMENTALARGS", + "links": [ + 179 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "69dd4689bfd69d37d7f997289cb980e45f8b19bb", + "Node name for S&R": "WanVideoExperimentalArgs" + }, + "widgets_values": [ + "1", + false, + false, + 0, + false, + 1, + 1.25, + 20, false ] + }, + { + "id": 107, + "type": "WanVideoSampler", + "pos": [ + 1323.5477294921875, + 340.802734375 + ], + "size": [ + 315, + 873.39453125 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 171 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 172 + }, + { + "name": "text_embeds", + "shape": 7, + "type": "WANVIDEOTEXTEMBEDS", + "link": 178 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": null + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "cache_args", + "shape": 7, + "type": "CACHEARGS", + "link": null + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + }, + { + "name": "experimental_args", + "shape": 7, + "type": "EXPERIMENTALARGS", + "link": 179 + }, + { + "name": "sigmas", + "shape": 7, + "type": "SIGMAS", + "link": null + }, + { + "name": "unianimate_poses", + "shape": 7, + "type": "UNIANIMATE_POSE", + "link": null + }, + { + "name": "fantasytalking_embeds", + "shape": 7, + "type": "FANTASYTALKING_EMBEDS", + "link": null + }, + { + "name": "uni3c_embeds", + "shape": 7, + "type": "UNI3C_EMBEDS", + "link": null + }, + { + "name": "multitalk_embeds", + "shape": 7, + "type": "MULTITALK_EMBEDS", + "link": null + }, + { + "name": "freeinit_args", + "shape": 7, + "type": "FREEINITARGS", + "link": null + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 174 + ] + }, + { + "name": "denoised_samples", + "type": "LATENT", + "links": null + } + ], + "properties": { + "cnr_id": "ComfyUI-WanVideoWrapper", + "ver": "d9b1f4d1a5aea91d101ae97a54714a5861af3f50", + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 8, + 1, + 5, + 68, + "fixed", + true, + "euler", + 0, + 1, + false, + "comfy", + 0, + -1, + false, + "" + ] + }, + { + "id": 30, + "type": "VHS_VideoCombine", + "pos": [ + 2029.0123291015625, + -636.0532836914062 + ], + "size": [ + 951.5560913085938, + 885.4362182617188 + ], + "flags": {}, + "order": 19, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 36 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideoWrapper_EchoShot", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideoWrapper_EchoShot_00014.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideoWrapper_EchoShot_00014.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideoWrapper_EchoShot_00014.mp4" + } + } + } + }, + { + "id": 108, + "type": "VHS_VideoCombine", + "pos": [ + 1989.6697998046875, + 300.8671875 + ], + "size": [ + 951.5560913085938, + 885.4362182617188 + ], + "flags": {}, + "order": 20, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 175 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "cnr_id": "comfyui-videohelpersuite", + "ver": "0a75c7958fe320efcb052f1d9f8451fd20c730a8", + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideoWrapper_EchoShot", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": false, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideoWrapper_EchoShot_00016.mp4", + "subfolder": "", + "type": "temp", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideoWrapper_EchoShot_00016.png", + "fullpath": "N:\\AI\\ComfyUI\\temp\\WanVideoWrapper_EchoShot_00016.mp4" + } + } + } } ], "links": [ - [ - 15, - 11, - 0, - 16, - 0, - "WANTEXTENCODER" - ], - [ - 30, - 16, - 0, - 27, - 2, - "WANVIDEOTEXTEMBEDS" - ], [ 36, 28, @@ -820,14 +1236,6 @@ 0, "IMAGE" ], - [ - 79, - 22, - 0, - 16, - 1, - "WANVIDEOMODEL" - ], [ 102, 78, @@ -852,14 +1260,6 @@ 0, "WANVIDEOMODEL" ], - [ - 150, - 35, - 0, - 22, - 0, - "WANCOMPILEARGS" - ], [ 152, 68, @@ -876,14 +1276,6 @@ 0, "WANVIDLORA" ], - [ - 157, - 99, - 0, - 27, - 4, - "FETAARGS" - ], [ 159, 38, @@ -899,19 +1291,115 @@ 75, 0, "WANVIDLORA" + ], + [ + 162, + 11, + 0, + 104, + 0, + "WANTEXTENCODER" + ], + [ + 163, + 22, + 0, + 104, + 1, + "WANVIDEOMODEL" + ], + [ + 166, + 104, + 0, + 27, + 2, + "WANVIDEOTEXTEMBEDS" + ], + [ + 169, + 38, + 0, + 106, + 0, + "WANVAE" + ], + [ + 171, + 22, + 0, + 107, + 0, + "WANVIDEOMODEL" + ], + [ + 172, + 78, + 0, + 107, + 1, + "WANVIDIMAGE_EMBEDS" + ], + [ + 174, + 107, + 0, + 106, + 1, + "LATENT" + ], + [ + 175, + 106, + 0, + 108, + 0, + "IMAGE" + ], + [ + 176, + 11, + 0, + 109, + 0, + "WANTEXTENCODER" + ], + [ + 177, + 22, + 0, + 109, + 1, + "WANVIDEOMODEL" + ], + [ + 178, + 109, + 0, + 107, + 2, + "WANVIDEOTEXTEMBEDS" + ], + [ + 179, + 112, + 0, + 107, + 10, + "EXPERIMENTALARGS" ] ], "groups": [], "config": {}, "extra": { "ds": { - "scale": 0.6115909044841898, + "scale": 0.5559917313492645, "offset": [ - 1225.8534561534846, - 1177.513152807732 + 365.30269486183914, + 693.992397217299 ] }, - "frontendVersion": "1.25.1", + "frontendVersion": "1.25.3", "node_versions": { "ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd", "comfy-core": "0.3.26", diff --git a/nodes.py b/nodes.py index 5dbd1c3..233338f 100644 --- a/nodes.py +++ b/nodes.py @@ -1965,7 +1965,8 @@ class WanVideoSampler: shot_num = len(text_embeds["prompt_embeds"]) shot_len = [latent_video_length//shot_num] * (shot_num-1) shot_len.append(latent_video_length-sum(shot_len)) - log.info(f"EchoShot - Number of shots in prompt: {shot_num}, Shot token lengths: {shot_len}") + rope_function = "default" #echoshot does not support comfy rope function + log.info(f"Number of shots in prompt: {shot_num}, Shot token lengths: {shot_len}") #region transformer settings #rope @@ -2991,7 +2992,7 @@ class WanVideoSampler: # cache generated samples videos = torch.stack(videos).cpu() # B C T H W if colormatch != "disabled": - videos = videos[0].permute(1, 2, 3, 0).cpu().numpy() + videos = videos[0].permute(1, 2, 3, 0).cpu().float().numpy() from color_matcher import ColorMatcher cm = ColorMatcher() cm_result_list = [] @@ -3236,6 +3237,7 @@ class WanVideoDecode: if is_looped: temp_latents = torch.cat([latents[:, :, -3:]] + [latents[:, :, :2]], dim=2) temp_images = vae.decode(temp_latents, device=device, end_=(end_image is not None), tiled=enable_vae_tiling, tile_size=(tile_x//vae.upsampling_factor, tile_y//vae.upsampling_factor), tile_stride=(tile_stride_x//vae.upsampling_factor, tile_stride_y//vae.upsampling_factor))[0] + temp_images = temp_images.cpu().float() temp_images = (temp_images - temp_images.min()) / (temp_images.max() - temp_images.min()) images = torch.cat([temp_images[:, 9:].to(images), images[:, 5:]], dim=1) diff --git a/wanvideo/modules/model.py b/wanvideo/modules/model.py index bd31312..47df295 100644 --- a/wanvideo/modules/model.py +++ b/wanvideo/modules/model.py @@ -27,8 +27,9 @@ from comfy import model_management as mm from ...utils import log, get_module_memory_mb from ...cache_methods.cache_methods import TeaCacheState, MagCacheState, EasyCacheState, relative_l1_distance from ...multitalk.multitalk import get_attn_map_with_target -from ...echoshot.echoshot import rope_apply_z, rope_apply_c +from ...echoshot.echoshot import rope_apply_z, rope_apply_c, rope_apply_echoshot +from comfy.model_management import get_torch_device, get_autocast_device from comfy.ldm.flux.math import apply_rope as apply_rope_comfy def apply_rope_comfy_chunked(xq, xk, freqs_cis, num_chunks=4): @@ -155,7 +156,6 @@ def rope_params(max_seq_len, dim, theta=10000, L_test=25, k=0): freqs = torch.polar(torch.ones_like(freqs), freqs) return freqs -from comfy.model_management import get_torch_device, get_autocast_device @torch.autocast(device_type=get_autocast_device(get_torch_device()), enabled=False) @torch.compiler.disable() def rope_apply(x, grid_sizes, freqs): @@ -359,33 +359,30 @@ class WanSelfAttention(nn.Module): # Split by frames if multiple prompts are provided if seq_chunks > 1 and current_step in video_attention_split_steps: outputs = [] - # Extract frame, height, width from grid_sizes - force to CPU scalars - frames = grid_sizes[0][0].item() - height = grid_sizes[0][1].item() - width = grid_sizes[0][2].item() + # Extract frame, height, width from grid_sizes + frames = grid_sizes[0][0] + height = grid_sizes[0][1] + width = grid_sizes[0][2] tokens_per_frame = height * width - actual_chunks = min(seq_chunks, frames) - if isinstance(actual_chunks, torch.Tensor): - actual_chunks = actual_chunks.item() - - frame_chunks = [] # Pre-calculate all chunk boundaries - start_frame = 0 + actual_chunks = torch.min(torch.tensor(seq_chunks, device=frames.device), frames) base_frames_per_chunk = frames // actual_chunks extra_frames = frames % actual_chunks - # Pre-calculate all chunks - for i in range(actual_chunks): - chunk_size = base_frames_per_chunk + (1 if i < extra_frames else 0) - end_frame = start_frame + chunk_size - frame_chunks.append((start_frame, end_frame)) - start_frame = end_frame + # Calculate all chunk boundaries + chunk_indices = torch.arange(actual_chunks, device=frames.device) + chunk_sizes = base_frames_per_chunk + (chunk_indices < extra_frames).long() + chunk_starts = torch.cumsum(torch.cat([torch.zeros(1, device=frames.device), chunk_sizes[:-1]]), dim=0).long() + chunk_ends = chunk_starts + chunk_sizes - # Process each chunk using the pre-calculated boundaries - for start_frame, end_frame in frame_chunks: - # Convert to token indices - start_idx = int(start_frame * tokens_per_frame) - end_idx = int(end_frame * tokens_per_frame) + # Process each chunk using tensor indexing + for i in range(actual_chunks.item()): + start_frame = chunk_starts[i] + end_frame = chunk_ends[i] + + # Convert to token indices using tensor operations + start_idx = start_frame * tokens_per_frame + end_idx = end_frame * tokens_per_frame chunk_q = q[:, start_idx:end_idx, :, :] chunk_k = k[:, start_idx:end_idx, :, :] @@ -706,7 +703,10 @@ class WanAttentionBlock(nn.Module): feta_scores = get_feta_scores(q, k) #RoPE - if self.rope_func == "comfy": + if inner_t is not None: + q=rope_apply_echoshot(q, grid_sizes, freqs, inner_t).to(q) + k=rope_apply_echoshot(k, grid_sizes, freqs, inner_t).to(k) + elif self.rope_func == "comfy": q, k = apply_rope_comfy(q, k, freqs) elif self.rope_func == "comfy_chunked": q, k = apply_rope_comfy_chunked(q, k, freqs)