From cecc49da31ea62b8244391dd1ccd8b87fd9b79bc Mon Sep 17 00:00:00 2001 From: Enrico Date: Mon, 19 Jan 2026 23:58:52 +0100 Subject: [PATCH] tensor fix --- audio_blender.py | 15 +++++++++++---- test_parsing.py | 30 ++++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 4 deletions(-) create mode 100644 test_parsing.py diff --git a/audio_blender.py b/audio_blender.py index 2489729..b0005c9 100644 --- a/audio_blender.py +++ b/audio_blender.py @@ -111,11 +111,18 @@ class AudioOverlapBlender: ) # Reshape for broadcasting - # Audio latent is typically [batch, channels, frames, features] - while alpha.dim() < prev_tail.dim(): - alpha = alpha.unsqueeze(0) + # Audio latent is [batch, channels, frames, freq_bins] + # Alpha needs to be [1, 1, actual_overlap, 1] to broadcast if prev_tail.dim() == 4: - alpha = alpha.unsqueeze(-1) # [1, 1, overlap, 1] + # Audio: [B, C, T, F] -> alpha: [1, 1, overlap, 1] + alpha = alpha.view(1, 1, actual_overlap, 1) + elif prev_tail.dim() == 5: + # Video: [B, C, T, H, W] -> alpha: [1, 1, overlap, 1, 1] + alpha = alpha.view(1, 1, actual_overlap, 1, 1) + else: + # Fallback: add dims at front + while alpha.dim() < prev_tail.dim(): + alpha = alpha.unsqueeze(0) alpha = alpha.expand_as(prev_tail) diff --git a/test_parsing.py b/test_parsing.py new file mode 100644 index 0000000..fe22197 --- /dev/null +++ b/test_parsing.py @@ -0,0 +1,30 @@ +import sys +sys.path.insert(0, r'D:\ComfyUI7\ComfyUI\custom_nodes\ComfyUI-Erosdiffusion-LTX2') + +from script_parser import parse_scene_script + +# Test script 1: User's format +script1 = """[00:00.00-00:04.04] the woman turns from left to right, static camera | audio:techno music | first:ComfyUI-zimage-diffusers-wrapper_00213_.png | end:ComfyUI-zimage-diffusers-wrapper_00213_.png +[00:04.04-00:08.00] the woman turns from right to left, static camera | first:ComfyUI-zimage-diffusers-wrapper_00213_.png | audio:techno music | end:ComfyUI-zimage-diffusers-wrapper_00212_.png""" + +# Test script 2: Missing parts +script2 = """[00:00-00:04] no prompt no guides +[00:04-00:08] just a prompt | first:image.png +[00:08-00:12] | audio:silence""" + +print("=== Test 1: User's script ===") +chunks = parse_scene_script(script1) +print(f"Parsed {len(chunks)} chunks:") +for i, c in enumerate(chunks): + print(f" Chunk {i}: {c.start_sec:.2f}s-{c.end_sec:.2f}s") + print(f" prompt: '{c.prompt[:50]}...'") + print(f" audio: '{c.audio_spec}'") + print(f" guides: {len(c.guides)}") + for g in c.guides: + print(f" - {g.position}: {g.image_ref}") + +print("\n=== Test 2: Missing parts ===") +chunks2 = parse_scene_script(script2) +print(f"Parsed {len(chunks2)} chunks:") +for i, c in enumerate(chunks2): + print(f" Chunk {i}: prompt='{c.prompt}', audio='{c.audio_spec}', guides={len(c.guides)}")