From 66d4c5b90d7dc606e0c18a2b0387fecf14d7b020 Mon Sep 17 00:00:00 2001 From: kijai <40791699+kijai@users.noreply.github.com> Date: Mon, 25 Aug 2025 12:50:41 +0300 Subject: [PATCH] Slice possible alpha channel away before encoding --- multitalk/nodes.py | 4 ++-- nodes.py | 2 ++ 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/multitalk/nodes.py b/multitalk/nodes.py index cfd7428..0ecea7d 100644 --- a/multitalk/nodes.py +++ b/multitalk/nodes.py @@ -78,8 +78,8 @@ class MultiTalkWav2VecEmbeds: "normalize_loudness": ("BOOLEAN", {"default": True, "tooltip": "Normalize the audio loudness to -23 LUFS"}), "num_frames": ("INT", {"default": 81, "min": 1, "max": 10000, "step": 1, "tooltip": "The total frame count to generate."}), "fps": ("FLOAT", {"default": 25.0, "min": 1.0, "max": 60.0, "step": 0.1}), - "audio_scale": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 100.0, "step": 0.1, "tooltip": "Strength of the audio conditioning"}), - "audio_cfg_scale": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 100.0, "step": 0.1, "tooltip": "When not 1.0, an extra model pass without audio conditioning is done: slower inference but more motion is allowed"}), + "audio_scale": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 100.0, "step": 0.01, "tooltip": "Strength of the audio conditioning"}), + "audio_cfg_scale": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 100.0, "step": 0.01, "tooltip": "When not 1.0, an extra model pass without audio conditioning is done: slower inference but more motion is allowed"}), "multi_audio_type": (["para", "add"], {"default": "para", "tooltip": "'para' overlay speakers in parallel, 'add' concatenate sequentially"}), }, "optional" : { diff --git a/nodes.py b/nodes.py index 8002fcc..79775f6 100644 --- a/nodes.py +++ b/nodes.py @@ -864,6 +864,7 @@ class WanVideoImageToVideoEncode: # Resize and rearrange the input image dimensions if start_image is not None: + start_image = start_image[..., :3] if start_image.shape[1] != H or start_image.shape[2] != W: resized_start_image = common_upscale(start_image.movedim(-1, 1), W, H, "lanczos", "disabled").movedim(0, 1) else: @@ -873,6 +874,7 @@ class WanVideoImageToVideoEncode: resized_start_image = add_noise_to_reference_video(resized_start_image, ratio=noise_aug_strength) if end_image is not None: + end_image = end_image[..., :3] if end_image.shape[1] != H or end_image.shape[2] != W: resized_end_image = common_upscale(end_image.movedim(-1, 1), W, H, "lanczos", "disabled").movedim(0, 1) else: