fix(ditto): Restore EMA Logic! Wire mouth_smoothing to MotionStitch instead of Audio2Motion v1.9.46

This commit is contained in:
Hawk Lee
2026-01-22 22:28:08 +08:00
parent d937f688e3
commit c936887aed
3 changed files with 14 additions and 52 deletions
+5 -18
View File
@@ -791,24 +791,10 @@ class AIIA_DittoSampler:
elif blink_mode == "Slow":
blink_min = 120
blink_max = 200
# Mouth Smoothing Logic [Fix v1.9.44]
# User requested Float Blend Factors (0.0 - 0.7)
# 0.0 = Raw, 0.5 = Normal Blend, 0.7 = Heavy Blend
if mouth_smoothing == "None (Raw)":
smo_k_d = 0.0
elif mouth_smoothing == "Light":
smo_k_d = 0.3
elif mouth_smoothing == "Heavy":
smo_k_d = 0.7
elif mouth_smoothing == "Custom (Manual)":
# Respect the integer input
pass
else: # Normal
smo_k_d = 0.5
# If "Normal" was default, we set 0.5.
# Note: We are hijacking smo_k_d (int) to pass a float.
# StreamSDK and Audio2Motion must be updated to handle this float.
# [Revert v1.9.46] Logic Separation
# smo_k_d: Controls Pose Smoothing Window (Audio2Motion). Uses direct INT input.
# mouth_smoothing: Controls Expression EMA Decay (MotionStitch). Passed via string.
# We no longer override smo_k_d based on mouth_smoothing.
# Map emo string to int
emo_map = {
@@ -823,6 +809,7 @@ class AIIA_DittoSampler:
"delta_yaw": hd_rot_y,
"delta_roll": hd_rot_r,
"mouth_amp": mouth_amp,
"mouth_smoothing": mouth_smoothing, # [New] Pass string to trigger EMA logic in MotionStitch
}
@@ -166,45 +166,20 @@ class Audio2Motion:
self.kp_cond = res_kp_seq[:, idx-1]
def _smo(self, res_kp_seq, s, e):
# [Upgrade v1.9.44] Support Float Factors (0.0 - 1.0) for Blending
# Legacy: Integers > 1 mean Window Size.
factor = float(self.smo_k_d)
if factor <= 0.01: # None (Raw) -> 0.0
return res_kp_seq
# New Logic: Weighted Blend
if factor < 1.0:
# Base Smoothing Layer: Window 5
k = 5
else:
# Legacy Logic: Use input as kernel size (must be int)
k = int(factor)
if k <= 1: return res_kp_seq
# Perform Smoothing
# [Revert v1.9.46] Back to Legacy Integer Window Smoothing (Pose Smoothing)
# Note: 'mouth_smoothing' (EMA) is now handled in MotionStitch, separate from this.
k = int(self.smo_k_d)
if k <= 1:
return res_kp_seq
new_res_kp_seq = res_kp_seq.copy()
n = res_kp_seq.shape[1]
half_k = k // 2
# Store smoothed result in temp array
smoothed_seq = res_kp_seq.copy()
for i in range(s, e):
ss = max(0, i - half_k)
ee = min(n, i + half_k + 1)
# Calculate Mean for this frame
smoothed_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1)
# Apply Blend if using factor mode
if factor < 1.0:
# Output = Raw * (1 - factor) + Smooth * factor
# 0.3 -> 70% Raw + 30% Smooth
# 0.7 -> 30% Raw + 70% Smooth
return res_kp_seq * (1.0 - factor) + smoothed_seq * factor
else:
# Legacy mode: Return smoothed directly
return smoothed_seq
res_kp_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1)
return res_kp_seq
def __call__(self, aud_cond, res_kp_seq=None, reset=False, step_len=None, seed=None):
"""
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "aiia"
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
version = "1.9.45"
version = "1.9.46"
license = {file = "LICENSE"}
readme = "README.md"
authors = [