fix(ditto): Restore EMA Logic! Wire mouth_smoothing to MotionStitch instead of Audio2Motion v1.9.46
This commit is contained in:
+5
-18
@@ -791,24 +791,10 @@ class AIIA_DittoSampler:
|
||||
elif blink_mode == "Slow":
|
||||
blink_min = 120
|
||||
blink_max = 200
|
||||
# Mouth Smoothing Logic [Fix v1.9.44]
|
||||
# User requested Float Blend Factors (0.0 - 0.7)
|
||||
# 0.0 = Raw, 0.5 = Normal Blend, 0.7 = Heavy Blend
|
||||
if mouth_smoothing == "None (Raw)":
|
||||
smo_k_d = 0.0
|
||||
elif mouth_smoothing == "Light":
|
||||
smo_k_d = 0.3
|
||||
elif mouth_smoothing == "Heavy":
|
||||
smo_k_d = 0.7
|
||||
elif mouth_smoothing == "Custom (Manual)":
|
||||
# Respect the integer input
|
||||
pass
|
||||
else: # Normal
|
||||
smo_k_d = 0.5
|
||||
# If "Normal" was default, we set 0.5.
|
||||
# Note: We are hijacking smo_k_d (int) to pass a float.
|
||||
# StreamSDK and Audio2Motion must be updated to handle this float.
|
||||
|
||||
# [Revert v1.9.46] Logic Separation
|
||||
# smo_k_d: Controls Pose Smoothing Window (Audio2Motion). Uses direct INT input.
|
||||
# mouth_smoothing: Controls Expression EMA Decay (MotionStitch). Passed via string.
|
||||
# We no longer override smo_k_d based on mouth_smoothing.
|
||||
|
||||
# Map emo string to int
|
||||
emo_map = {
|
||||
@@ -823,6 +809,7 @@ class AIIA_DittoSampler:
|
||||
"delta_yaw": hd_rot_y,
|
||||
"delta_roll": hd_rot_r,
|
||||
"mouth_amp": mouth_amp,
|
||||
"mouth_smoothing": mouth_smoothing, # [New] Pass string to trigger EMA logic in MotionStitch
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -166,45 +166,20 @@ class Audio2Motion:
|
||||
self.kp_cond = res_kp_seq[:, idx-1]
|
||||
|
||||
def _smo(self, res_kp_seq, s, e):
|
||||
# [Upgrade v1.9.44] Support Float Factors (0.0 - 1.0) for Blending
|
||||
# Legacy: Integers > 1 mean Window Size.
|
||||
factor = float(self.smo_k_d)
|
||||
|
||||
if factor <= 0.01: # None (Raw) -> 0.0
|
||||
return res_kp_seq
|
||||
|
||||
# New Logic: Weighted Blend
|
||||
if factor < 1.0:
|
||||
# Base Smoothing Layer: Window 5
|
||||
k = 5
|
||||
else:
|
||||
# Legacy Logic: Use input as kernel size (must be int)
|
||||
k = int(factor)
|
||||
if k <= 1: return res_kp_seq
|
||||
|
||||
# Perform Smoothing
|
||||
# [Revert v1.9.46] Back to Legacy Integer Window Smoothing (Pose Smoothing)
|
||||
# Note: 'mouth_smoothing' (EMA) is now handled in MotionStitch, separate from this.
|
||||
k = int(self.smo_k_d)
|
||||
if k <= 1:
|
||||
return res_kp_seq
|
||||
|
||||
new_res_kp_seq = res_kp_seq.copy()
|
||||
n = res_kp_seq.shape[1]
|
||||
half_k = k // 2
|
||||
|
||||
# Store smoothed result in temp array
|
||||
smoothed_seq = res_kp_seq.copy()
|
||||
|
||||
for i in range(s, e):
|
||||
ss = max(0, i - half_k)
|
||||
ee = min(n, i + half_k + 1)
|
||||
# Calculate Mean for this frame
|
||||
smoothed_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1)
|
||||
|
||||
# Apply Blend if using factor mode
|
||||
if factor < 1.0:
|
||||
# Output = Raw * (1 - factor) + Smooth * factor
|
||||
# 0.3 -> 70% Raw + 30% Smooth
|
||||
# 0.7 -> 30% Raw + 70% Smooth
|
||||
return res_kp_seq * (1.0 - factor) + smoothed_seq * factor
|
||||
else:
|
||||
# Legacy mode: Return smoothed directly
|
||||
return smoothed_seq
|
||||
res_kp_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1)
|
||||
return res_kp_seq
|
||||
|
||||
def __call__(self, aud_cond, res_kp_seq=None, reset=False, step_len=None, seed=None):
|
||||
"""
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "aiia"
|
||||
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
|
||||
version = "1.9.45"
|
||||
version = "1.9.46"
|
||||
license = {file = "LICENSE"}
|
||||
readme = "README.md"
|
||||
authors = [
|
||||
|
||||
Reference in New Issue
Block a user