diff --git a/aiia_ditto_nodes.py b/aiia_ditto_nodes.py index 030d4d5..1c3173c 100644 --- a/aiia_ditto_nodes.py +++ b/aiia_ditto_nodes.py @@ -791,24 +791,10 @@ class AIIA_DittoSampler: elif blink_mode == "Slow": blink_min = 120 blink_max = 200 - # Mouth Smoothing Logic [Fix v1.9.44] - # User requested Float Blend Factors (0.0 - 0.7) - # 0.0 = Raw, 0.5 = Normal Blend, 0.7 = Heavy Blend - if mouth_smoothing == "None (Raw)": - smo_k_d = 0.0 - elif mouth_smoothing == "Light": - smo_k_d = 0.3 - elif mouth_smoothing == "Heavy": - smo_k_d = 0.7 - elif mouth_smoothing == "Custom (Manual)": - # Respect the integer input - pass - else: # Normal - smo_k_d = 0.5 - # If "Normal" was default, we set 0.5. - # Note: We are hijacking smo_k_d (int) to pass a float. - # StreamSDK and Audio2Motion must be updated to handle this float. - + # [Revert v1.9.46] Logic Separation + # smo_k_d: Controls Pose Smoothing Window (Audio2Motion). Uses direct INT input. + # mouth_smoothing: Controls Expression EMA Decay (MotionStitch). Passed via string. + # We no longer override smo_k_d based on mouth_smoothing. # Map emo string to int emo_map = { @@ -823,6 +809,7 @@ class AIIA_DittoSampler: "delta_yaw": hd_rot_y, "delta_roll": hd_rot_r, "mouth_amp": mouth_amp, + "mouth_smoothing": mouth_smoothing, # [New] Pass string to trigger EMA logic in MotionStitch } diff --git a/libs/Ditto/core/atomic_components/audio2motion.py b/libs/Ditto/core/atomic_components/audio2motion.py index 7baa7b4..1067202 100644 --- a/libs/Ditto/core/atomic_components/audio2motion.py +++ b/libs/Ditto/core/atomic_components/audio2motion.py @@ -166,45 +166,20 @@ class Audio2Motion: self.kp_cond = res_kp_seq[:, idx-1] def _smo(self, res_kp_seq, s, e): - # [Upgrade v1.9.44] Support Float Factors (0.0 - 1.0) for Blending - # Legacy: Integers > 1 mean Window Size. - factor = float(self.smo_k_d) - - if factor <= 0.01: # None (Raw) -> 0.0 - return res_kp_seq - - # New Logic: Weighted Blend - if factor < 1.0: - # Base Smoothing Layer: Window 5 - k = 5 - else: - # Legacy Logic: Use input as kernel size (must be int) - k = int(factor) - if k <= 1: return res_kp_seq - - # Perform Smoothing + # [Revert v1.9.46] Back to Legacy Integer Window Smoothing (Pose Smoothing) + # Note: 'mouth_smoothing' (EMA) is now handled in MotionStitch, separate from this. + k = int(self.smo_k_d) + if k <= 1: + return res_kp_seq + new_res_kp_seq = res_kp_seq.copy() n = res_kp_seq.shape[1] half_k = k // 2 - - # Store smoothed result in temp array - smoothed_seq = res_kp_seq.copy() - for i in range(s, e): ss = max(0, i - half_k) ee = min(n, i + half_k + 1) - # Calculate Mean for this frame - smoothed_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1) - - # Apply Blend if using factor mode - if factor < 1.0: - # Output = Raw * (1 - factor) + Smooth * factor - # 0.3 -> 70% Raw + 30% Smooth - # 0.7 -> 30% Raw + 70% Smooth - return res_kp_seq * (1.0 - factor) + smoothed_seq * factor - else: - # Legacy mode: Return smoothed directly - return smoothed_seq + res_kp_seq[:, i, :202] = np.mean(new_res_kp_seq[:, ss:ee, :202], axis=1) + return res_kp_seq def __call__(self, aud_cond, res_kp_seq=None, reset=False, step_len=None, seed=None): """ diff --git a/pyproject.toml b/pyproject.toml index 630f146..323a8c4 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "aiia" description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser." -version = "1.9.45" +version = "1.9.46" license = {file = "LICENSE"} readme = "README.md" authors = [