v1.9.189: Predictive Pitch Bias & Volumetric Jaw Excitation. Implemented constant chin-tuck pressure and boosted lower lip gain to 1.2 to fix head-up drift and lazy mouth opening.

This commit is contained in:
Hawk Lee
2026-01-25 01:16:18 +08:00
parent 03eecb15f1
commit 474fc79e36
3 changed files with 13 additions and 16 deletions
@@ -119,8 +119,8 @@ class Audio2Motion:
self.brownian_momentum = np.zeros_like(self.kp_cond) # [v1.9.139] Postural inertia
self.look_up_timer = 0 # [v1.9.141] Timer for anti-stall recovery
self.is_recovering = False # [v1.9.155] Hysteresis state flag
self.target_bias_deg = 0.0 # [v1.9.162/163] For scope visibility
self.current_push = 0.0 # [v1.9.164] SAFETY: Initialize missing variable to prevent thread crash
self.target_bias_deg = -3.5 # [v1.9.189] Aesthetic Sweet Spot (Target Delta)
self.current_push = 0.0012 # [v1.9.189] Constant Chin-Tuck Pressure
# [v1.9.170] Pure Photo Anchor (Reverted Neutralizer)
self.photo_base_neutralizer = np.zeros_like(self.s_kp_cond)
@@ -248,15 +248,13 @@ class Audio2Motion:
# [v1.9.164] Predictive Physics Push
new_drift[0, 1:67] += self.current_push
# [v1.9.152] Absolute Postural Force (Boosted to 0.0030)
# [v1.9.189] Decoupled Axis Force (Chin-Tuck Mode)
if self.is_recovering:
# [v1.9.158] Decoupled Axis Force
# High tension for Pitch (1:67), Soft tension for Yaw/Roll/T (67:202)
new_drift[0, 1:67] -= 0.0030
new_drift[0, 67:202] -= 0.0010 # Softened Yaw/Roll pulse
# Add drift (push down) to recover from looking up too much
new_drift[0, 1:67] += 0.0035
new_drift[0, 67:202] -= 0.0010
if self.look_up_timer > 100:
new_drift[0, 1:67] -= 0.0030 # Double impulse for Pitch
new_drift[0, 1:67] += 0.0035
self.brownian_momentum = self.brownian_momentum * 0.92 + new_drift
self.brownian_pos += self.brownian_momentum
@@ -388,7 +386,7 @@ class Audio2Motion:
if self.clip_idx % 20 == 0:
mode_s = "SPEECH" if is_talking else "IDLE"
print(f"[v1.9.170 {mode_s}] Pressure: {pressure*100:.0f}% (Safety Net Active)")
print(f"[v1.9.189 {mode_s}] Pressure: {pressure*100:.0f}% (Delta={delta_p:+.2f} Target={self.target_bias_deg:+.1f})")
# [v1.9.156] Virtual Last Frame for Startup Stabilization
# If this is the VERY first chunk, we treat the source photo as the "prev frame"
@@ -533,8 +533,8 @@ class MotionStitch:
self.fix_exp_a2 = (1 - _a1) + _a1 * _a2
self.fix_exp_a3 = _a2
# [Debug v1.9.188] Verify Code Sync
print(f"[AIIA Debug] MotionStitch Setup: v1.9.188. LATEST VERSION LOADED.")
# [Debug v1.9.189] Verify Code Sync
print(f"[AIIA Debug] MotionStitch Setup: v1.9.189. LATEST VERSION LOADED.")
if self.drive_eye and self.delta_eye_arr is not None:
@@ -701,9 +701,8 @@ class MotionStitch:
# Only boost if the AI is actually trying to open the mouth (y > 0)
mask = y_intent > 0.001
if np.any(mask):
# Gain(y) = 1.0 + 0.8 * exp(-y / 0.04)
# High gain for small y, moves to 1.0 as y increases.
gain = 1.0 + 0.8 * np.exp(-y_intent[mask] / 0.04)
# [v1.9.189] Volumetric Jaw Excitation (Boosted to 1.2)
gain = 1.0 + 1.2 * np.exp(-y_intent[mask] / 0.04)
exp_reshaped[:, lower_lip, 1][mask] *= gain
# 2. Volumetric Mouth Micro-Motion (Breathing + Corners 7,8)
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "aiia"
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
version = "1.9.188"
version = "1.9.189"
license = {file = "LICENSE"}
readme = "README.md"
authors = [