v1.10.0: Add Speech Pitch Offset control and fix video combine UI bug

This commit is contained in:
Hawk Lee
2026-01-27 19:07:50 +08:00
parent ff187c9c00
commit ef2c434d01
5 changed files with 73 additions and 248 deletions
+4
View File
@@ -377,6 +377,10 @@ ComfyUI/models/EchoMimicV3/
- 只有低于此比例的音量才会被视为静音。
- `smo_k_d`: (默认 3) 运动平滑系数。数值越大动作越柔和,可抑制面部抖动。
- `hd_rot_p` / `y` / `r`: 头部旋转微调 (Pitch/Yaw/Roll)。
- `speech_pitch`: (v1.10.0 New) **说话时俯仰角补偿 (Speech Pitch Offset)**。
- 用于修正"说话时头抬得太高"或"需要低头说话"的场景。此偏移量仅在说话期间生效,并随语音强度平滑切入切出。
- **正值 (+) = 低头 (Look Down)**。例如 `5.0` 表示说话时微微低头。
- **负值 (-) = 抬头 (Look Up)**。
- `mouth_smoothing`: (v1.9.5 New) **嘴型惯性平滑 (Mouth Motion Inertia)**。
- 防止模型输出的嘴型瞬间开合(如爆破音时),增加物理惯性感。
- **`None (Raw)`**: 无平滑,模型原始输出。追求极致对口型,容忍偶尔快速开合。
-212
View File
@@ -1,212 +0,0 @@
import torch
import numpy as np
import os
import random
import tempfile
import soundfile as sf
import warnings
import sys
import subprocess
import folder_paths
from huggingface_hub import snapshot_download
# Suppress annoying warnings
warnings.filterwarnings("ignore", category=FutureWarning)
warnings.filterwarnings("ignore", category=UserWarning, module="onnxruntime")
os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE"
os.environ["ONNXRUNTIME_QUIET"] = "1"
# Lazy-loaded global variable
CosyVoice = None
def _install_cosyvoice_if_needed():
global CosyVoice
if CosyVoice is not None: return
try:
from cosyvoice.cli.cosyvoice import CosyVoice as CV
CosyVoice = CV
return
except ImportError: pass
try:
libs_dir = os.path.join(os.path.dirname(__file__), "libs")
cosyvoice_dir = os.path.join(libs_dir, "CosyVoice")
matcha_dir = os.path.join(cosyvoice_dir, "third_party", "Matcha-TTS")
if not os.path.exists(libs_dir): os.makedirs(libs_dir, exist_ok=True)
if not os.path.exists(cosyvoice_dir):
subprocess.check_call(["git", "clone", "--recursive", "https://github.com/FunAudioLLM/CosyVoice.git", cosyvoice_dir])
if cosyvoice_dir not in sys.path: sys.path.insert(0, cosyvoice_dir)
if matcha_dir not in sys.path: sys.path.insert(0, matcha_dir)
from cosyvoice.cli.cosyvoice import CosyVoice as CV
CosyVoice = CV
except Exception as e:
print(f"[AIIA] Failed to install/import CosyVoice: {e}")
class AIIA_CosyVoice_ModelLoader:
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"model_name": ([
"FunAudioLLM/Fun-CosyVoice3-0.5B-2512",
"FunAudioLLM/CosyVoice2-0.5B",
"CosyVoice-300M",
"CosyVoice-300M-SFT",
"CosyVoice-300M-Instruct"
],),
"use_fp16": ("BOOLEAN", {"default": True}),
}
}
RETURN_TYPES = ("COSYVOICE_MODEL",)
RETURN_NAMES = ("model",)
FUNCTION = "load_model"
CATEGORY = "AIIA/Loaders"
def load_model(self, model_name, use_fp16):
_install_cosyvoice_if_needed()
if model_name.startswith("FunAudioLLM/"):
model_dir = os.path.join(folder_paths.models_dir, "cosyvoice", model_name.split("/")[-1])
if not os.path.exists(model_dir):
snapshot_download(repo_id=model_name, local_dir=model_dir)
else:
model_dir = os.path.join(folder_paths.models_dir, "cosyvoice", model_name)
from cosyvoice.cli.cosyvoice import AutoModel
is_v3 = os.path.exists(os.path.join(model_dir, "cosyvoice3.yaml"))
is_v2 = os.path.exists(os.path.join(model_dir, "cosyvoice2.yaml")) or (not is_v3 and os.path.exists(os.path.join(model_dir, "flow.pt")))
print(f"[AIIA] Loading {'V3' if is_v3 else ('V2' if is_v2 else 'V1')} model from {model_dir}")
model_instance = AutoModel(model_dir=model_dir, fp16=use_fp16)
# Identity detection
available_spks = []
spk2info_path = os.path.join(model_dir, "spk2info.pt")
if os.path.exists(spk2info_path):
try: available_spks = list(torch.load(spk2info_path, map_location='cpu').keys())
except: pass
if "instruct" in model_dir.lower() and not is_v2 and not is_v3:
available_spks = sorted(list(set(available_spks + ["中文男", "中文女", "英文男", "英文女", "日语男", "粤语女", "韩语女"])))
return ({"model": model_instance, "model_dir": model_dir, "is_v3": is_v3, "is_v2": is_v2, "available_spks": available_spks},)
class AIIA_CosyVoice_V1_TTS:
"""Specialized node for 300M (V1) models with Surgical Fix for Male voices."""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"model": ("COSYVOICE_MODEL",),
"tts_text": ("STRING", {"multiline": True, "default": "你好,这是V1专号节点的测试。"}),
"instruct_text": ("STRING", {"multiline": True, "default": "Theo 'Crimson', is a fiery, passionate rebel leader."}),
"spk_id": ("STRING", {"default": "中文男"}),
"speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1}),
"seed": ("INT", {"default": 42, "min": -1, "max": 2147483647}),
},
"optional": {
"reference_audio": ("AUDIO",),
"prompt_text": ("STRING", {"multiline": True, "default": ""}),
}
}
RETURN_TYPES = ("AUDIO",)
FUNCTION = "generate"
CATEGORY = "AIIA/Synthesis"
def generate(self, model, tts_text, instruct_text, spk_id, speed, seed, reference_audio=None, prompt_text=""):
cosyvoice_model = model["model"]
if seed >= 0:
torch.manual_seed(seed)
if torch.cuda.is_available(): torch.cuda.manual_seed_all(seed)
# 1. Surgical Fix Logic for V1
# Check if it's actually V1
if model.get("is_v2") or model.get("is_v3"):
print("[AIIA] Warning: V1 node used with V2/V3 model. Falling back to native wrapper.")
output = cosyvoice_model.inference_instruct(tts_text, instruct_text, None, speed=speed)
else:
# PURE V1 SURGICAL PATH
if instruct_text:
print(f"[AIIA] V1 Surgical Instruct | Spk: {spk_id}")
clean_inst = instruct_text.strip().split("<|")[0].strip() + "<|endofprompt|>"
def gen():
chunks = cosyvoice_model.frontend.text_normalize(tts_text, split=True)
for c in chunks:
mi = cosyvoice_model.frontend.frontend_instruct(c, spk_id, clean_inst)
if 'llm_embedding' in mi: del mi['llm_embedding']
for o in cosyvoice_model.model.tts(**mi, stream=False, speed=speed): yield o
output = gen()
elif reference_audio is not None and prompt_text:
print("[AIIA] V1 Zero-shot")
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp:
wav = reference_audio["waveform"].squeeze().cpu().numpy()
if wav.ndim == 2: wav = wav.T
sf.write(tmp.name, wav, cosyvoice_model.sample_rate)
output = cosyvoice_model.inference_zero_shot(tts_text, prompt_text, tmp.name, speed=speed)
os.unlink(tmp.name)
else:
print(f"[AIIA] V1 SFT | Spk: {spk_id}")
output = cosyvoice_model.inference_sft(tts_text, spk_id, speed=speed)
all_speech = [c['tts_speech'] for c in output]
final_wav = torch.cat(all_speech, dim=-1)
return ({"waveform": final_wav.unsqueeze(0).cpu(), "sample_rate": cosyvoice_model.sample_rate},)
class AIIA_CosyVoice_V2V3_TTS:
"""Native node for 0.5B (V2/V3) models using official APIs."""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"model": ("COSYVOICE_MODEL",),
"tts_text": ("STRING", {"multiline": True, "default": "你好,这是V2/V3专用节点的测试。"}),
"instruct_text": ("STRING", {"multiline": True, "default": ""}),
"spk_id": ("STRING", {"default": ""}),
"speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1}),
"seed": ("INT", {"default": 42, "min": -1, "max": 2147483647}),
},
"optional": {
"reference_audio": ("AUDIO",),
}
}
RETURN_TYPES = ("AUDIO",)
FUNCTION = "generate"
CATEGORY = "AIIA/Synthesis"
def generate(self, model, tts_text, instruct_text, spk_id, speed, seed, reference_audio=None):
cosyvoice_model = model["model"]
if seed >= 0:
torch.manual_seed(seed)
if torch.cuda.is_available(): torch.cuda.manual_seed_all(seed)
ref_path = None
if reference_audio:
with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp:
ref_path = tmp.name
wav = reference_audio["waveform"].squeeze().cpu().numpy()
if wav.ndim == 2: wav = wav.T
sf.write(ref_path, wav, cosyvoice_model.sample_rate)
try:
if model["is_v3"]:
print(f"[AIIA] V3 Native | Spk: {spk_id}")
output = cosyvoice_model.inference_instruct2(tts_text, instruct_text, ref_path, zero_shot_spk_id=spk_id, speed=speed)
else:
print(f"[AIIA] V2 Native | Spk: {spk_id}")
output = cosyvoice_model.inference_instruct(tts_text, instruct_text, ref_path, zero_shot_spk_id=spk_id, speed=speed)
all_speech = [c['tts_speech'] for c in output]
final_wav = torch.cat(all_speech, dim=-1)
finally:
if ref_path and os.path.exists(ref_path): os.unlink(ref_path)
return ({"waveform": final_wav.unsqueeze(0).cpu(), "sample_rate": cosyvoice_model.sample_rate},)
NODE_CLASS_MAPPINGS = {
"AIIA_CosyVoice_ModelLoader": AIIA_CosyVoice_ModelLoader,
"AIIA_CosyVoice_V1_TTS": AIIA_CosyVoice_V1_TTS,
"AIIA_CosyVoice_V2V3_TTS": AIIA_CosyVoice_V2V3_TTS
}
NODE_DISPLAY_NAME_MAPPINGS = {
"AIIA_CosyVoice_ModelLoader": "CosyVoice Model Loader (AIIA)",
"AIIA_CosyVoice_V1_TTS": "CosyVoice V1 (300M) TTS",
"AIIA_CosyVoice_V2V3_TTS": "CosyVoice V2/V3 (0.5B+) TTS"
}
+38 -19
View File
@@ -499,6 +499,7 @@ class AIIA_DittoSampler:
"hd_rot_p": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}),
"hd_rot_y": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}),
"hd_rot_r": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}),
"speech_pitch": ("FLOAT", {"default": 0.0, "min": -20.0, "max": 20.0, "step": 1.0, "tooltip": "Pitch offset applied ONLY during speech. Positive = Look Down, Negative = Look Up."}),
"mouth_amp": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 2.0, "step": 0.05}),
"blink_amp": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 2.0, "step": 0.05}),
"relax_on_silence": ("BOOLEAN", {"default": True, "label_on": "Relax Face on Silence", "label_off": "Disabled"}),
@@ -522,7 +523,7 @@ class AIIA_DittoSampler:
FUNCTION = "generate"
CATEGORY = "AIIA/Ditto"
def generate(self, pipe, ref_image, audio, sampling_steps, fps, crop_scale, emo, drive_eye, chk_eye_blink, smo_k_d, hd_rot_p, hd_rot_y, hd_rot_r, mouth_amp, blink_amp, relax_on_silence, ref_threshold, blink_mode, speech_only_blink, silence_release, mouth_smoothing, save_to_disk, seed, prompt=None, unique_id=None):
def generate(self, pipe, ref_image, audio, sampling_steps, fps, crop_scale, emo, drive_eye, chk_eye_blink, smo_k_d, hd_rot_p, hd_rot_y, hd_rot_r, speech_pitch, mouth_amp, blink_amp, relax_on_silence, ref_threshold, blink_mode, speech_only_blink, silence_release, mouth_smoothing, save_to_disk, seed, prompt=None, unique_id=None):
# pipe is the dict we returned in Loader
master_sdk = pipe["sdk"]
cfg_pkl = pipe["cfg_pkl"]
@@ -760,7 +761,8 @@ class AIIA_DittoSampler:
# [v1.9.710] Continuous Vitality Planner (Distance-to-Boundary Envelope)
# Instead of a simple inverse of speech, we calculate an envelope based on
# the distance to the nearest speech boundary (onset/offset).
# This allows vitality to return during long speech segments while protecting the "Snap" at edges.
# This allows vitality to safely fade out near boundaries (preventing snap/drift)
# while keeping the character alive during both Silence AND Speech.
# 1. Identify Boundaries
# Using target_alpha (raw 0/1 VAD) to find start/end of speech blocks.
@@ -773,18 +775,21 @@ class AIIA_DittoSampler:
for i in range(num_frames):
# Distance to the nearest boundary
d = min(abs(i - b) for b in boundaries)
# Ramp Settings
# Silence Recovery: 25 frames (1.0s)
# Speech Recovery: 100 frames (4.0s) - User preferred 4s for long speech
is_speech = target_alpha[i] > 0.5
ramp_scale = 100.0 if is_speech else 25.0
if is_speech:
# Speech Recovery: How fast vitality returns after starting to speak.
# Was 100.0 (4s) -> Too static during short sentences.
# Changed to 25.0 (1.0s) -> Vitality returns quickly but smoothly.
ramp_scale = 25.0
target_peak = 0.8 # [Tweaked] Increase speech vitality (was 0.6)
else:
# Silence Recovery:
ramp_scale = 25.0 # 1.0s
target_peak = 1.0 # Full idle motion
# Calculate local weight (0.0 at boundary, ramping to target)
weight = min(1.0, d / ramp_scale)
# Peak Vitality: Silence=1.0, Speech=0.6 (Keep speech motion slightly lower)
target_peak = 0.6 if is_speech else 1.0
envelope[i] = weight * target_peak
# 3. Apply Procedural Motion
@@ -799,22 +804,36 @@ class AIIA_DittoSampler:
info_dict["vad_alpha"] = alpha
# Active Micro-Motion
# Only apply if we have weight (Silence or transition)
# Note: During Speech, weight=0.0, so d_pitch=0.0 -> Result = global hd_rot_p.
# weight is 0.0 at boundaries (Ensures return to Reference Pose)
# weight ramps up to target_peak (0.8/1.0) in middle of segments.
t = i / 25.0
# [v1.9.700] Tuned Vitality (Faster/Stronger)
d_pitch = (math.sin(t * 0.45) - 0.5) * 1.5 * weight
d_yaw = (math.sin(t * 0.75) * 0.8 + math.sin(t * 2.5) * 0.2) * idle_amp * weight
d_roll = math.cos(t * 0.5) * 0.5 * weight
# [v1.9.99] Tuned Vitality Formula
# Removed -0.5 bias from pitch (was looking down).
# Added faster roll component.
# Increased high-freq yaw component for speech.
# Mouth Breathing (uses same weight or separate?)
# Breathing should probably use the same weight to fade out during speech.
d_pitch = (math.sin(t * 0.45)) * 1.5 * weight
# Yaw: Mix slow sway (breathing) and faster micro-movements
# idle_amp default is 7.0 degrees.
d_yaw = (math.sin(t * 0.75) * 0.7 + math.sin(t * 2.5) * 0.3) * idle_amp * weight
# Roll: Add slight complexity
d_roll = (math.cos(t * 0.5) * 0.5 + math.sin(t * 1.5) * 0.2) * weight
# Mouth Breathing (uses same weight)
d_mouth = (math.sin(t * 2.5) + 1.0) * 0.5 * 0.005 * weight
# [v1.9.99] Speech Pitch Bias (User Control)
# Allows correcting "head too high/low" during speech.
# Smoothly fades in/out based on VAD alpha (mouth opening).
# Positive = Look Down, Negative = Look Up
speech_pitch_offset = speech_pitch * alpha
# Apply to dict
info_dict["delta_pitch"] = hd_rot_p + d_pitch
info_dict["delta_pitch"] = hd_rot_p + d_pitch + speech_pitch_offset
info_dict["delta_yaw"] = hd_rot_y + d_yaw
info_dict["delta_roll"] = hd_rot_r + d_roll
info_dict["delta_mouth"] = d_mouth
+30 -16
View File
@@ -12,15 +12,15 @@ function toggleWidget(node, widget, show = false) {
// console.log(`AIIA Debug (toggleWidget): Toggling '${widget.name}'. Should show: ${show}`);
// --- Debug End ---
if (!widget) return;
if (!origProps[widget.name]) {
origProps[widget.name] = {
origType: widget.type,
origComputeSize: widget.computeSize
};
}
widget.type = show ? origProps[widget.name].origType : "AIIA_HIDDEN";
widget.computeSize = show ? origProps[widget.name].origComputeSize : () => [0, -4];
if (!widget) return;
if (!origProps[widget.name]) {
origProps[widget.name] = {
origType: widget.type,
origComputeSize: widget.computeSize
};
}
widget.type = show ? origProps[widget.name].origType : "AIIA_HIDDEN";
widget.computeSize = show ? origProps[widget.name].origComputeSize : () => [0, -4];
node.setDirtyCanvas(true);
}
@@ -28,7 +28,7 @@ function toggleWidget(node, widget, show = false) {
function chainCallback(object, property, callback) {
if (object[property]) {
const original = object[property];
object[property] = function() {
object[property] = function () {
original.apply(this, arguments);
callback.apply(this, arguments);
};
@@ -41,15 +41,15 @@ app.registerExtension({
name: "AIIA.VideoNodes.DynamicWidgets.Final",
async beforeRegisterNodeDef(nodeType, nodeData, app) {
if (nodeData.name === "AIIA_VideoCombine") {
const widgetsByFormat = nodeData.input.required.format[1].formats;
if (!widgetsByFormat) return;
chainCallback(nodeType.prototype, "onNodeCreated", function() {
chainCallback(nodeType.prototype, "onNodeCreated", function () {
const node = this;
const formatWidget = findWidgetByName(node, "format");
if (!formatWidget) return;
const allDynamicWidgetNames = new Set(Object.values(widgetsByFormat).flat().map(p => p[0]));
const updateWidgetsVisibility = (formatValue) => {
@@ -57,23 +57,37 @@ app.registerExtension({
const visibleWidgetNames = new Set(
(widgetsByFormat[formatValue] || []).map(p => p[0])
);
for (const widgetName of allDynamicWidgetNames) {
const widget = findWidgetByName(node, widgetName);
if (widget) {
toggleWidget(node, widget, visibleWidgetNames.has(widgetName));
}
}
node.setSize([node.size[0], node.computeSize()[1]]);
};
// 为format widget的callback链接上更新函数
chainCallback(formatWidget, "callback", updateWidgetsVisibility);
// Expose function for onConfigure
node.aiiaUpdateVideoWidgets = updateWidgetsVisibility;
// 初始加载时触发
updateWidgetsVisibility(formatWidget.value);
});
chainCallback(nodeType.prototype, "onConfigure", function () {
const node = this;
// 使用 requestAnimationFrame 确保在所有widget值被写入后再执行更新
requestAnimationFrame(() => {
const formatWidget = findWidgetByName(node, "format");
if (formatWidget && node.aiiaUpdateVideoWidgets) {
node.aiiaUpdateVideoWidgets(formatWidget.value);
}
});
});
}
}
});
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "aiia"
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
version = "1.9.710"
version = "1.10.0"
license = {file = "LICENSE"}
readme = "README.md"
authors = [