diff --git a/README.md b/README.md index 962701a..4c6fbf1 100755 --- a/README.md +++ b/README.md @@ -377,6 +377,10 @@ ComfyUI/models/EchoMimicV3/ - 只有低于此比例的音量才会被视为静音。 - `smo_k_d`: (默认 3) 运动平滑系数。数值越大动作越柔和,可抑制面部抖动。 - `hd_rot_p` / `y` / `r`: 头部旋转微调 (Pitch/Yaw/Roll)。 + - `speech_pitch`: (v1.10.0 New) **说话时俯仰角补偿 (Speech Pitch Offset)**。 + - 用于修正"说话时头抬得太高"或"需要低头说话"的场景。此偏移量仅在说话期间生效,并随语音强度平滑切入切出。 + - **正值 (+) = 低头 (Look Down)**。例如 `5.0` 表示说话时微微低头。 + - **负值 (-) = 抬头 (Look Up)**。 - `mouth_smoothing`: (v1.9.5 New) **嘴型惯性平滑 (Mouth Motion Inertia)**。 - 防止模型输出的嘴型瞬间开合(如爆破音时),增加物理惯性感。 - **`None (Raw)`**: 无平滑,模型原始输出。追求极致对口型,容忍偶尔快速开合。 diff --git a/aiia_cosyvoice_nodes.py.bak b/aiia_cosyvoice_nodes.py.bak deleted file mode 100755 index b5a7e8b..0000000 --- a/aiia_cosyvoice_nodes.py.bak +++ /dev/null @@ -1,212 +0,0 @@ -import torch -import numpy as np -import os -import random -import tempfile -import soundfile as sf -import warnings -import sys -import subprocess -import folder_paths -from huggingface_hub import snapshot_download - -# Suppress annoying warnings -warnings.filterwarnings("ignore", category=FutureWarning) -warnings.filterwarnings("ignore", category=UserWarning, module="onnxruntime") -os.environ["KMP_DUPLICATE_LIB_OK"] = "TRUE" -os.environ["ONNXRUNTIME_QUIET"] = "1" - -# Lazy-loaded global variable -CosyVoice = None - -def _install_cosyvoice_if_needed(): - global CosyVoice - if CosyVoice is not None: return - try: - from cosyvoice.cli.cosyvoice import CosyVoice as CV - CosyVoice = CV - return - except ImportError: pass - - try: - libs_dir = os.path.join(os.path.dirname(__file__), "libs") - cosyvoice_dir = os.path.join(libs_dir, "CosyVoice") - matcha_dir = os.path.join(cosyvoice_dir, "third_party", "Matcha-TTS") - if not os.path.exists(libs_dir): os.makedirs(libs_dir, exist_ok=True) - if not os.path.exists(cosyvoice_dir): - subprocess.check_call(["git", "clone", "--recursive", "https://github.com/FunAudioLLM/CosyVoice.git", cosyvoice_dir]) - if cosyvoice_dir not in sys.path: sys.path.insert(0, cosyvoice_dir) - if matcha_dir not in sys.path: sys.path.insert(0, matcha_dir) - from cosyvoice.cli.cosyvoice import CosyVoice as CV - CosyVoice = CV - except Exception as e: - print(f"[AIIA] Failed to install/import CosyVoice: {e}") - -class AIIA_CosyVoice_ModelLoader: - @classmethod - def INPUT_TYPES(cls): - return { - "required": { - "model_name": ([ - "FunAudioLLM/Fun-CosyVoice3-0.5B-2512", - "FunAudioLLM/CosyVoice2-0.5B", - "CosyVoice-300M", - "CosyVoice-300M-SFT", - "CosyVoice-300M-Instruct" - ],), - "use_fp16": ("BOOLEAN", {"default": True}), - } - } - RETURN_TYPES = ("COSYVOICE_MODEL",) - RETURN_NAMES = ("model",) - FUNCTION = "load_model" - CATEGORY = "AIIA/Loaders" - - def load_model(self, model_name, use_fp16): - _install_cosyvoice_if_needed() - if model_name.startswith("FunAudioLLM/"): - model_dir = os.path.join(folder_paths.models_dir, "cosyvoice", model_name.split("/")[-1]) - if not os.path.exists(model_dir): - snapshot_download(repo_id=model_name, local_dir=model_dir) - else: - model_dir = os.path.join(folder_paths.models_dir, "cosyvoice", model_name) - - from cosyvoice.cli.cosyvoice import AutoModel - is_v3 = os.path.exists(os.path.join(model_dir, "cosyvoice3.yaml")) - is_v2 = os.path.exists(os.path.join(model_dir, "cosyvoice2.yaml")) or (not is_v3 and os.path.exists(os.path.join(model_dir, "flow.pt"))) - - print(f"[AIIA] Loading {'V3' if is_v3 else ('V2' if is_v2 else 'V1')} model from {model_dir}") - model_instance = AutoModel(model_dir=model_dir, fp16=use_fp16) - - # Identity detection - available_spks = [] - spk2info_path = os.path.join(model_dir, "spk2info.pt") - if os.path.exists(spk2info_path): - try: available_spks = list(torch.load(spk2info_path, map_location='cpu').keys()) - except: pass - if "instruct" in model_dir.lower() and not is_v2 and not is_v3: - available_spks = sorted(list(set(available_spks + ["中文男", "中文女", "英文男", "英文女", "日语男", "粤语女", "韩语女"]))) - - return ({"model": model_instance, "model_dir": model_dir, "is_v3": is_v3, "is_v2": is_v2, "available_spks": available_spks},) - -class AIIA_CosyVoice_V1_TTS: - """Specialized node for 300M (V1) models with Surgical Fix for Male voices.""" - @classmethod - def INPUT_TYPES(cls): - return { - "required": { - "model": ("COSYVOICE_MODEL",), - "tts_text": ("STRING", {"multiline": True, "default": "你好,这是V1专号节点的测试。"}), - "instruct_text": ("STRING", {"multiline": True, "default": "Theo 'Crimson', is a fiery, passionate rebel leader."}), - "spk_id": ("STRING", {"default": "中文男"}), - "speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1}), - "seed": ("INT", {"default": 42, "min": -1, "max": 2147483647}), - }, - "optional": { - "reference_audio": ("AUDIO",), - "prompt_text": ("STRING", {"multiline": True, "default": ""}), - } - } - RETURN_TYPES = ("AUDIO",) - FUNCTION = "generate" - CATEGORY = "AIIA/Synthesis" - - def generate(self, model, tts_text, instruct_text, spk_id, speed, seed, reference_audio=None, prompt_text=""): - cosyvoice_model = model["model"] - if seed >= 0: - torch.manual_seed(seed) - if torch.cuda.is_available(): torch.cuda.manual_seed_all(seed) - - # 1. Surgical Fix Logic for V1 - # Check if it's actually V1 - if model.get("is_v2") or model.get("is_v3"): - print("[AIIA] Warning: V1 node used with V2/V3 model. Falling back to native wrapper.") - output = cosyvoice_model.inference_instruct(tts_text, instruct_text, None, speed=speed) - else: - # PURE V1 SURGICAL PATH - if instruct_text: - print(f"[AIIA] V1 Surgical Instruct | Spk: {spk_id}") - clean_inst = instruct_text.strip().split("<|")[0].strip() + "<|endofprompt|>" - def gen(): - chunks = cosyvoice_model.frontend.text_normalize(tts_text, split=True) - for c in chunks: - mi = cosyvoice_model.frontend.frontend_instruct(c, spk_id, clean_inst) - if 'llm_embedding' in mi: del mi['llm_embedding'] - for o in cosyvoice_model.model.tts(**mi, stream=False, speed=speed): yield o - output = gen() - elif reference_audio is not None and prompt_text: - print("[AIIA] V1 Zero-shot") - with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp: - wav = reference_audio["waveform"].squeeze().cpu().numpy() - if wav.ndim == 2: wav = wav.T - sf.write(tmp.name, wav, cosyvoice_model.sample_rate) - output = cosyvoice_model.inference_zero_shot(tts_text, prompt_text, tmp.name, speed=speed) - os.unlink(tmp.name) - else: - print(f"[AIIA] V1 SFT | Spk: {spk_id}") - output = cosyvoice_model.inference_sft(tts_text, spk_id, speed=speed) - - all_speech = [c['tts_speech'] for c in output] - final_wav = torch.cat(all_speech, dim=-1) - return ({"waveform": final_wav.unsqueeze(0).cpu(), "sample_rate": cosyvoice_model.sample_rate},) - -class AIIA_CosyVoice_V2V3_TTS: - """Native node for 0.5B (V2/V3) models using official APIs.""" - @classmethod - def INPUT_TYPES(cls): - return { - "required": { - "model": ("COSYVOICE_MODEL",), - "tts_text": ("STRING", {"multiline": True, "default": "你好,这是V2/V3专用节点的测试。"}), - "instruct_text": ("STRING", {"multiline": True, "default": ""}), - "spk_id": ("STRING", {"default": ""}), - "speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1}), - "seed": ("INT", {"default": 42, "min": -1, "max": 2147483647}), - }, - "optional": { - "reference_audio": ("AUDIO",), - } - } - RETURN_TYPES = ("AUDIO",) - FUNCTION = "generate" - CATEGORY = "AIIA/Synthesis" - - def generate(self, model, tts_text, instruct_text, spk_id, speed, seed, reference_audio=None): - cosyvoice_model = model["model"] - if seed >= 0: - torch.manual_seed(seed) - if torch.cuda.is_available(): torch.cuda.manual_seed_all(seed) - - ref_path = None - if reference_audio: - with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp: - ref_path = tmp.name - wav = reference_audio["waveform"].squeeze().cpu().numpy() - if wav.ndim == 2: wav = wav.T - sf.write(ref_path, wav, cosyvoice_model.sample_rate) - - try: - if model["is_v3"]: - print(f"[AIIA] V3 Native | Spk: {spk_id}") - output = cosyvoice_model.inference_instruct2(tts_text, instruct_text, ref_path, zero_shot_spk_id=spk_id, speed=speed) - else: - print(f"[AIIA] V2 Native | Spk: {spk_id}") - output = cosyvoice_model.inference_instruct(tts_text, instruct_text, ref_path, zero_shot_spk_id=spk_id, speed=speed) - - all_speech = [c['tts_speech'] for c in output] - final_wav = torch.cat(all_speech, dim=-1) - finally: - if ref_path and os.path.exists(ref_path): os.unlink(ref_path) - - return ({"waveform": final_wav.unsqueeze(0).cpu(), "sample_rate": cosyvoice_model.sample_rate},) - -NODE_CLASS_MAPPINGS = { - "AIIA_CosyVoice_ModelLoader": AIIA_CosyVoice_ModelLoader, - "AIIA_CosyVoice_V1_TTS": AIIA_CosyVoice_V1_TTS, - "AIIA_CosyVoice_V2V3_TTS": AIIA_CosyVoice_V2V3_TTS -} -NODE_DISPLAY_NAME_MAPPINGS = { - "AIIA_CosyVoice_ModelLoader": "CosyVoice Model Loader (AIIA)", - "AIIA_CosyVoice_V1_TTS": "CosyVoice V1 (300M) TTS", - "AIIA_CosyVoice_V2V3_TTS": "CosyVoice V2/V3 (0.5B+) TTS" -} diff --git a/aiia_ditto_nodes.py b/aiia_ditto_nodes.py index c65460b..32c72ec 100644 --- a/aiia_ditto_nodes.py +++ b/aiia_ditto_nodes.py @@ -499,6 +499,7 @@ class AIIA_DittoSampler: "hd_rot_p": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}), "hd_rot_y": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}), "hd_rot_r": ("FLOAT", {"default": 0.0, "min": -30.0, "max": 30.0, "step": 1.0}), + "speech_pitch": ("FLOAT", {"default": 0.0, "min": -20.0, "max": 20.0, "step": 1.0, "tooltip": "Pitch offset applied ONLY during speech. Positive = Look Down, Negative = Look Up."}), "mouth_amp": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 2.0, "step": 0.05}), "blink_amp": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 2.0, "step": 0.05}), "relax_on_silence": ("BOOLEAN", {"default": True, "label_on": "Relax Face on Silence", "label_off": "Disabled"}), @@ -522,7 +523,7 @@ class AIIA_DittoSampler: FUNCTION = "generate" CATEGORY = "AIIA/Ditto" - def generate(self, pipe, ref_image, audio, sampling_steps, fps, crop_scale, emo, drive_eye, chk_eye_blink, smo_k_d, hd_rot_p, hd_rot_y, hd_rot_r, mouth_amp, blink_amp, relax_on_silence, ref_threshold, blink_mode, speech_only_blink, silence_release, mouth_smoothing, save_to_disk, seed, prompt=None, unique_id=None): + def generate(self, pipe, ref_image, audio, sampling_steps, fps, crop_scale, emo, drive_eye, chk_eye_blink, smo_k_d, hd_rot_p, hd_rot_y, hd_rot_r, speech_pitch, mouth_amp, blink_amp, relax_on_silence, ref_threshold, blink_mode, speech_only_blink, silence_release, mouth_smoothing, save_to_disk, seed, prompt=None, unique_id=None): # pipe is the dict we returned in Loader master_sdk = pipe["sdk"] cfg_pkl = pipe["cfg_pkl"] @@ -760,7 +761,8 @@ class AIIA_DittoSampler: # [v1.9.710] Continuous Vitality Planner (Distance-to-Boundary Envelope) # Instead of a simple inverse of speech, we calculate an envelope based on # the distance to the nearest speech boundary (onset/offset). - # This allows vitality to return during long speech segments while protecting the "Snap" at edges. + # This allows vitality to safely fade out near boundaries (preventing snap/drift) + # while keeping the character alive during both Silence AND Speech. # 1. Identify Boundaries # Using target_alpha (raw 0/1 VAD) to find start/end of speech blocks. @@ -773,18 +775,21 @@ class AIIA_DittoSampler: for i in range(num_frames): # Distance to the nearest boundary d = min(abs(i - b) for b in boundaries) - - # Ramp Settings - # Silence Recovery: 25 frames (1.0s) - # Speech Recovery: 100 frames (4.0s) - User preferred 4s for long speech is_speech = target_alpha[i] > 0.5 - ramp_scale = 100.0 if is_speech else 25.0 + + if is_speech: + # Speech Recovery: How fast vitality returns after starting to speak. + # Was 100.0 (4s) -> Too static during short sentences. + # Changed to 25.0 (1.0s) -> Vitality returns quickly but smoothly. + ramp_scale = 25.0 + target_peak = 0.8 # [Tweaked] Increase speech vitality (was 0.6) + else: + # Silence Recovery: + ramp_scale = 25.0 # 1.0s + target_peak = 1.0 # Full idle motion # Calculate local weight (0.0 at boundary, ramping to target) weight = min(1.0, d / ramp_scale) - - # Peak Vitality: Silence=1.0, Speech=0.6 (Keep speech motion slightly lower) - target_peak = 0.6 if is_speech else 1.0 envelope[i] = weight * target_peak # 3. Apply Procedural Motion @@ -799,22 +804,36 @@ class AIIA_DittoSampler: info_dict["vad_alpha"] = alpha # Active Micro-Motion - # Only apply if we have weight (Silence or transition) - # Note: During Speech, weight=0.0, so d_pitch=0.0 -> Result = global hd_rot_p. + # weight is 0.0 at boundaries (Ensures return to Reference Pose) + # weight ramps up to target_peak (0.8/1.0) in middle of segments. t = i / 25.0 - # [v1.9.700] Tuned Vitality (Faster/Stronger) - d_pitch = (math.sin(t * 0.45) - 0.5) * 1.5 * weight - d_yaw = (math.sin(t * 0.75) * 0.8 + math.sin(t * 2.5) * 0.2) * idle_amp * weight - d_roll = math.cos(t * 0.5) * 0.5 * weight + # [v1.9.99] Tuned Vitality Formula + # Removed -0.5 bias from pitch (was looking down). + # Added faster roll component. + # Increased high-freq yaw component for speech. - # Mouth Breathing (uses same weight or separate?) - # Breathing should probably use the same weight to fade out during speech. + d_pitch = (math.sin(t * 0.45)) * 1.5 * weight + + # Yaw: Mix slow sway (breathing) and faster micro-movements + # idle_amp default is 7.0 degrees. + d_yaw = (math.sin(t * 0.75) * 0.7 + math.sin(t * 2.5) * 0.3) * idle_amp * weight + + # Roll: Add slight complexity + d_roll = (math.cos(t * 0.5) * 0.5 + math.sin(t * 1.5) * 0.2) * weight + + # Mouth Breathing (uses same weight) d_mouth = (math.sin(t * 2.5) + 1.0) * 0.5 * 0.005 * weight + + # [v1.9.99] Speech Pitch Bias (User Control) + # Allows correcting "head too high/low" during speech. + # Smoothly fades in/out based on VAD alpha (mouth opening). + # Positive = Look Down, Negative = Look Up + speech_pitch_offset = speech_pitch * alpha # Apply to dict - info_dict["delta_pitch"] = hd_rot_p + d_pitch + info_dict["delta_pitch"] = hd_rot_p + d_pitch + speech_pitch_offset info_dict["delta_yaw"] = hd_rot_y + d_yaw info_dict["delta_roll"] = hd_rot_r + d_roll info_dict["delta_mouth"] = d_mouth diff --git a/js/aiia_video_nodes.js b/js/aiia_video_nodes.js index 1c86639..9fee6f8 100755 --- a/js/aiia_video_nodes.js +++ b/js/aiia_video_nodes.js @@ -12,15 +12,15 @@ function toggleWidget(node, widget, show = false) { // console.log(`AIIA Debug (toggleWidget): Toggling '${widget.name}'. Should show: ${show}`); // --- Debug End --- - if (!widget) return; - if (!origProps[widget.name]) { - origProps[widget.name] = { - origType: widget.type, - origComputeSize: widget.computeSize - }; - } - widget.type = show ? origProps[widget.name].origType : "AIIA_HIDDEN"; - widget.computeSize = show ? origProps[widget.name].origComputeSize : () => [0, -4]; + if (!widget) return; + if (!origProps[widget.name]) { + origProps[widget.name] = { + origType: widget.type, + origComputeSize: widget.computeSize + }; + } + widget.type = show ? origProps[widget.name].origType : "AIIA_HIDDEN"; + widget.computeSize = show ? origProps[widget.name].origComputeSize : () => [0, -4]; node.setDirtyCanvas(true); } @@ -28,7 +28,7 @@ function toggleWidget(node, widget, show = false) { function chainCallback(object, property, callback) { if (object[property]) { const original = object[property]; - object[property] = function() { + object[property] = function () { original.apply(this, arguments); callback.apply(this, arguments); }; @@ -41,15 +41,15 @@ app.registerExtension({ name: "AIIA.VideoNodes.DynamicWidgets.Final", async beforeRegisterNodeDef(nodeType, nodeData, app) { if (nodeData.name === "AIIA_VideoCombine") { - + const widgetsByFormat = nodeData.input.required.format[1].formats; if (!widgetsByFormat) return; - chainCallback(nodeType.prototype, "onNodeCreated", function() { + chainCallback(nodeType.prototype, "onNodeCreated", function () { const node = this; const formatWidget = findWidgetByName(node, "format"); if (!formatWidget) return; - + const allDynamicWidgetNames = new Set(Object.values(widgetsByFormat).flat().map(p => p[0])); const updateWidgetsVisibility = (formatValue) => { @@ -57,23 +57,37 @@ app.registerExtension({ const visibleWidgetNames = new Set( (widgetsByFormat[formatValue] || []).map(p => p[0]) ); - + for (const widgetName of allDynamicWidgetNames) { const widget = findWidgetByName(node, widgetName); if (widget) { toggleWidget(node, widget, visibleWidgetNames.has(widgetName)); } } - + node.setSize([node.size[0], node.computeSize()[1]]); }; // 为format widget的callback链接上更新函数 chainCallback(formatWidget, "callback", updateWidgetsVisibility); - + + // Expose function for onConfigure + node.aiiaUpdateVideoWidgets = updateWidgetsVisibility; + // 初始加载时触发 updateWidgetsVisibility(formatWidget.value); }); + + chainCallback(nodeType.prototype, "onConfigure", function () { + const node = this; + // 使用 requestAnimationFrame 确保在所有widget值被写入后再执行更新 + requestAnimationFrame(() => { + const formatWidget = findWidgetByName(node, "format"); + if (formatWidget && node.aiiaUpdateVideoWidgets) { + node.aiiaUpdateVideoWidgets(formatWidget.value); + } + }); + }); } } }); \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml index 6a2da48..3ee21df 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "aiia" description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser." -version = "1.9.710" +version = "1.10.0" license = {file = "LICENSE"} readme = "README.md" authors = [