From e968fc688bedc352c4cb95b6979920f0052b4602 Mon Sep 17 00:00:00 2001 From: Hawk Lee Date: Thu, 29 Jan 2026 13:47:52 +0800 Subject: [PATCH] feat: implement subtitle calibration and subtitle-to-segments conversion (v1.10.3) --- README.md | 16 +++- aiia_ditto_nodes.py | 28 +++++- aiia_subtitle_nodes.py | 198 ++++++++++++++++++++++++++++++++++++++++- pyproject.toml | 2 +- 4 files changed, 236 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 4c6fbf1..684ba42 100755 --- a/README.md +++ b/README.md @@ -1018,7 +1018,8 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860 **[v1.7.0 New]** 无需 STT,直接从生成过程中提取精准时间轴。 - **Input**: - - `segments_info`: 来自 `AIIA Dialogue TTS` 的输出。 + - `segments_info`: 来自 `AIIA Dialogue TTS` 或 `AIIA Generate Segments` 的输出。 + - `calibration_info` (可选): **[v1.10.2 新增]** 接入 `AIIA Generate Speaker Segments` 的输出。用于将估算的时间轴自动“吸附”到真实的 VAD 语音活动区间,解决 VibeVoice 等批处理引擎的时间轴偏移问题。 - **Output**: - `SRT`: 通用字幕格式。 - `ASS`: 高级排版字幕格式 (自动区分角色颜色)。 @@ -1026,7 +1027,18 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860 - **CosyVoice**: 使用生成时的精确时长。 - **VibeVoice**: 使用**智能插值算法 (Smart Interpolation)**,根据字符长度自动计算长音频段内的单句时间轴。 -#### 4.4 AIIA Subtitle Preview (字幕预览) +#### 4.4 AIIA Subtitle to Segments (字幕转分段) + +**[v1.10.3 New]** 将现有的 SRT/ASS 字幕文件转换为 `segments_info` 格式,以便进行时间轴重新校准。 + +- **Input**: + - `subtitle_text`: SRT 或 ASS 格式的文本内容。 + - `subtitle_path` (可选): 字幕文件的本地路径(如果提供,将优先读取文件)。 +- **Output**: + - `segments_info`: 标准化的 JSON 字符串,可直接输入到 `AIIA Subtitle Gen`。 +- **用途**: 结合 `Subtitle Gen` 的 `calibration_info` 输入,可以将**旧的、不准的字幕**自动对齐到**新的、精准的音轨**上。 + +#### 4.5 AIIA Subtitle Preview (字幕预览) **[v1.7.1 New]** 实时校验音画同步效果。 diff --git a/aiia_ditto_nodes.py b/aiia_ditto_nodes.py index 32c72ec..256a136 100644 --- a/aiia_ditto_nodes.py +++ b/aiia_ditto_nodes.py @@ -748,7 +748,30 @@ class AIIA_DittoSampler: current_alpha = max(target, current_alpha - release_step) dataset_alpha[i] = current_alpha - + + # [v1.10.0] Independent Head Pitch Envelope + # We want the head to nod/tilt SLOWLY when speech starts (0.8s), + # while the mouth opens INSTANTLY (0.08s). + # So we calculate a second alpha specifically for the head pitch offset. + + head_pitch_alpha = np.zeros(num_frames, dtype=np.float32) + current_head_alpha = target_alpha[0] + # [v1.10.0 Tuned] Faster attack to catch initial head lift. + # 0.05 (0.8s) was too slow -> Head lifted before correction. + # 0.20 (0.2s) is balanced -> Fast enough to clamp lift, smooth enough to avoid snap. + head_attack_step = 0.20 + # Use same release step as mouth to return to neutral naturally + + for i in range(num_frames): + target = target_alpha[i] + if target > current_head_alpha: + # Slow Attack + current_head_alpha = min(target, current_head_alpha + head_attack_step) + else: + # Same Release + current_head_alpha = max(target, current_head_alpha - release_step) + head_pitch_alpha[i] = current_head_alpha + # Log VAD stats for debugging non_silence_count = np.count_nonzero(target_alpha) logging.info(f"[Ditto] VAD Stats: {non_silence_count}/{num_frames} frames active. RMS Mean: {np.mean(rms):.4f}, Min: {np.min(rms):.4f}, Max: {np.max(rms):.4f}") @@ -830,7 +853,8 @@ class AIIA_DittoSampler: # Allows correcting "head too high/low" during speech. # Smoothly fades in/out based on VAD alpha (mouth opening). # Positive = Look Down, Negative = Look Up - speech_pitch_offset = speech_pitch * alpha + # [v1.10.0] Use smoothed alpha for head to avoid "snap" motion. + speech_pitch_offset = speech_pitch * head_pitch_alpha[i] # Apply to dict info_dict["delta_pitch"] = hd_rot_p + d_pitch + speech_pitch_offset diff --git a/aiia_subtitle_nodes.py b/aiia_subtitle_nodes.py index c9b04a0..c84f8f0 100755 --- a/aiia_subtitle_nodes.py +++ b/aiia_subtitle_nodes.py @@ -18,6 +18,7 @@ class AIIA_Subtitle_Gen: "save_file": ("BOOLEAN", {"default": False, "label_on": "Save to Disk", "label_off": "Memory Only"}), }, "optional": { + "calibration_info": ("WHISPER_CHUNKS",), "ass_style": ("STRING", {"default": "Default", "multiline": False}), "filename_prefix": ("STRING", {"default": "aiia_subtitle"}), } @@ -29,7 +30,7 @@ class AIIA_Subtitle_Gen: CATEGORY = "AIIA/Subtitle" OUTPUT_NODE = True - def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle"): + def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle", calibration_info=None): try: segments = json.loads(segments_info) except Exception as e: @@ -40,6 +41,15 @@ class AIIA_Subtitle_Gen: print("[AIIA Subtitle] Segments info must be a list of dicts.") return ("", "") + # --- Subtitle Calibration (v1.10.2) --- + if calibration_info and "chunks" in calibration_info: + print(f"[AIIA Subtitle] Calibrating {len(segments)} segments using {len(calibration_info['chunks'])} high-precision chunks.") + segments = self._calibrate_segments(segments, calibration_info["chunks"]) + + if not isinstance(segments, list): + print("[AIIA Subtitle] Segments info must be a list of dicts.") + return ("", "") + srt_out = "" ass_out = "" @@ -169,6 +179,72 @@ class AIIA_Subtitle_Gen: return f"{hours:02}:{minutes:02}:{secs:02},{millis:03}" + def _calibrate_segments(self, segments, chunks): + """ + Calibrate estimated segments using high-precision VAD chunks. + Algorithm: Iterative sequence matching with overlap weight. + """ + calibrated = [] + chunk_idx = 0 + num_chunks = len(chunks) + + for i, seg in enumerate(segments): + seg_start = seg["start"] + seg_end = seg["end"] + + best_match_start = -1 + best_match_end = -1 + + # Find chunks that overlap with this segment + # We look ahead starting from chunk_idx to maintain sequential order + matched_chunks = [] + + # Tolerance window: How far we can look for a matching chunk if no direct overlap + # 1.0s is reasonable for VibeVoice drift + lookahead_limit = 5 + + find_idx = chunk_idx + while find_idx < num_chunks and len(matched_chunks) < lookahead_limit: + chunk = chunks[find_idx] + c_start, c_end = chunk["timestamp"] + + # Check Overlap + overlap = min(seg_end, c_end) - max(seg_start, c_start) + + # If significant overlap, or if it's the very first chunk and we are near start + if overlap > 0.05 or (i == 0 and find_idx == 0 and abs(c_start - seg_start) < 2.0): + matched_chunks.append(find_idx) + + # Break if we've passed the segment significantly + if c_start > seg_end + 1.0: + break + find_idx += 1 + + if matched_chunks: + # Use the range of all matched chunks + # This handles cases where one sentence is split into multiple VAD chunks due to pauses + min_s = chunks[matched_chunks[0]]["timestamp"][0] + max_e = chunks[matched_chunks[-1]]["timestamp"][1] + + # Update chunk_idx to favor the next chunk for subsequent segments + chunk_idx = matched_chunks[-1] + 1 + + seg["start"] = round(min_s, 3) + seg["end"] = round(max_e, 3) + else: + # No match found within window, keep original estimated timing but + # ensure it doesn't overlap backwards after calibration + if i > 0: + prev_end = segments[i-1]["end"] + if seg["start"] < prev_end: + diff = prev_end - seg["start"] + seg["start"] += diff + seg["end"] += diff + + calibrated.append(seg) + + return calibrated + def _format_ass_time(self, seconds): # H:MM:SS.cs (centiseconds) td = datetime.timedelta(seconds=seconds) @@ -224,12 +300,128 @@ class AIIA_Subtitle_Preview: return {"ui": {"text": [subtitle_content], "audio": [audio_info] if audio_info else []}} +class AIIA_Subtitle_To_Segments: + """Convert SRT/ASS text or files into segments_info format.""" + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "subtitle_text": ("STRING", {"multiline": True, "default": ""}), + }, + "optional": { + "subtitle_path": ("STRING", {"default": ""}), + } + } + + RETURN_TYPES = ("STRING",) + RETURN_NAMES = ("segments_info",) + FUNCTION = "convert" + CATEGORY = "AIIA/Subtitle" + + def convert(self, subtitle_text, subtitle_path=""): + import re + content = subtitle_text.strip() + + # If path provided and exists, read it + if subtitle_path and os.path.exists(subtitle_path): + try: + with open(subtitle_path, 'r', encoding='utf-8', errors='ignore') as f: + content = f.read().strip() + except Exception as e: + print(f"[AIIA Subtitle Convert] Error reading file: {e}") + + if not content: + return (json.dumps([]),) + + segments = [] + + # Detect Format + if "Dialogue:" in content: + segments = self._parse_ass(content) + elif " --> " in content: + segments = self._parse_srt(content) + else: + print("[AIIA Subtitle Convert] Unknown format or empty content.") + + return (json.dumps(segments, ensure_ascii=False, indent=2),) + + def _parse_srt(self, text): + import re + segments = [] + # Pattern: Index, Time, Text + # Handles \n and \r\n + blocks = re.split(r'\n\s*\n', text.strip()) + for block in blocks: + lines = [l.strip() for l in block.split('\n') if l.strip()] + if len(lines) < 2: continue + + # Find time line + time_match = re.search(r'(\d+:\d+:\d+,\d+) --> (\d+:\d+:\d+,\d+)', lines[0] if "-->" in lines[0] else lines[1]) + if not time_match: continue + + start_s = self._time_to_seconds(time_match.group(1), "srt") + end_s = self._time_to_seconds(time_match.group(2), "srt") + + # Content is everything after the time line + idx = 1 if "-->" in lines[0] else 2 + content = " ".join(lines[idx:]) + + segments.append({ + "start": round(start_s, 3), + "end": round(end_s, 3), + "text": content, + "speaker": "Unknown" + }) + return segments + + def _parse_ass(self, text): + import re + segments = [] + # Look for Dialogue: lines + for line in text.split('\n'): + if line.startswith("Dialogue:"): + # Dialogue: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text + parts = line.split(',', 9) + if len(parts) < 10: continue + + start_s = self._time_to_seconds(parts[1].strip(), "ass") + end_s = self._time_to_seconds(parts[2].strip(), "ass") + speaker = parts[4].strip() or "Unknown" + content = parts[9].strip().replace('\\N', ' ').replace('\\n', ' ') + # Clean ASS tags like {\pos(x,y)} + content = re.sub(r'\{.*?\}', '', content) + + segments.append({ + "start": round(start_s, 3), + "end": round(end_s, 3), + "text": content, + "speaker": speaker + }) + return segments + + def _time_to_seconds(self, t_str, fmt): + try: + if fmt == "srt": + # HH:MM:SS,mmm + h, m, s_ms = t_str.split(':') + s, ms = s_ms.split(',') + return int(h)*3600 + int(m)*60 + int(s) + int(ms)/1000.0 + else: + # H:MM:SS.cc + h, m, s_cs = t_str.split(':') + s, cs = s_cs.split('.') + return int(h)*3600 + int(m)*60 + int(s) + int(cs)/100.0 + except: + return 0.0 + NODE_CLASS_MAPPINGS = { "AIIA_Subtitle_Gen": AIIA_Subtitle_Gen, - "AIIA_Subtitle_Preview": AIIA_Subtitle_Preview + "AIIA_Subtitle_Preview": AIIA_Subtitle_Preview, + "AIIA_Subtitle_To_Segments": AIIA_Subtitle_To_Segments } NODE_DISPLAY_NAME_MAPPINGS = { "AIIA_Subtitle_Gen": "📝 AIIA Subtitle Generation", - "AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview" + "AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview", + "AIIA_Subtitle_To_Segments": "🔄 AIIA Subtitle to Segments" } diff --git a/pyproject.toml b/pyproject.toml index 90aaa28..e6eec38 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "aiia" description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser." -version = "1.10.1" +version = "1.10.3" license = {file = "LICENSE"} readme = "README.md" authors = [