feat: implement subtitle calibration and subtitle-to-segments conversion (v1.10.3)

This commit is contained in:
Hawk Lee
2026-01-29 13:47:52 +08:00
parent fe8a8dda7b
commit e968fc688b
4 changed files with 236 additions and 8 deletions
+14 -2
View File
@@ -1018,7 +1018,8 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860
**[v1.7.0 New]** 无需 STT,直接从生成过程中提取精准时间轴。
- **Input**:
- `segments_info`: 来自 `AIIA Dialogue TTS` 的输出。
- `segments_info`: 来自 `AIIA Dialogue TTS` 或 `AIIA Generate Segments` 的输出。
- `calibration_info` (可选): **[v1.10.2 新增]** 接入 `AIIA Generate Speaker Segments` 的输出。用于将估算的时间轴自动“吸附”到真实的 VAD 语音活动区间,解决 VibeVoice 等批处理引擎的时间轴偏移问题。
- **Output**:
- `SRT`: 通用字幕格式。
- `ASS`: 高级排版字幕格式 (自动区分角色颜色)。
@@ -1026,7 +1027,18 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860
- **CosyVoice**: 使用生成时的精确时长。
- **VibeVoice**: 使用**智能插值算法 (Smart Interpolation)**,根据字符长度自动计算长音频段内的单句时间轴。
#### 4.4 AIIA Subtitle Preview (字幕预览)
#### 4.4 AIIA Subtitle to Segments (字幕转分段)
**[v1.10.3 New]** 将现有的 SRT/ASS 字幕文件转换为 `segments_info` 格式,以便进行时间轴重新校准。
- **Input**:
- `subtitle_text`: SRT 或 ASS 格式的文本内容。
- `subtitle_path` (可选): 字幕文件的本地路径(如果提供,将优先读取文件)。
- **Output**:
- `segments_info`: 标准化的 JSON 字符串,可直接输入到 `AIIA Subtitle Gen`。
- **用途**: 结合 `Subtitle Gen` 的 `calibration_info` 输入,可以将**旧的、不准的字幕**自动对齐到**新的、精准的音轨**上。
#### 4.5 AIIA Subtitle Preview (字幕预览)
**[v1.7.1 New]** 实时校验音画同步效果。
+26 -2
View File
@@ -748,7 +748,30 @@ class AIIA_DittoSampler:
current_alpha = max(target, current_alpha - release_step)
dataset_alpha[i] = current_alpha
# [v1.10.0] Independent Head Pitch Envelope
# We want the head to nod/tilt SLOWLY when speech starts (0.8s),
# while the mouth opens INSTANTLY (0.08s).
# So we calculate a second alpha specifically for the head pitch offset.
head_pitch_alpha = np.zeros(num_frames, dtype=np.float32)
current_head_alpha = target_alpha[0]
# [v1.10.0 Tuned] Faster attack to catch initial head lift.
# 0.05 (0.8s) was too slow -> Head lifted before correction.
# 0.20 (0.2s) is balanced -> Fast enough to clamp lift, smooth enough to avoid snap.
head_attack_step = 0.20
# Use same release step as mouth to return to neutral naturally
for i in range(num_frames):
target = target_alpha[i]
if target > current_head_alpha:
# Slow Attack
current_head_alpha = min(target, current_head_alpha + head_attack_step)
else:
# Same Release
current_head_alpha = max(target, current_head_alpha - release_step)
head_pitch_alpha[i] = current_head_alpha
# Log VAD stats for debugging
non_silence_count = np.count_nonzero(target_alpha)
logging.info(f"[Ditto] VAD Stats: {non_silence_count}/{num_frames} frames active. RMS Mean: {np.mean(rms):.4f}, Min: {np.min(rms):.4f}, Max: {np.max(rms):.4f}")
@@ -830,7 +853,8 @@ class AIIA_DittoSampler:
# Allows correcting "head too high/low" during speech.
# Smoothly fades in/out based on VAD alpha (mouth opening).
# Positive = Look Down, Negative = Look Up
speech_pitch_offset = speech_pitch * alpha
# [v1.10.0] Use smoothed alpha for head to avoid "snap" motion.
speech_pitch_offset = speech_pitch * head_pitch_alpha[i]
# Apply to dict
info_dict["delta_pitch"] = hd_rot_p + d_pitch + speech_pitch_offset
+195 -3
View File
@@ -18,6 +18,7 @@ class AIIA_Subtitle_Gen:
"save_file": ("BOOLEAN", {"default": False, "label_on": "Save to Disk", "label_off": "Memory Only"}),
},
"optional": {
"calibration_info": ("WHISPER_CHUNKS",),
"ass_style": ("STRING", {"default": "Default", "multiline": False}),
"filename_prefix": ("STRING", {"default": "aiia_subtitle"}),
}
@@ -29,7 +30,7 @@ class AIIA_Subtitle_Gen:
CATEGORY = "AIIA/Subtitle"
OUTPUT_NODE = True
def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle"):
def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle", calibration_info=None):
try:
segments = json.loads(segments_info)
except Exception as e:
@@ -40,6 +41,15 @@ class AIIA_Subtitle_Gen:
print("[AIIA Subtitle] Segments info must be a list of dicts.")
return ("", "")
# --- Subtitle Calibration (v1.10.2) ---
if calibration_info and "chunks" in calibration_info:
print(f"[AIIA Subtitle] Calibrating {len(segments)} segments using {len(calibration_info['chunks'])} high-precision chunks.")
segments = self._calibrate_segments(segments, calibration_info["chunks"])
if not isinstance(segments, list):
print("[AIIA Subtitle] Segments info must be a list of dicts.")
return ("", "")
srt_out = ""
ass_out = ""
@@ -169,6 +179,72 @@ class AIIA_Subtitle_Gen:
return f"{hours:02}:{minutes:02}:{secs:02},{millis:03}"
def _calibrate_segments(self, segments, chunks):
"""
Calibrate estimated segments using high-precision VAD chunks.
Algorithm: Iterative sequence matching with overlap weight.
"""
calibrated = []
chunk_idx = 0
num_chunks = len(chunks)
for i, seg in enumerate(segments):
seg_start = seg["start"]
seg_end = seg["end"]
best_match_start = -1
best_match_end = -1
# Find chunks that overlap with this segment
# We look ahead starting from chunk_idx to maintain sequential order
matched_chunks = []
# Tolerance window: How far we can look for a matching chunk if no direct overlap
# 1.0s is reasonable for VibeVoice drift
lookahead_limit = 5
find_idx = chunk_idx
while find_idx < num_chunks and len(matched_chunks) < lookahead_limit:
chunk = chunks[find_idx]
c_start, c_end = chunk["timestamp"]
# Check Overlap
overlap = min(seg_end, c_end) - max(seg_start, c_start)
# If significant overlap, or if it's the very first chunk and we are near start
if overlap > 0.05 or (i == 0 and find_idx == 0 and abs(c_start - seg_start) < 2.0):
matched_chunks.append(find_idx)
# Break if we've passed the segment significantly
if c_start > seg_end + 1.0:
break
find_idx += 1
if matched_chunks:
# Use the range of all matched chunks
# This handles cases where one sentence is split into multiple VAD chunks due to pauses
min_s = chunks[matched_chunks[0]]["timestamp"][0]
max_e = chunks[matched_chunks[-1]]["timestamp"][1]
# Update chunk_idx to favor the next chunk for subsequent segments
chunk_idx = matched_chunks[-1] + 1
seg["start"] = round(min_s, 3)
seg["end"] = round(max_e, 3)
else:
# No match found within window, keep original estimated timing but
# ensure it doesn't overlap backwards after calibration
if i > 0:
prev_end = segments[i-1]["end"]
if seg["start"] < prev_end:
diff = prev_end - seg["start"]
seg["start"] += diff
seg["end"] += diff
calibrated.append(seg)
return calibrated
def _format_ass_time(self, seconds):
# H:MM:SS.cs (centiseconds)
td = datetime.timedelta(seconds=seconds)
@@ -224,12 +300,128 @@ class AIIA_Subtitle_Preview:
return {"ui": {"text": [subtitle_content], "audio": [audio_info] if audio_info else []}}
class AIIA_Subtitle_To_Segments:
"""Convert SRT/ASS text or files into segments_info format."""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"subtitle_text": ("STRING", {"multiline": True, "default": ""}),
},
"optional": {
"subtitle_path": ("STRING", {"default": ""}),
}
}
RETURN_TYPES = ("STRING",)
RETURN_NAMES = ("segments_info",)
FUNCTION = "convert"
CATEGORY = "AIIA/Subtitle"
def convert(self, subtitle_text, subtitle_path=""):
import re
content = subtitle_text.strip()
# If path provided and exists, read it
if subtitle_path and os.path.exists(subtitle_path):
try:
with open(subtitle_path, 'r', encoding='utf-8', errors='ignore') as f:
content = f.read().strip()
except Exception as e:
print(f"[AIIA Subtitle Convert] Error reading file: {e}")
if not content:
return (json.dumps([]),)
segments = []
# Detect Format
if "Dialogue:" in content:
segments = self._parse_ass(content)
elif " --> " in content:
segments = self._parse_srt(content)
else:
print("[AIIA Subtitle Convert] Unknown format or empty content.")
return (json.dumps(segments, ensure_ascii=False, indent=2),)
def _parse_srt(self, text):
import re
segments = []
# Pattern: Index, Time, Text
# Handles \n and \r\n
blocks = re.split(r'\n\s*\n', text.strip())
for block in blocks:
lines = [l.strip() for l in block.split('\n') if l.strip()]
if len(lines) < 2: continue
# Find time line
time_match = re.search(r'(\d+:\d+:\d+,\d+) --> (\d+:\d+:\d+,\d+)', lines[0] if "-->" in lines[0] else lines[1])
if not time_match: continue
start_s = self._time_to_seconds(time_match.group(1), "srt")
end_s = self._time_to_seconds(time_match.group(2), "srt")
# Content is everything after the time line
idx = 1 if "-->" in lines[0] else 2
content = " ".join(lines[idx:])
segments.append({
"start": round(start_s, 3),
"end": round(end_s, 3),
"text": content,
"speaker": "Unknown"
})
return segments
def _parse_ass(self, text):
import re
segments = []
# Look for Dialogue: lines
for line in text.split('\n'):
if line.startswith("Dialogue:"):
# Dialogue: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
parts = line.split(',', 9)
if len(parts) < 10: continue
start_s = self._time_to_seconds(parts[1].strip(), "ass")
end_s = self._time_to_seconds(parts[2].strip(), "ass")
speaker = parts[4].strip() or "Unknown"
content = parts[9].strip().replace('\\N', ' ').replace('\\n', ' ')
# Clean ASS tags like {\pos(x,y)}
content = re.sub(r'\{.*?\}', '', content)
segments.append({
"start": round(start_s, 3),
"end": round(end_s, 3),
"text": content,
"speaker": speaker
})
return segments
def _time_to_seconds(self, t_str, fmt):
try:
if fmt == "srt":
# HH:MM:SS,mmm
h, m, s_ms = t_str.split(':')
s, ms = s_ms.split(',')
return int(h)*3600 + int(m)*60 + int(s) + int(ms)/1000.0
else:
# H:MM:SS.cc
h, m, s_cs = t_str.split(':')
s, cs = s_cs.split('.')
return int(h)*3600 + int(m)*60 + int(s) + int(cs)/100.0
except:
return 0.0
NODE_CLASS_MAPPINGS = {
"AIIA_Subtitle_Gen": AIIA_Subtitle_Gen,
"AIIA_Subtitle_Preview": AIIA_Subtitle_Preview
"AIIA_Subtitle_Preview": AIIA_Subtitle_Preview,
"AIIA_Subtitle_To_Segments": AIIA_Subtitle_To_Segments
}
NODE_DISPLAY_NAME_MAPPINGS = {
"AIIA_Subtitle_Gen": "📝 AIIA Subtitle Generation",
"AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview"
"AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview",
"AIIA_Subtitle_To_Segments": "🔄 AIIA Subtitle to Segments"
}
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "aiia"
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
version = "1.10.1"
version = "1.10.3"
license = {file = "LICENSE"}
readme = "README.md"
authors = [