feat: implement subtitle calibration and subtitle-to-segments conversion (v1.10.3)
This commit is contained in:
@@ -1018,7 +1018,8 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860
|
||||
**[v1.7.0 New]** 无需 STT,直接从生成过程中提取精准时间轴。
|
||||
|
||||
- **Input**:
|
||||
- `segments_info`: 来自 `AIIA Dialogue TTS` 的输出。
|
||||
- `segments_info`: 来自 `AIIA Dialogue TTS` 或 `AIIA Generate Segments` 的输出。
|
||||
- `calibration_info` (可选): **[v1.10.2 新增]** 接入 `AIIA Generate Speaker Segments` 的输出。用于将估算的时间轴自动“吸附”到真实的 VAD 语音活动区间,解决 VibeVoice 等批处理引擎的时间轴偏移问题。
|
||||
- **Output**:
|
||||
- `SRT`: 通用字幕格式。
|
||||
- `ASS`: 高级排版字幕格式 (自动区分角色颜色)。
|
||||
@@ -1026,7 +1027,18 @@ https://github.com/user-attachments/assets/9a5502c5-79e3-4fc8-8a2d-2cbdbdbbc860
|
||||
- **CosyVoice**: 使用生成时的精确时长。
|
||||
- **VibeVoice**: 使用**智能插值算法 (Smart Interpolation)**,根据字符长度自动计算长音频段内的单句时间轴。
|
||||
|
||||
#### 4.4 AIIA Subtitle Preview (字幕预览)
|
||||
#### 4.4 AIIA Subtitle to Segments (字幕转分段)
|
||||
|
||||
**[v1.10.3 New]** 将现有的 SRT/ASS 字幕文件转换为 `segments_info` 格式,以便进行时间轴重新校准。
|
||||
|
||||
- **Input**:
|
||||
- `subtitle_text`: SRT 或 ASS 格式的文本内容。
|
||||
- `subtitle_path` (可选): 字幕文件的本地路径(如果提供,将优先读取文件)。
|
||||
- **Output**:
|
||||
- `segments_info`: 标准化的 JSON 字符串,可直接输入到 `AIIA Subtitle Gen`。
|
||||
- **用途**: 结合 `Subtitle Gen` 的 `calibration_info` 输入,可以将**旧的、不准的字幕**自动对齐到**新的、精准的音轨**上。
|
||||
|
||||
#### 4.5 AIIA Subtitle Preview (字幕预览)
|
||||
|
||||
**[v1.7.1 New]** 实时校验音画同步效果。
|
||||
|
||||
|
||||
+26
-2
@@ -748,7 +748,30 @@ class AIIA_DittoSampler:
|
||||
current_alpha = max(target, current_alpha - release_step)
|
||||
|
||||
dataset_alpha[i] = current_alpha
|
||||
|
||||
|
||||
# [v1.10.0] Independent Head Pitch Envelope
|
||||
# We want the head to nod/tilt SLOWLY when speech starts (0.8s),
|
||||
# while the mouth opens INSTANTLY (0.08s).
|
||||
# So we calculate a second alpha specifically for the head pitch offset.
|
||||
|
||||
head_pitch_alpha = np.zeros(num_frames, dtype=np.float32)
|
||||
current_head_alpha = target_alpha[0]
|
||||
# [v1.10.0 Tuned] Faster attack to catch initial head lift.
|
||||
# 0.05 (0.8s) was too slow -> Head lifted before correction.
|
||||
# 0.20 (0.2s) is balanced -> Fast enough to clamp lift, smooth enough to avoid snap.
|
||||
head_attack_step = 0.20
|
||||
# Use same release step as mouth to return to neutral naturally
|
||||
|
||||
for i in range(num_frames):
|
||||
target = target_alpha[i]
|
||||
if target > current_head_alpha:
|
||||
# Slow Attack
|
||||
current_head_alpha = min(target, current_head_alpha + head_attack_step)
|
||||
else:
|
||||
# Same Release
|
||||
current_head_alpha = max(target, current_head_alpha - release_step)
|
||||
head_pitch_alpha[i] = current_head_alpha
|
||||
|
||||
# Log VAD stats for debugging
|
||||
non_silence_count = np.count_nonzero(target_alpha)
|
||||
logging.info(f"[Ditto] VAD Stats: {non_silence_count}/{num_frames} frames active. RMS Mean: {np.mean(rms):.4f}, Min: {np.min(rms):.4f}, Max: {np.max(rms):.4f}")
|
||||
@@ -830,7 +853,8 @@ class AIIA_DittoSampler:
|
||||
# Allows correcting "head too high/low" during speech.
|
||||
# Smoothly fades in/out based on VAD alpha (mouth opening).
|
||||
# Positive = Look Down, Negative = Look Up
|
||||
speech_pitch_offset = speech_pitch * alpha
|
||||
# [v1.10.0] Use smoothed alpha for head to avoid "snap" motion.
|
||||
speech_pitch_offset = speech_pitch * head_pitch_alpha[i]
|
||||
|
||||
# Apply to dict
|
||||
info_dict["delta_pitch"] = hd_rot_p + d_pitch + speech_pitch_offset
|
||||
|
||||
+195
-3
@@ -18,6 +18,7 @@ class AIIA_Subtitle_Gen:
|
||||
"save_file": ("BOOLEAN", {"default": False, "label_on": "Save to Disk", "label_off": "Memory Only"}),
|
||||
},
|
||||
"optional": {
|
||||
"calibration_info": ("WHISPER_CHUNKS",),
|
||||
"ass_style": ("STRING", {"default": "Default", "multiline": False}),
|
||||
"filename_prefix": ("STRING", {"default": "aiia_subtitle"}),
|
||||
}
|
||||
@@ -29,7 +30,7 @@ class AIIA_Subtitle_Gen:
|
||||
CATEGORY = "AIIA/Subtitle"
|
||||
OUTPUT_NODE = True
|
||||
|
||||
def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle"):
|
||||
def generate_subtitle(self, segments_info, format="SRT", save_file=False, ass_style="Default", filename_prefix="aiia_subtitle", calibration_info=None):
|
||||
try:
|
||||
segments = json.loads(segments_info)
|
||||
except Exception as e:
|
||||
@@ -40,6 +41,15 @@ class AIIA_Subtitle_Gen:
|
||||
print("[AIIA Subtitle] Segments info must be a list of dicts.")
|
||||
return ("", "")
|
||||
|
||||
# --- Subtitle Calibration (v1.10.2) ---
|
||||
if calibration_info and "chunks" in calibration_info:
|
||||
print(f"[AIIA Subtitle] Calibrating {len(segments)} segments using {len(calibration_info['chunks'])} high-precision chunks.")
|
||||
segments = self._calibrate_segments(segments, calibration_info["chunks"])
|
||||
|
||||
if not isinstance(segments, list):
|
||||
print("[AIIA Subtitle] Segments info must be a list of dicts.")
|
||||
return ("", "")
|
||||
|
||||
srt_out = ""
|
||||
ass_out = ""
|
||||
|
||||
@@ -169,6 +179,72 @@ class AIIA_Subtitle_Gen:
|
||||
|
||||
return f"{hours:02}:{minutes:02}:{secs:02},{millis:03}"
|
||||
|
||||
def _calibrate_segments(self, segments, chunks):
|
||||
"""
|
||||
Calibrate estimated segments using high-precision VAD chunks.
|
||||
Algorithm: Iterative sequence matching with overlap weight.
|
||||
"""
|
||||
calibrated = []
|
||||
chunk_idx = 0
|
||||
num_chunks = len(chunks)
|
||||
|
||||
for i, seg in enumerate(segments):
|
||||
seg_start = seg["start"]
|
||||
seg_end = seg["end"]
|
||||
|
||||
best_match_start = -1
|
||||
best_match_end = -1
|
||||
|
||||
# Find chunks that overlap with this segment
|
||||
# We look ahead starting from chunk_idx to maintain sequential order
|
||||
matched_chunks = []
|
||||
|
||||
# Tolerance window: How far we can look for a matching chunk if no direct overlap
|
||||
# 1.0s is reasonable for VibeVoice drift
|
||||
lookahead_limit = 5
|
||||
|
||||
find_idx = chunk_idx
|
||||
while find_idx < num_chunks and len(matched_chunks) < lookahead_limit:
|
||||
chunk = chunks[find_idx]
|
||||
c_start, c_end = chunk["timestamp"]
|
||||
|
||||
# Check Overlap
|
||||
overlap = min(seg_end, c_end) - max(seg_start, c_start)
|
||||
|
||||
# If significant overlap, or if it's the very first chunk and we are near start
|
||||
if overlap > 0.05 or (i == 0 and find_idx == 0 and abs(c_start - seg_start) < 2.0):
|
||||
matched_chunks.append(find_idx)
|
||||
|
||||
# Break if we've passed the segment significantly
|
||||
if c_start > seg_end + 1.0:
|
||||
break
|
||||
find_idx += 1
|
||||
|
||||
if matched_chunks:
|
||||
# Use the range of all matched chunks
|
||||
# This handles cases where one sentence is split into multiple VAD chunks due to pauses
|
||||
min_s = chunks[matched_chunks[0]]["timestamp"][0]
|
||||
max_e = chunks[matched_chunks[-1]]["timestamp"][1]
|
||||
|
||||
# Update chunk_idx to favor the next chunk for subsequent segments
|
||||
chunk_idx = matched_chunks[-1] + 1
|
||||
|
||||
seg["start"] = round(min_s, 3)
|
||||
seg["end"] = round(max_e, 3)
|
||||
else:
|
||||
# No match found within window, keep original estimated timing but
|
||||
# ensure it doesn't overlap backwards after calibration
|
||||
if i > 0:
|
||||
prev_end = segments[i-1]["end"]
|
||||
if seg["start"] < prev_end:
|
||||
diff = prev_end - seg["start"]
|
||||
seg["start"] += diff
|
||||
seg["end"] += diff
|
||||
|
||||
calibrated.append(seg)
|
||||
|
||||
return calibrated
|
||||
|
||||
def _format_ass_time(self, seconds):
|
||||
# H:MM:SS.cs (centiseconds)
|
||||
td = datetime.timedelta(seconds=seconds)
|
||||
@@ -224,12 +300,128 @@ class AIIA_Subtitle_Preview:
|
||||
|
||||
return {"ui": {"text": [subtitle_content], "audio": [audio_info] if audio_info else []}}
|
||||
|
||||
class AIIA_Subtitle_To_Segments:
|
||||
"""Convert SRT/ASS text or files into segments_info format."""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"subtitle_text": ("STRING", {"multiline": True, "default": ""}),
|
||||
},
|
||||
"optional": {
|
||||
"subtitle_path": ("STRING", {"default": ""}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("STRING",)
|
||||
RETURN_NAMES = ("segments_info",)
|
||||
FUNCTION = "convert"
|
||||
CATEGORY = "AIIA/Subtitle"
|
||||
|
||||
def convert(self, subtitle_text, subtitle_path=""):
|
||||
import re
|
||||
content = subtitle_text.strip()
|
||||
|
||||
# If path provided and exists, read it
|
||||
if subtitle_path and os.path.exists(subtitle_path):
|
||||
try:
|
||||
with open(subtitle_path, 'r', encoding='utf-8', errors='ignore') as f:
|
||||
content = f.read().strip()
|
||||
except Exception as e:
|
||||
print(f"[AIIA Subtitle Convert] Error reading file: {e}")
|
||||
|
||||
if not content:
|
||||
return (json.dumps([]),)
|
||||
|
||||
segments = []
|
||||
|
||||
# Detect Format
|
||||
if "Dialogue:" in content:
|
||||
segments = self._parse_ass(content)
|
||||
elif " --> " in content:
|
||||
segments = self._parse_srt(content)
|
||||
else:
|
||||
print("[AIIA Subtitle Convert] Unknown format or empty content.")
|
||||
|
||||
return (json.dumps(segments, ensure_ascii=False, indent=2),)
|
||||
|
||||
def _parse_srt(self, text):
|
||||
import re
|
||||
segments = []
|
||||
# Pattern: Index, Time, Text
|
||||
# Handles \n and \r\n
|
||||
blocks = re.split(r'\n\s*\n', text.strip())
|
||||
for block in blocks:
|
||||
lines = [l.strip() for l in block.split('\n') if l.strip()]
|
||||
if len(lines) < 2: continue
|
||||
|
||||
# Find time line
|
||||
time_match = re.search(r'(\d+:\d+:\d+,\d+) --> (\d+:\d+:\d+,\d+)', lines[0] if "-->" in lines[0] else lines[1])
|
||||
if not time_match: continue
|
||||
|
||||
start_s = self._time_to_seconds(time_match.group(1), "srt")
|
||||
end_s = self._time_to_seconds(time_match.group(2), "srt")
|
||||
|
||||
# Content is everything after the time line
|
||||
idx = 1 if "-->" in lines[0] else 2
|
||||
content = " ".join(lines[idx:])
|
||||
|
||||
segments.append({
|
||||
"start": round(start_s, 3),
|
||||
"end": round(end_s, 3),
|
||||
"text": content,
|
||||
"speaker": "Unknown"
|
||||
})
|
||||
return segments
|
||||
|
||||
def _parse_ass(self, text):
|
||||
import re
|
||||
segments = []
|
||||
# Look for Dialogue: lines
|
||||
for line in text.split('\n'):
|
||||
if line.startswith("Dialogue:"):
|
||||
# Dialogue: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
|
||||
parts = line.split(',', 9)
|
||||
if len(parts) < 10: continue
|
||||
|
||||
start_s = self._time_to_seconds(parts[1].strip(), "ass")
|
||||
end_s = self._time_to_seconds(parts[2].strip(), "ass")
|
||||
speaker = parts[4].strip() or "Unknown"
|
||||
content = parts[9].strip().replace('\\N', ' ').replace('\\n', ' ')
|
||||
# Clean ASS tags like {\pos(x,y)}
|
||||
content = re.sub(r'\{.*?\}', '', content)
|
||||
|
||||
segments.append({
|
||||
"start": round(start_s, 3),
|
||||
"end": round(end_s, 3),
|
||||
"text": content,
|
||||
"speaker": speaker
|
||||
})
|
||||
return segments
|
||||
|
||||
def _time_to_seconds(self, t_str, fmt):
|
||||
try:
|
||||
if fmt == "srt":
|
||||
# HH:MM:SS,mmm
|
||||
h, m, s_ms = t_str.split(':')
|
||||
s, ms = s_ms.split(',')
|
||||
return int(h)*3600 + int(m)*60 + int(s) + int(ms)/1000.0
|
||||
else:
|
||||
# H:MM:SS.cc
|
||||
h, m, s_cs = t_str.split(':')
|
||||
s, cs = s_cs.split('.')
|
||||
return int(h)*3600 + int(m)*60 + int(s) + int(cs)/100.0
|
||||
except:
|
||||
return 0.0
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"AIIA_Subtitle_Gen": AIIA_Subtitle_Gen,
|
||||
"AIIA_Subtitle_Preview": AIIA_Subtitle_Preview
|
||||
"AIIA_Subtitle_Preview": AIIA_Subtitle_Preview,
|
||||
"AIIA_Subtitle_To_Segments": AIIA_Subtitle_To_Segments
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"AIIA_Subtitle_Gen": "📝 AIIA Subtitle Generation",
|
||||
"AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview"
|
||||
"AIIA_Subtitle_Preview": "🎬 AIIA Subtitle Preview",
|
||||
"AIIA_Subtitle_To_Segments": "🔄 AIIA Subtitle to Segments"
|
||||
}
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "aiia"
|
||||
description = "The Ultimate AI Audio/Video toolkit for ComfyUI. Features an enhanced Ditto (with optimizations that outperform official demos and other SOTA talking head models in lip-sync accuracy and natural motion), EchoMimic V3 & FLOAT, VibeVoice & CosyVoice 3.0 (Zero-Shot Voice Cloning), Multi-Role Podcast Generation, and a powerful Media Browser."
|
||||
version = "1.10.1"
|
||||
version = "1.10.3"
|
||||
license = {file = "LICENSE"}
|
||||
readme = "README.md"
|
||||
authors = [
|
||||
|
||||
Reference in New Issue
Block a user