diff --git a/TODO.md b/TODO.md index c2ee378..80e60d2 100644 --- a/TODO.md +++ b/TODO.md @@ -3,3 +3,8 @@ ## Podcast / TTS 增强 - [ ] **LLM 驱动的情感标注**:使用轻量 LLM 分析输入对白文本,自动为每句添加情感控制标签(如 `excited`、`calm`、`serious`),用于控制 TTS 模型的情感表达。适用于支持情感标签的模型(CosyVoice v2/v3、F5-TTS 等),配合 Split/Stitch 流程实现逐句情感分配。 + +## Stitch / 切分精度 + +- [ ] **Silero VAD 集成**:用极轻量 VAD 模型(~2MB ONNX)替代纯能量检测,作为切分边界的最终裁决者。ASR 负责粗定位,VAD 精确修正语音起止点,彻底解决尾音误切、清辅音误判等问题。`pip install silero-vad` 或 `torch.hub.load('snakers4/silero-vad', 'silero_vad')`。 +- [ ] **Forced Alignment(强制对齐)**:用 Wav2Vec2 基础的对齐模型(如 MFA / mms-fa)替代 ASR 做时间对齐。输入音频+讲稿,输出每个字/音素的毫秒级位置,彻底避免 ASR 错字/漏字的匹配漂移。 diff --git a/aiia_podcast_stitcher.py b/aiia_podcast_stitcher.py index 2ac9f61..3586dda 100644 --- a/aiia_podcast_stitcher.py +++ b/aiia_podcast_stitcher.py @@ -312,12 +312,13 @@ class AIIA_Podcast_Stitcher: def _refine_cut_point(self, wav, sr, time_s, search_radius=0.15, direction="both"): """ - 在 time_s 附近找到能量最低的点作为切割位置。 + 在 time_s 附近找到能量最低的"静音山谷"作为切割位置。 + 使用 20ms 窗口 + 滑动平均平滑,避免被瞬时低能量(清辅音等)欺骗。 direction: - "before" — 只在 [time_s - radius, time_s] 搜索(用于 cut_start,远离语音) - "after" — 只在 [time_s, time_s + radius] 搜索(用于 cut_end,远离语音) - "both" — 在 ±radius 搜索(向后兼容) + "before" — 只在 [time_s - radius, time_s] 搜索(用于 cut_start) + "after" — 只在 [time_s, time_s + radius] 搜索 + "both" — 在 ±radius 搜索(用于 cut_end,寻找最近的静音谷) """ center = int(time_s * sr) radius = int(search_radius * sr) @@ -332,30 +333,43 @@ class AIIA_Podcast_Stitcher: start = max(0, center - radius) end = min(len(wav), center + radius) - if end - start < 2: + # 搜索区间太短(<50ms)则不微调 + if end - start < int(sr * 0.05): return time_s - # 计算短时能量(10ms 窗口) - window_size = max(1, int(0.01 * sr)) segment = wav[start:end] - n_windows = len(segment) // window_size + + # 20ms 窗口,10ms 步长(跨越大部分短暂闭气停顿) + window_size = max(1, int(0.02 * sr)) + step_size = max(1, int(0.01 * sr)) + + n_windows = (len(segment) - window_size) // step_size + 1 if n_windows < 2: return time_s energies = [] + positions = [] for i in range(n_windows): - w = segment[i * window_size:(i + 1) * window_size] + w = segment[i * step_size : i * step_size + window_size] energies.append(np.sqrt(np.mean(w ** 2))) + positions.append(start + i * step_size + window_size // 2) - # 找能量最低的窗口 - min_idx = np.argmin(energies) - best_sample = start + min_idx * window_size + window_size // 2 - return best_sample / sr + # 5-point 滑动平均平滑(把锯齿状毛刺抹平,寻找宽阔的静音带) + kernel_size = min(5, len(energies)) + kernel = np.ones(kernel_size) / kernel_size + smoothed = np.convolve(energies, kernel, mode='same') + + if len(smoothed) == 0: + return time_s + + # 找平滑后的能量最低点 + min_idx = np.argmin(smoothed) + return positions[min_idx] / sr def _expand_to_midpoints(self, boundaries: list, total_duration: float) -> list: """将切割点扩展到相邻句子间隙中,但限制最大扩展量以避免吃进下一句。""" MAX_EXPAND_START = 0.15 # cut_start 向前扩展:最多 150ms(保留吸气/起音余量) - MAX_EXPAND_END = 0.05 # cut_end 向后扩展:最多 50ms(保守,避免吃到下一句) + MAX_EXPAND_END = 0.10 # cut_end 向后扩展:最多 100ms(补偿 ASR 尾部时间戳早退) if len(boundaries) <= 1: if boundaries: @@ -479,13 +493,13 @@ class AIIA_Podcast_Stitcher: cut_start = boundary.get("cut_start", boundary["start"]) cut_end = boundary.get("cut_end", boundary["end"]) - # 基于能量的边界微调:找到语音实际结束/开始位置 - cut_start = self._refine_cut_point(wav, sr, cut_start, direction="before") - cut_end = self._refine_cut_point(wav, sr, cut_end, direction="before") + # 基于能量的边界微调(平滑能量包络 + 方向性搜索) + cut_start = self._refine_cut_point(wav, sr, cut_start, search_radius=0.15, direction="before") + cut_end = self._refine_cut_point(wav, sr, cut_end, search_radius=0.10, direction="both") - # 应用 padding(仅对 cut_start,给吸气/起音留余量;cut_end 不加 padding 避免吃下一句) + # 应用 padding:cut_start 全额保留起音余量,cut_end 少量保留尾音衰减 cut_start = max(0, cut_start - padding) - cut_end = min(len(wav) / sr, cut_end) + cut_end = min(len(wav) / sr, cut_end + padding * 0.3) # 防重叠:确保 cut_start 不早于同一说话人上一个片段的 cut_end if cut_start < prev_cut_end[speaker]: