fix: total ASCII stabilization and metadata index alignment

This commit is contained in:
Hawk Lee
2026-02-04 16:09:21 +08:00
parent 4dd27695fe
commit 6c7e3cd469
2 changed files with 35 additions and 34 deletions
+23 -13
View File
@@ -3,18 +3,18 @@ import json
import re
AIIA_EMOTION_LIST = [
"None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)",
"温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)",
"低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)",
"哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)",
"自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)"
"None", "Happy", "Sad", "Angry", "Excited",
"Gentle", "Serious", "Fearful", "Surprised",
"Whispering", "Shouting", "Shy", "Seductive",
"Crying", "Laughter", "Embarrassed", "Disappointed",
"Proud", "Doubtful", "Anxious", "Calm"
]
AIIA_DIALECT_LIST = [
"None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)",
"四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)",
"客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)",
"河南话 (Henan)", "陕西话 (Shaanxi)", "湖南话 (Hunan)"
"None", "Mandarin", "Cantonese", "Shanghainese",
"Sichuanese", "Northeastern", "Hokkien",
"Hakka", "Tianjinese", "Shandongnese",
"Henan", "Shaanxi", "Hunan"
]
class AIIA_Podcast_Script_Parser:
@@ -346,10 +346,20 @@ class AIIA_Dialogue_TTS:
time_ptr[0] += 1.0
def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode,
max_batch_char=1000, cfg_scale=1.5, temperature=0.8, top_k=20, top_p=0.95,
qwen_model=None, cosyvoice_model=None, vibevoice_model=None,
qwen_base_model=None, qwen_custom_model=None, qwen_design_model=None, **kwargs):
def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode, **kwargs):
# Extract optional and model-specific params from kwargs
max_batch_char = kwargs.get("max_batch_char", 1000)
cfg_scale = kwargs.get("cfg_scale", 1.5)
temperature = kwargs.get("temperature", 0.8)
top_k = kwargs.get("top_k", 20)
top_p = kwargs.get("top_p", 0.95)
cosyvoice_model = kwargs.get("cosyvoice_model")
vibevoice_model = kwargs.get("vibevoice_model")
qwen_model = kwargs.get("qwen_model")
qwen_base_model = kwargs.get("qwen_base_model")
qwen_custom_model = kwargs.get("qwen_custom_model")
qwen_design_model = kwargs.get("qwen_design_model")
# Robustness: ensure max_batch_char is correctly picked up even if shifted or provided as kwarg
max_batch_char = kwargs.get("max_batch_char", max_batch_char)
import json
+12 -21
View File
@@ -11,31 +11,22 @@ import subprocess
QWEN_SPEAKER_LIST = ["Vivian", "Serena", "Uncle_Fu", "Dylan", "Eric", "Ryan", "Aiden", "Ono_Anna", "Sohee"]
QWEN_PRESET_NOTE = "Presets (9 premium timbres): Vivian/Serena/Uncle_Fu (CN), Dylan/Eric/Ryan/Aiden (EN), Ono_Anna (JP), Sohee (KR)"
QWEN_EMOTION_LIST = [
"None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)",
"温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)",
"低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)",
"哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)",
"自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)"
"None", "Happy", "Sad", "Angry", "Excited",
"Gentle", "Serious", "Fearful", "Surprised",
"Whispering", "Shouting", "Shy", "Seductive",
"Crying", "Laughter", "Embarrassed", "Disappointed",
"Proud", "Doubtful", "Anxious", "Calm"
]
QWEN_EXPRESSION_LIST = [
"None",
"带点羞涩的 (With a hint of shyness)",
"语气充满诱惑力 (Seductive tone)",
"语气带着哭腔 (Crying tone)",
"稍微带一点点笑意 (With a slight smile)",
"语气显得非常疲惫 (Sounding very tired)",
"语速稍快,显得有些急促 (Hurried tone)",
"充满自信且响亮的 (Confident and loud)",
"稍微有点犹豫 and 不确定 (Hesitant and uncertain)",
"语气极其冷淡 (Extremely cold tone)",
"温柔且轻声细语的 (Gentle and whispering)"
"None", "Shyness", "Seductive", "Crying", "Smiling",
"Tired", "Hurried", "Confident", "Hesitant", "Cold", "Whispering"
]
QWEN_DIALECT_LIST = [
"None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)",
"四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)",
"客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)"
"None", "Mandarin", "Cantonese", "Shanghainese",
"Sichuanese", "Northeastern", "Hokkien",
"Hakka", "Tianjinese", "Shandongnese"
]
def _install_qwen_tts_if_needed():
@@ -503,8 +494,8 @@ class AIIA_Qwen_Dialogue_TTS:
design = kwargs.get(f"speaker_{spk_key}_design", "")
# 0. Specialized Qwen Routing from Bundle or slots
def get_model_from_bundle(mode, ref=None):
# 1. Check primary qwen_model (might be a bundle)
main_q = kwargs.get("qwen_model")
# 1. Check primary qwen_model (passed as positional argument)
main_q = qwen_model
if main_q and main_q.get("is_bundle"):
if mode == "Clone" or ref is not None:
return main_q.get("base") or main_q.get("default")