fix: total ASCII stabilization and metadata index alignment
This commit is contained in:
+23
-13
@@ -3,18 +3,18 @@ import json
|
||||
import re
|
||||
|
||||
AIIA_EMOTION_LIST = [
|
||||
"None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)",
|
||||
"温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)",
|
||||
"低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)",
|
||||
"哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)",
|
||||
"自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)"
|
||||
"None", "Happy", "Sad", "Angry", "Excited",
|
||||
"Gentle", "Serious", "Fearful", "Surprised",
|
||||
"Whispering", "Shouting", "Shy", "Seductive",
|
||||
"Crying", "Laughter", "Embarrassed", "Disappointed",
|
||||
"Proud", "Doubtful", "Anxious", "Calm"
|
||||
]
|
||||
|
||||
AIIA_DIALECT_LIST = [
|
||||
"None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)",
|
||||
"四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)",
|
||||
"客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)",
|
||||
"河南话 (Henan)", "陕西话 (Shaanxi)", "湖南话 (Hunan)"
|
||||
"None", "Mandarin", "Cantonese", "Shanghainese",
|
||||
"Sichuanese", "Northeastern", "Hokkien",
|
||||
"Hakka", "Tianjinese", "Shandongnese",
|
||||
"Henan", "Shaanxi", "Hunan"
|
||||
]
|
||||
|
||||
class AIIA_Podcast_Script_Parser:
|
||||
@@ -346,10 +346,20 @@ class AIIA_Dialogue_TTS:
|
||||
time_ptr[0] += 1.0
|
||||
|
||||
|
||||
def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode,
|
||||
max_batch_char=1000, cfg_scale=1.5, temperature=0.8, top_k=20, top_p=0.95,
|
||||
qwen_model=None, cosyvoice_model=None, vibevoice_model=None,
|
||||
qwen_base_model=None, qwen_custom_model=None, qwen_design_model=None, **kwargs):
|
||||
def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode, **kwargs):
|
||||
# Extract optional and model-specific params from kwargs
|
||||
max_batch_char = kwargs.get("max_batch_char", 1000)
|
||||
cfg_scale = kwargs.get("cfg_scale", 1.5)
|
||||
temperature = kwargs.get("temperature", 0.8)
|
||||
top_k = kwargs.get("top_k", 20)
|
||||
top_p = kwargs.get("top_p", 0.95)
|
||||
|
||||
cosyvoice_model = kwargs.get("cosyvoice_model")
|
||||
vibevoice_model = kwargs.get("vibevoice_model")
|
||||
qwen_model = kwargs.get("qwen_model")
|
||||
qwen_base_model = kwargs.get("qwen_base_model")
|
||||
qwen_custom_model = kwargs.get("qwen_custom_model")
|
||||
qwen_design_model = kwargs.get("qwen_design_model")
|
||||
# Robustness: ensure max_batch_char is correctly picked up even if shifted or provided as kwarg
|
||||
max_batch_char = kwargs.get("max_batch_char", max_batch_char)
|
||||
import json
|
||||
|
||||
+12
-21
@@ -11,31 +11,22 @@ import subprocess
|
||||
QWEN_SPEAKER_LIST = ["Vivian", "Serena", "Uncle_Fu", "Dylan", "Eric", "Ryan", "Aiden", "Ono_Anna", "Sohee"]
|
||||
QWEN_PRESET_NOTE = "Presets (9 premium timbres): Vivian/Serena/Uncle_Fu (CN), Dylan/Eric/Ryan/Aiden (EN), Ono_Anna (JP), Sohee (KR)"
|
||||
QWEN_EMOTION_LIST = [
|
||||
"None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)",
|
||||
"温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)",
|
||||
"低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)",
|
||||
"哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)",
|
||||
"自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)"
|
||||
"None", "Happy", "Sad", "Angry", "Excited",
|
||||
"Gentle", "Serious", "Fearful", "Surprised",
|
||||
"Whispering", "Shouting", "Shy", "Seductive",
|
||||
"Crying", "Laughter", "Embarrassed", "Disappointed",
|
||||
"Proud", "Doubtful", "Anxious", "Calm"
|
||||
]
|
||||
|
||||
QWEN_EXPRESSION_LIST = [
|
||||
"None",
|
||||
"带点羞涩的 (With a hint of shyness)",
|
||||
"语气充满诱惑力 (Seductive tone)",
|
||||
"语气带着哭腔 (Crying tone)",
|
||||
"稍微带一点点笑意 (With a slight smile)",
|
||||
"语气显得非常疲惫 (Sounding very tired)",
|
||||
"语速稍快,显得有些急促 (Hurried tone)",
|
||||
"充满自信且响亮的 (Confident and loud)",
|
||||
"稍微有点犹豫 and 不确定 (Hesitant and uncertain)",
|
||||
"语气极其冷淡 (Extremely cold tone)",
|
||||
"温柔且轻声细语的 (Gentle and whispering)"
|
||||
"None", "Shyness", "Seductive", "Crying", "Smiling",
|
||||
"Tired", "Hurried", "Confident", "Hesitant", "Cold", "Whispering"
|
||||
]
|
||||
|
||||
QWEN_DIALECT_LIST = [
|
||||
"None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)",
|
||||
"四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)",
|
||||
"客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)"
|
||||
"None", "Mandarin", "Cantonese", "Shanghainese",
|
||||
"Sichuanese", "Northeastern", "Hokkien",
|
||||
"Hakka", "Tianjinese", "Shandongnese"
|
||||
]
|
||||
|
||||
def _install_qwen_tts_if_needed():
|
||||
@@ -503,8 +494,8 @@ class AIIA_Qwen_Dialogue_TTS:
|
||||
design = kwargs.get(f"speaker_{spk_key}_design", "")
|
||||
# 0. Specialized Qwen Routing from Bundle or slots
|
||||
def get_model_from_bundle(mode, ref=None):
|
||||
# 1. Check primary qwen_model (might be a bundle)
|
||||
main_q = kwargs.get("qwen_model")
|
||||
# 1. Check primary qwen_model (passed as positional argument)
|
||||
main_q = qwen_model
|
||||
if main_q and main_q.get("is_bundle"):
|
||||
if mode == "Clone" or ref is not None:
|
||||
return main_q.get("base") or main_q.get("default")
|
||||
|
||||
Reference in New Issue
Block a user