diff --git a/aiia_podcast_nodes.py b/aiia_podcast_nodes.py index b9db03e..0f51d66 100755 --- a/aiia_podcast_nodes.py +++ b/aiia_podcast_nodes.py @@ -3,18 +3,18 @@ import json import re AIIA_EMOTION_LIST = [ - "None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)", - "温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)", - "低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)", - "哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)", - "自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)" + "None", "Happy", "Sad", "Angry", "Excited", + "Gentle", "Serious", "Fearful", "Surprised", + "Whispering", "Shouting", "Shy", "Seductive", + "Crying", "Laughter", "Embarrassed", "Disappointed", + "Proud", "Doubtful", "Anxious", "Calm" ] AIIA_DIALECT_LIST = [ - "None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)", - "四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)", - "客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)", - "河南话 (Henan)", "陕西话 (Shaanxi)", "湖南话 (Hunan)" + "None", "Mandarin", "Cantonese", "Shanghainese", + "Sichuanese", "Northeastern", "Hokkien", + "Hakka", "Tianjinese", "Shandongnese", + "Henan", "Shaanxi", "Hunan" ] class AIIA_Podcast_Script_Parser: @@ -346,10 +346,20 @@ class AIIA_Dialogue_TTS: time_ptr[0] += 1.0 - def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode, - max_batch_char=1000, cfg_scale=1.5, temperature=0.8, top_k=20, top_p=0.95, - qwen_model=None, cosyvoice_model=None, vibevoice_model=None, - qwen_base_model=None, qwen_custom_model=None, qwen_design_model=None, **kwargs): + def process_dialogue(self, dialogue_json, tts_engine, pause_duration, speed_global, batch_mode, **kwargs): + # Extract optional and model-specific params from kwargs + max_batch_char = kwargs.get("max_batch_char", 1000) + cfg_scale = kwargs.get("cfg_scale", 1.5) + temperature = kwargs.get("temperature", 0.8) + top_k = kwargs.get("top_k", 20) + top_p = kwargs.get("top_p", 0.95) + + cosyvoice_model = kwargs.get("cosyvoice_model") + vibevoice_model = kwargs.get("vibevoice_model") + qwen_model = kwargs.get("qwen_model") + qwen_base_model = kwargs.get("qwen_base_model") + qwen_custom_model = kwargs.get("qwen_custom_model") + qwen_design_model = kwargs.get("qwen_design_model") # Robustness: ensure max_batch_char is correctly picked up even if shifted or provided as kwarg max_batch_char = kwargs.get("max_batch_char", max_batch_char) import json diff --git a/aiia_qwen_nodes.py b/aiia_qwen_nodes.py index 974bee2..6dbd077 100644 --- a/aiia_qwen_nodes.py +++ b/aiia_qwen_nodes.py @@ -11,31 +11,22 @@ import subprocess QWEN_SPEAKER_LIST = ["Vivian", "Serena", "Uncle_Fu", "Dylan", "Eric", "Ryan", "Aiden", "Ono_Anna", "Sohee"] QWEN_PRESET_NOTE = "Presets (9 premium timbres): Vivian/Serena/Uncle_Fu (CN), Dylan/Eric/Ryan/Aiden (EN), Ono_Anna (JP), Sohee (KR)" QWEN_EMOTION_LIST = [ - "None", "开心 (Happy)", "悲伤 (Sad)", "生气 (Angry)", "兴奋 (Excited)", - "温柔 (Gentle)", "严肃 (Serious)", "恐惧 (Fearful)", "惊讶 (Surprised)", - "低语 (Whispering)", "呐喊 (Shouting)", "羞涩 (Shy)", "诱惑 (Seductive)", - "哭腔 (Crying)", "笑声 (Laughter)", "尴尬 (Embarrassed)", "失望 (Disappointed)", - "自豪 (Proud)", "疑惑 (Doubtful)", "焦虑 (Anxious)", "平静 (Calm)" + "None", "Happy", "Sad", "Angry", "Excited", + "Gentle", "Serious", "Fearful", "Surprised", + "Whispering", "Shouting", "Shy", "Seductive", + "Crying", "Laughter", "Embarrassed", "Disappointed", + "Proud", "Doubtful", "Anxious", "Calm" ] QWEN_EXPRESSION_LIST = [ - "None", - "带点羞涩的 (With a hint of shyness)", - "语气充满诱惑力 (Seductive tone)", - "语气带着哭腔 (Crying tone)", - "稍微带一点点笑意 (With a slight smile)", - "语气显得非常疲惫 (Sounding very tired)", - "语速稍快,显得有些急促 (Hurried tone)", - "充满自信且响亮的 (Confident and loud)", - "稍微有点犹豫 and 不确定 (Hesitant and uncertain)", - "语气极其冷淡 (Extremely cold tone)", - "温柔且轻声细语的 (Gentle and whispering)" + "None", "Shyness", "Seductive", "Crying", "Smiling", + "Tired", "Hurried", "Confident", "Hesitant", "Cold", "Whispering" ] QWEN_DIALECT_LIST = [ - "None", "普通话 (Mandarin)", "粤语 (Cantonese)", "上海话 (Shanghainese)", - "四川话 (Sichuanese)", "东北话 (Northeastern)", "闽南话 (Hokkien)", - "客家话 (Hakka)", "天津话 (Tianjinese)", "山东话 (Shandongnese)" + "None", "Mandarin", "Cantonese", "Shanghainese", + "Sichuanese", "Northeastern", "Hokkien", + "Hakka", "Tianjinese", "Shandongnese" ] def _install_qwen_tts_if_needed(): @@ -503,8 +494,8 @@ class AIIA_Qwen_Dialogue_TTS: design = kwargs.get(f"speaker_{spk_key}_design", "") # 0. Specialized Qwen Routing from Bundle or slots def get_model_from_bundle(mode, ref=None): - # 1. Check primary qwen_model (might be a bundle) - main_q = kwargs.get("qwen_model") + # 1. Check primary qwen_model (passed as positional argument) + main_q = qwen_model if main_q and main_q.get("is_bundle"): if mode == "Clone" or ref is not None: return main_q.get("base") or main_q.get("default")