diff --git a/ailab_edgeTTS.py b/ailab_edgeTTS.py index 57aa765..84b5b39 100644 --- a/ailab_edgeTTS.py +++ b/ailab_edgeTTS.py @@ -1,141 +1,174 @@ -# ComfyUI-EdgeTTS V1.1.0 -# A simplified Edge TTS node for ComfyUI -# Uses Microsoft Edge's online text-to-speech service -# Outputs standard ComfyUI audio format - -import os -import edge_tts -import asyncio -import re -import torch -import torchaudio -import json - -class EdgeTTS: - - @staticmethod - def load_voices(): - - try: - config_path = os.path.join(os.path.dirname(__file__), "config.json") - with open(config_path, 'r', encoding='utf-8') as f: - config = json.load(f) - voices = [] - tooltips = {} - - default_voice = config.get("default_voice") - - for language, voice_list in config["edge_tts_voices"].items(): - for voice, description in voice_list: - voices.append(voice) - tooltips[voice] = f"{language}: {description}" - - if default_voice in voices: - voices.remove(default_voice) - voices.insert(0, default_voice) - - return voices, tooltips - except: - return ( - ["zh-CN-XiaoxiaoNeural", "en-US-JennyNeural", "ja-JP-NanamiNeural"], - { - "zh-CN-XiaoxiaoNeural": "Chinese: Female, cheerful", - "en-US-JennyNeural": "English: Female, casual", - "ja-JP-NanamiNeural": "Japanese: Female, natural" - } - ) - - DEFAULT_VOICES, VOICE_TOOLTIPS = load_voices.__func__() - - @classmethod - def INPUT_TYPES(s): - return { - "required": { - "text": ("STRING", {"multiline": True, "placeholder": "Enter text to convert to speech"}), - "voice": (s.DEFAULT_VOICES, {"default": s.DEFAULT_VOICES[0], "tooltip": "Select a voice for text-to-speech"}), - }, - "optional": { - "speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1, "tooltip": "Speech rate (0.5 to 2.0)"}), - "pitch": ("INT", {"default": 0, "min": -20, "max": 20, "step": 1, "tooltip": "Voice pitch adjustment (-20 to +20 Hz)"}) - } - } - - RETURN_TYPES = ("AUDIO",) - FUNCTION = "tts" - CATEGORY = "🧪AILab/🔊Audio" - - async def generate_speech(self, text, voice, speed, pitch): - """Generate speech from text using Edge TTS""" - speed_percent = int((speed - 1.0) * 100) - rate = "+0%" if speed_percent == 0 else f"{speed_percent:+d}%" - - temp_file = f"temp_tts_{os.getpid()}.wav" - try: - text = text.strip() - if not text: - raise ValueError("Input text cannot be empty") - - communicate = edge_tts.Communicate( - text=text, - voice=voice, - rate=rate, - pitch=f"{pitch:+d}Hz" - ) - - try: - await communicate.save(temp_file) - except edge_tts.exceptions.NoAudioReceived: - default_voice = self.DEFAULT_VOICES[0] - if voice != default_voice: - print(f"Warning: Failed with voice {voice}, trying default voice {default_voice}") - communicate = edge_tts.Communicate( - text=text, - voice=default_voice, - rate=rate, - pitch=f"{pitch:+d}Hz" - ) - await communicate.save(temp_file) - else: - raise - - waveform, sample_rate = torchaudio.load(temp_file) - if waveform.shape[0] > 1: - waveform = waveform.mean(dim=0, keepdim=True) - waveform = waveform / (waveform.abs().max() + 1e-6) - return {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate} - - finally: - if os.path.exists(temp_file): - try: - os.remove(temp_file) - except: - pass - - def tts(self, text, voice, speed=1.0, pitch=0): - """Convert text to speech""" - if not text.strip(): - raise ValueError("Text cannot be empty") - - text = re.sub(r'\s+', ' ', text).strip() - - try: - try: - loop = asyncio.get_event_loop() - except RuntimeError: - loop = asyncio.new_event_loop() - asyncio.set_event_loop(loop) - - audio_data = loop.run_until_complete(self.generate_speech(text, voice, speed, pitch)) - return (audio_data,) - except Exception as e: - print(f"TTS Error: {str(e)}") - empty_waveform = torch.zeros((1, 1, 16000)) - return ({"waveform": empty_waveform, "sample_rate": 16000},) - -NODE_CLASS_MAPPINGS = { - "EdgeTTS": EdgeTTS -} - -NODE_DISPLAY_NAME_MAPPINGS = { - "EdgeTTS": "Edge TTS 🔊" +# ComfyUI-EdgeTTS V1.2.0 +# A simplified Edge TTS node for ComfyUI +# Uses Microsoft Edge's online text-to-speech service +# Outputs standard ComfyUI audio format + +import os +import edge_tts +import asyncio +import re +import torch +import torchaudio +import json + +class EdgeTTS: + _voice_data_cache = None + + @classmethod + def get_voice_data(cls): + if cls._voice_data_cache is None: + cls._voice_data_cache = cls.load_voices() + return cls._voice_data_cache + + @staticmethod + def load_voices(): + + try: + config_path = os.path.join(os.path.dirname(__file__), "config.json") + with open(config_path, 'r', encoding='utf-8') as f: + config = json.load(f) + voices = [] + tooltips = {} + voice_ids = {} + + default_voice = config.get("default_voice") + default_display_name = None + + for language, voice_list in config["edge_tts_voices"].items(): + for voice, description in voice_list: + parts = voice.split('-') + if len(parts) >= 3: + lang_name = language.split('-')[0] if '-' in language else language + display_name = f"[{lang_name}] {parts[0]}-{parts[1]} {parts[-1].replace('Neural', '').replace('Multilingual', '')}" + else: + display_name = voice + voices.append(display_name) + tooltips[display_name] = f"{language}: {description}" + voice_ids[display_name] = voice + + if voice == default_voice: + default_display_name = display_name + + if default_display_name and default_display_name in voices: + voices.remove(default_display_name) + voices.insert(0, default_display_name) + + return voices, tooltips, voice_ids + except: + return ( + ["[English] en-US Jenny", "[Chinese] zh-CN Xiaoxiao", "[Japanese] ja-JP Nanami"], + {"[English] en-US Jenny": "English-US: Female, casual", "[Chinese] zh-CN Xiaoxiao": "Chinese-Mainland: Female, cheerful", "[Japanese] ja-JP Nanami": "Japanese: Female, natural"}, + {"[English] en-US Jenny": "en-US-JennyNeural", "[Chinese] zh-CN Xiaoxiao": "zh-CN-XiaoxiaoNeural", "[Japanese] ja-JP Nanami": "ja-JP-NanamiNeural"} + ) + + @property + def DEFAULT_VOICES(self): + return self.get_voice_data()[0] + + @property + def VOICE_TOOLTIPS(self): + return self.get_voice_data()[1] + + @property + def VOICE_IDS(self): + return self.get_voice_data()[2] + + @classmethod + def INPUT_TYPES(cls): + voices, _, _ = cls.get_voice_data() + return { + "required": { + "text": ("STRING", {"multiline": True, "placeholder": "Enter text to convert to speech"}), + "voice": (voices, {"default": voices[0], "tooltip": "Select a voice for text-to-speech"}), + }, + "optional": { + "speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 2.0, "step": 0.1, "tooltip": "Speech rate (0.5 to 2.0)"}), + "pitch": ("INT", {"default": 0, "min": -20, "max": 20, "step": 1, "tooltip": "Voice pitch adjustment (-20 to +20 Hz)"}) + } + } + + RETURN_TYPES = ("AUDIO",) + FUNCTION = "tts" + CATEGORY = "🧪AILab/🔊Audio" + + async def generate_speech(self, text, voice, speed, pitch): + """Generate speech from text using Edge TTS""" + speed_percent = int((speed - 1.0) * 100) + rate = "+0%" if speed_percent == 0 else f"{speed_percent:+d}%" + + temp_file = f"temp_tts_{os.getpid()}.wav" + try: + text = text.strip() + if not text: + raise ValueError("Input text cannot be empty") + + communicate = edge_tts.Communicate( + text=text, + voice=voice, + rate=rate, + pitch=f"{pitch:+d}Hz" + ) + + try: + await communicate.save(temp_file) + except edge_tts.exceptions.NoAudioReceived: + + default_display_name = self.DEFAULT_VOICES[0] + default_voice = self.VOICE_IDS.get(default_display_name, default_display_name) + + if voice != default_voice: + print(f"Warning: Failed with voice {voice}, trying default voice {default_voice}") + communicate = edge_tts.Communicate( + text=text, + voice=default_voice, + rate=rate, + pitch=f"{pitch:+d}Hz" + ) + await communicate.save(temp_file) + else: + raise + + waveform, sample_rate = torchaudio.load(temp_file) + if waveform.shape[0] > 1: + waveform = waveform.mean(dim=0, keepdim=True) + waveform = waveform / (waveform.abs().max() + 1e-6) + return {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate} + + finally: + if os.path.exists(temp_file): + try: + os.remove(temp_file) + except: + pass + + def tts(self, text, voice, speed=1.0, pitch=0): + """Convert text to speech""" + if not text.strip(): + raise ValueError("Text cannot be empty") + + text = re.sub(r'\s+', ' ', text).strip() + + + actual_voice = self.VOICE_IDS.get(voice, voice) + + try: + try: + loop = asyncio.get_event_loop() + except RuntimeError: + loop = asyncio.new_event_loop() + asyncio.set_event_loop(loop) + + audio_data = loop.run_until_complete(self.generate_speech(text, actual_voice, speed, pitch)) + return (audio_data,) + except Exception as e: + print(f"TTS Error: {str(e)}") + empty_waveform = torch.zeros((1, 1, 16000)) + return ({"waveform": empty_waveform, "sample_rate": 16000},) + +NODE_CLASS_MAPPINGS = { + "EdgeTTS": EdgeTTS +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "EdgeTTS": "Edge TTS 🔊" } \ No newline at end of file diff --git a/config.json b/config.json index 3786d0e..abff963 100644 --- a/config.json +++ b/config.json @@ -3,7 +3,7 @@ "default_language": "en-US", "default_voice": "en-US-JennyNeural", "default_speed": 1.0, - "default_audio_format": "wav", + "default_audio_format": "mp3", "ffmpeg_path": "ffmpeg", "edge_tts_voices": { "Chinese-Mainland": [ @@ -12,7 +12,10 @@ ["zh-CN-YunjianNeural", "Male, Sports, Novel, Passion"], ["zh-CN-YunxiNeural", "Male, Novel, Lively, Sunshine"], ["zh-CN-YunxiaNeural", "Male, Cartoon, Novel, Cute"], - ["zh-CN-YunyangNeural", "Male, News, Professional, Reliable"] + ["zh-CN-YunyangNeural", "Male, News, Professional, Reliable"], + ["zh-CN-XiaoxiaoMultilingualNeural", "Female, Friendly, Positive"], + ["zh-CN-YunfanMultilingualNeural", "Male, Friendly, Positive"], + ["zh-CN-YunxiaoMultilingualNeural", "Male, Friendly, Positive"] ], "Chinese-Cantonese": [ ["zh-HK-HiuGaaiNeural", "Female, Friendly, Positive"], @@ -37,22 +40,74 @@ ["en-US-EricNeural", "Male, Rational"], ["en-US-MichelleNeural", "Female, Friendly, Pleasant"], ["en-US-RogerNeural", "Male, Lively"], - ["en-US-SteffanNeural", "Male, Rational"] + ["en-US-SteffanNeural", "Male, Rational"], + ["en-US-AvaMultilingualNeural", "Female, Expressive, Caring, Pleasant, Friendly"], + ["en-US-AndrewMultilingualNeural", "Male, Warm, Confident, Authentic, Honest"], + ["en-US-EmmaMultilingualNeural", "Female, Cheerful, Clear, Conversational"], + ["en-US-BrianMultilingualNeural", "Male, Approachable, Casual, Sincere"] ], "English-GB": [ ["en-GB-LibbyNeural", "Female, Friendly, Positive"], ["en-GB-MaisieNeural", "Female, Friendly, Positive"], ["en-GB-RyanNeural", "Male, Friendly, Positive"], ["en-GB-SoniaNeural", "Female, Friendly, Positive"], - ["en-GB-ThomasNeural", "Male, Friendly, Positive"] + ["en-GB-ThomasNeural", "Male, Friendly, Positive"], + ["en-GB-AdaMultilingualNeural", "Female, Friendly, Positive"], + ["en-GB-OllieMultilingualNeural", "Male, Friendly, Positive"] ], "English-AU": [ ["en-AU-NatashaNeural", "Female, Friendly, Positive"], ["en-AU-WilliamNeural", "Male, Friendly, Positive"] ], + "English-CA": [ + ["en-CA-ClaraNeural", "Female, Friendly, Positive"], + ["en-CA-LiamNeural", "Male, Friendly, Positive"] + ], + "English-HK": [ + ["en-HK-SamNeural", "Male, Friendly, Positive"], + ["en-HK-YanNeural", "Female, Friendly, Positive"] + ], + "English-IN": [ + ["en-IN-NeerjaExpressiveNeural", "Female, Friendly, Positive"], + ["en-IN-NeerjaNeural", "Female, Friendly, Positive"], + ["en-IN-PrabhatNeural", "Male, Friendly, Positive"] + ], + "English-IE": [ + ["en-IE-ConnorNeural", "Male, Friendly, Positive"], + ["en-IE-EmilyNeural", "Female, Friendly, Positive"] + ], + "English-KE": [ + ["en-KE-AsiliaNeural", "Female, Friendly, Positive"], + ["en-KE-ChilembaNeural", "Male, Friendly, Positive"] + ], + "English-NZ": [ + ["en-NZ-MitchellNeural", "Male, Friendly, Positive"], + ["en-NZ-MollyNeural", "Female, Friendly, Positive"] + ], + "English-NG": [ + ["en-NG-AbeoNeural", "Male, Friendly, Positive"], + ["en-NG-EzinneNeural", "Female, Friendly, Positive"] + ], + "English-PH": [ + ["en-PH-JamesNeural", "Male, Friendly, Positive"], + ["en-PH-RosaNeural", "Female, Friendly, Positive"] + ], + "English-SG": [ + ["en-SG-LunaNeural", "Female, Friendly, Positive"], + ["en-SG-WayneNeural", "Male, Friendly, Positive"] + ], + "English-ZA": [ + ["en-ZA-LeahNeural", "Female, Friendly, Positive"], + ["en-ZA-LukeNeural", "Male, Friendly, Positive"] + ], + "English-TZ": [ + ["en-TZ-ElimuNeural", "Male, Friendly, Positive"], + ["en-TZ-ImaniNeural", "Female, Friendly, Positive"] + ], "Japanese": [ ["ja-JP-NanamiNeural", "Female, Friendly, Positive"], - ["ja-JP-KeitaNeural", "Male, Friendly, Positive"] + ["ja-JP-KeitaNeural", "Male, Friendly, Positive"], + ["ja-JP-MasaruMultilingualNeural", "Male, Friendly, Positive"] ], "Korean": [ ["ko-KR-SunHiNeural", "Female, Friendly, Positive"], @@ -94,7 +149,10 @@ "Spanish": [ ["es-ES-ElviraNeural", "Female, Friendly, Positive"], ["es-ES-AlvaroNeural", "Male, Friendly, Positive"], - ["es-ES-XimenaNeural", "Female, Friendly, Positive"] + ["es-ES-XimenaNeural", "Female, Friendly, Positive"], + ["es-ES-IsidoraMultilingualNeural", "Female, Friendly, Positive"], + ["es-ES-ArabellaMultilingualNeural", "Female, Friendly, Positive"], + ["es-ES-TristanMultilingualNeural", "Male, Friendly, Positive"] ], "Spanish-MX": [ ["es-MX-DaliaNeural", "Female, Friendly, Positive"], @@ -109,12 +167,16 @@ ["it-IT-ElsaNeural", "Female, Friendly, Positive"], ["it-IT-DiegoNeural", "Male, Friendly, Positive"], ["it-IT-IsabellaNeural", "Female, Friendly, Positive"], - ["it-IT-GiuseppeMultilingualNeural", "Male, Friendly, Positive"] + ["it-IT-GiuseppeMultilingualNeural", "Male, Friendly, Positive"], + ["it-IT-IsabellaMultilingualNeural", "Female, Friendly, Positive"], + ["it-IT-MarcelloMultilingualNeural", "Male, Friendly, Positive"], + ["it-IT-AlessioMultilingualNeural", "Male, Friendly, Positive"] ], "Portuguese-BR": [ ["pt-BR-FranciscaNeural", "Female, Friendly, Positive"], ["pt-BR-AntonioNeural", "Male, Friendly, Positive"], - ["pt-BR-ThalitaMultilingualNeural", "Female, Friendly, Positive"] + ["pt-BR-ThalitaMultilingualNeural", "Female, Friendly, Positive"], + ["pt-BR-MacerioMultilingualNeural", "Male, Friendly, Positive"] ], "Portuguese-PT": [ ["pt-PT-RaquelNeural", "Female, Friendly, Positive"],