From b6763f7d3e760c5beffb82b1d9346c0c1f818383 Mon Sep 17 00:00:00 2001 From: AI Lab <129358391+1038lab@users.noreply.github.com> Date: Fri, 24 Jan 2025 02:29:05 -0800 Subject: [PATCH] Add files via upload --- ailab_edgeTTS.py | 52 ++++++++++++------ pyproject.toml | 6 +-- update.md | 136 +++++++++++++++++++++++++++++++++++++++++++++++ 3 files changed, 176 insertions(+), 18 deletions(-) create mode 100644 update.md diff --git a/ailab_edgeTTS.py b/ailab_edgeTTS.py index b53160a..57aa765 100644 --- a/ailab_edgeTTS.py +++ b/ailab_edgeTTS.py @@ -1,24 +1,21 @@ +# ComfyUI-EdgeTTS V1.1.0 +# A simplified Edge TTS node for ComfyUI +# Uses Microsoft Edge's online text-to-speech service +# Outputs standard ComfyUI audio format + import os import edge_tts import asyncio import re import torch import torchaudio -import nest_asyncio import json -nest_asyncio.apply() - class EdgeTTS: - """ - A simplified Edge TTS node for ComfyUI - Uses Microsoft Edge's online text-to-speech service - Outputs standard ComfyUI audio format - """ - + @staticmethod def load_voices(): - """Load available voices from config file""" + try: config_path = os.path.join(os.path.dirname(__file__), "config.json") with open(config_path, 'r', encoding='utf-8') as f: @@ -74,13 +71,32 @@ class EdgeTTS: temp_file = f"temp_tts_{os.getpid()}.wav" try: + text = text.strip() + if not text: + raise ValueError("Input text cannot be empty") + communicate = edge_tts.Communicate( text=text, voice=voice, rate=rate, pitch=f"{pitch:+d}Hz" ) - await communicate.save(temp_file) + + try: + await communicate.save(temp_file) + except edge_tts.exceptions.NoAudioReceived: + default_voice = self.DEFAULT_VOICES[0] + if voice != default_voice: + print(f"Warning: Failed with voice {voice}, trying default voice {default_voice}") + communicate = edge_tts.Communicate( + text=text, + voice=default_voice, + rate=rate, + pitch=f"{pitch:+d}Hz" + ) + await communicate.save(temp_file) + else: + raise waveform, sample_rate = torchaudio.load(temp_file) if waveform.shape[0] > 1: @@ -98,17 +114,23 @@ class EdgeTTS: def tts(self, text, voice, speed=1.0, pitch=0): """Convert text to speech""" if not text.strip(): - raise ValueError("Text is empty") + raise ValueError("Text cannot be empty") - # Clean text text = re.sub(r'\s+', ' ', text).strip() try: - audio_data = asyncio.run(self.generate_speech(text, voice, speed, pitch)) + try: + loop = asyncio.get_event_loop() + except RuntimeError: + loop = asyncio.new_event_loop() + asyncio.set_event_loop(loop) + + audio_data = loop.run_until_complete(self.generate_speech(text, voice, speed, pitch)) return (audio_data,) except Exception as e: print(f"TTS Error: {str(e)}") - raise e + empty_waveform = torch.zeros((1, 1, 16000)) + return ({"waveform": empty_waveform, "sample_rate": 16000},) NODE_CLASS_MAPPINGS = { "EdgeTTS": EdgeTTS diff --git a/pyproject.toml b/pyproject.toml index fadd9cd..e9fda82 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "comfyui-edgetts" description = "ComfyUI-EdgeTTS is a powerful text-to-speech node for ComfyUI, leveraging Microsoft's Edge TTS capabilities. It enables seamless conversion of text into natural-sounding speech, supporting multiple languages and voices. Ideal for enhancing user interactions, this node is easy to integrate and customize, making it perfect for various applications." -version = "1.0.0" +version = "1.1.0" license = {file = "LICENSE"} dependencies = [ "edge-tts>=6.1.7", @@ -12,9 +12,9 @@ dependencies = [ [project.urls] Repository = "https://github.com/1038lab/ComfyUI-EdgeTTS" -# Used by Comfy Registry https://comfyregistry.org +# Used by Comfy Registry https://comfyregistry.org [tool.comfy] PublisherId = "ailab" DisplayName = "ComfyUI-EdgeTTS" -Icon = "" +Icon = "🔊" \ No newline at end of file diff --git a/update.md b/update.md new file mode 100644 index 0000000..ca2db8a --- /dev/null +++ b/update.md @@ -0,0 +1,136 @@ +# ComfyUI-EdgeTTS Update Log + +## v1.1.0 (2025/1/24) + +### Voice Statistics +- This update adds 19 new languages and 38 new voices. +- A total of 46 language categories and 114 voices are now supported. +- The proportion of new voices accounts for approximately a 33.3% increase. + +### Voice Categories Restructuring +- Reorganized Chinese voices into more specific categories: + - Chinese-Mainland: Standard Mandarin + - Chinese-Cantonese: Hong Kong Cantonese + - Chinese-Taiwan: Taiwan Mandarin + - Chinese-Dialect: Regional dialects + +- Split English voices into regional variants: + - English-US: American English + - English-GB: British English + - English-AU: Australian English + +- Split French voices into regional variants: + - French-FR: France French + - French-CA: Canadian French + - French-CH: Swiss French + +- Split German voices into regional variants: + - German-DE: Germany German + - German-AT: Austrian German + - German-CH: Swiss German + +- Split Spanish voices into regional variants: + - Spanish-ES: Spain Spanish + - Spanish-MX: Mexican Spanish + +- Split Portuguese voices into regional variants: + - Portuguese-BR: Brazilian Portuguese + - Portuguese-PT: Portugal Portuguese + +### Chinese Voice Updates +- Added more detailed characteristics for existing Chinese voices + - XiaoxiaoNeural: Added "News, Novel" traits + - XiaoyiNeural: Added "Cartoon, Novel" traits + - YunjianNeural: Added "Sports, Novel" traits + - YunxiNeural: Added "Novel" trait + - YunxiaNeural: Added "Cartoon, Novel" traits + - YunyangNeural: Added "News" trait + - liaoning-XiaobeiNeural: Added "Dialect" trait + - shaanxi-XiaoniNeural: Added "Dialect" trait + +### New Language Support +- Added support for new languages: + 1. Bengali (India) + - BashkarNeural (Male) + - TanishaaNeural (Female) + + 2. Malay + - OsmanNeural (Male) + - YasminNeural (Female) + + 3. Tamil (India and Singapore) + - PallaviNeural (Female, IN) + - ValluvarNeural (Male, IN) + - AnbuNeural (Male, SG) + - VenbaNeural (Female, SG) + + 4. Telugu + - MohanNeural (Male) + - ShrutiNeural (Female) + + 5. Marathi + - AarohiNeural (Female) + - ManoharNeural (Male) + + 6. Swahili + - ZuriNeural (Female) + - RafikiNeural (Male) + + 7. Persian + - DilaraNeural (Female) + - FaridNeural (Male) + +2. Vietnamese + - HoaiMyNeural (Female) + - NamMinhNeural (Male) + +2. Thai + - NiwatNeural (Male) + - PremwadeeNeural (Female) + +3. Ukrainian + - OstapNeural (Male) + - PolinaNeural (Female) + +4. Greek + - AthinaNeural (Female) + - NestorasNeural (Male) + +5. Czech + - AntoninNeural (Male) + - VlastaNeural (Female) + +6. Finnish + - HarriNeural (Male) + - NooraNeural (Female) + +7. Danish + - ChristelNeural (Female) + - JeppeNeural (Male) + +8. Norwegian + - FinnNeural (Male) + - PernilleNeural (Female) + +9. Swedish + - MattiasNeural (Male) + - SofieNeural (Female) + +10. Romanian + - AlinaNeural (Female) + - EmilNeural (Male) + +11. Slovak + - LukasNeural (Male) + - ViktoriaNeural (Female) + +12. Slovenian + - PetraNeural (Female) + - RokNeural (Male) + +### Notes +- All new voices maintain consistent description format: "Friendly, Positive" +- Regional variants now clearly indicated in category names +- All voices support Neural engine technology +- Each language typically includes at least one male and one female voice +- Current voice distribution covers major world languages and regional variants \ No newline at end of file