From 15a05575e40816d05d165654d1079d1f38aa39ad Mon Sep 17 00:00:00 2001 From: Aryan Date: Mon, 13 Oct 2025 16:56:14 +0530 Subject: [PATCH] Added ElevenLabs TTS --- README.md | 41 ++++++++++++++- __init__.py | 5 +- elevenlabs_tts.py | 130 ++++++++++++++++++++++++++++++++++++++++++++++ gemini_node.py | 1 - gemini_tts.py | 14 ++--- gpt_image1.py | 1 - 6 files changed, 180 insertions(+), 12 deletions(-) create mode 100644 elevenlabs_tts.py diff --git a/README.md b/README.md index 9789dce..aef92ba 100644 --- a/README.md +++ b/README.md @@ -13,6 +13,8 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl * **Google Imagen Generator & Edit:** Create and edit images with Google's Imagen models, with support for Vertex AI. * **Nano Banana:** A creative image generation node using a specialized Gemini model. * **Veo Text-to-Video:** Generate high-quality video clips from text prompts using Google's Veo model via Vertex AI. +* **ElevenLabs TTS:** Generate high-quality speech from text using ElevenLabs' diverse range of voices and models. +* **Gemini TTS:** Create speech from text using Google's Gemini models. * **Seamless Integration:** All nodes are designed to work seamlessly with standard ComfyUI inputs (IMAGE, MASK, STRING) and outputs, allowing you to chain them into complex and creative workflows. * **Secure & Simple:** Simply provide your API key in the node's input field to get started. @@ -44,8 +46,9 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl All nodes in this collection require API keys to function. * **FLUX Nodes (Replicate):** You will need a [Replicate API Token](https://replicate.com/account/api-tokens). -* **Gemini, Imagen, and Nano Banana Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys). +* **Gemini, Imagen, Nano Banana, and Gemini TTS Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys). * **GPT Image Edit Node:** You will need an [OpenAI API Key](https://platform.openai.com/api-keys). +* **ElevenLabs TTS Node:** You will need an [ElevenLabs API Key](https://elevenlabs.io/). * **Vertex AI Nodes (Imagen Edit, Veo):** You will need a Google Cloud Project ID, a service account with appropriate permissions, and the location for the resources. You can paste your key directly into the `api_key` field on the corresponding node. For Vertex AI nodes, you will need to provide the project ID, location, and path to your service account JSON file. @@ -174,6 +177,42 @@ Generate short, high-quality video clips from a text description using Google's --- +### ElevenLabs TTS + +Generate speech from text using the ElevenLabs API. + +* **Category:** `audio/generation` +* **Inputs:** + * `text`: The text to convert to speech. + * `api_key`: Your API key from ElevenLabs. + * `voice_id`: The ID of the voice to use for generation. + * `model_id`: The ElevenLabs model to use. + * `output_format`: The desired output audio format. + * `stability`: Controls the stability and variability of the generated speech. + * `similarity_boost`: Enhances the similarity of the generated speech to the chosen voice. + * `speed`: Adjusts the speaking rate. + * `style`: Controls the expressiveness of the speech. + * `use_speaker_boost`: A boolean to enable or disable speaker boost. + * `seed`: A seed for ensuring reproducible results. +* **Output:** + * `audio`: The generated audio waveform and sample rate. + +### Gemini TTS + +Generate speech from text using Google's Gemini TTS models. + +* **Category:** `audio/generation` +* **Inputs:** + * `text`: The text to be converted into speech. + * `api_key`: Your API key from Google AI Studio. + * `model`: The specific Gemini model to use for generation. + * `voice_id`: The prebuilt voice to use for the output. + * `temperature`: Controls the randomness and creativity of the output. + * `seed`: A seed for ensuring reproducible results. + * `system_prompt` (Optional): A system-level instruction to guide the model's behavior. +* **Output:** + * `audio`: The generated audio waveform and sample rate. + ## Acknowledgements diff --git a/__init__.py b/__init__.py index e545377..ec292c7 100644 --- a/__init__.py +++ b/__init__.py @@ -8,8 +8,9 @@ from .veo import NODE_CLASS_MAPPINGS as VEO_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS from .gemini_segment import NODE_CLASS_MAPPINGS as GEMINI_SEGMENT_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_SEGMENT_DISPLAY from .nano_banana import NODE_CLASS_MAPPINGS as NANO_BANANA_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as NANO_BANANA_DISPLAY from .gemini_tts import NODE_CLASS_MAPPINGS as GEMINI_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_TTS_DISPLAY +from .elevenlabs_tts import NODE_CLASS_MAPPINGS as ELEVENLABS_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as ELEVENLABS_TTS_DISPLAY -NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS} -NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY} +NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS, **ELEVENLABS_TTS_MAPPINGS} +NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY, **ELEVENLABS_TTS_DISPLAY} __all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS'] \ No newline at end of file diff --git a/elevenlabs_tts.py b/elevenlabs_tts.py new file mode 100644 index 0000000..b5a275a --- /dev/null +++ b/elevenlabs_tts.py @@ -0,0 +1,130 @@ +import os +import io +import requests +import torchaudio + +class ElevenLabsTTSNode: + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "text": ("STRING", {"multiline": True, "default": ""}), + "model_id": ([ + "eleven_multilingual_v2", + "eleven_turbo_v2_5", + "eleven_flash_v2_5", + "eleven_flash_v2", + "eleven_turbo_v2", + "eleven_multilingual_v1", + "eleven_v3" + ],), + "output_format": ([ + "mp3_44100_128", + "mp3_22050_32", + "mp3_44100_32", + "mp3_44100_64", + "mp3_44100_96", + "mp3_44100_192", + "pcm_8000", + "pcm_16000", + "pcm_22050", + "pcm_24000", + "pcm_44100", + "pcm_48000", + "ulaw_8000", + "alaw_8000", + "opus_48000_32", + "opus_48000_64", + "opus_48000_96", + "opus_48000_128", + "opus_48000_192" + ],), + "voice_id": ("STRING", {"multiline": False, "default": "oPM3trUCF4e0vTcsrMQr"}), + "stability": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}), + "similarity_boost": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}), + "speed": ("FLOAT", {"default": 1.0, "min": 0.25, "max": 2.0, "step": 0.01}), + "style": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}), + "use_speaker_boost": ("BOOLEAN", {"default": True}), + "seed": ("INT", {"default": 40, "min": 0, "max": 4294967294}), + "api_key": ("STRING", {"multiline": False, "default": ""}), + }, + "optional": { + "previous_text": ("STRING", {"multiline": True, "default": ""}), + "next_text": ("STRING", {"multiline": True, "default": ""}), + } + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio",) + FUNCTION = "generate_speech" + CATEGORY = "audio/generation" + + def generate_speech(self, text, api_key, voice_id, model_id, output_format, + stability, similarity_boost, speed, style, use_speaker_boost, seed, + previous_text="", next_text=""): + + if not text.strip(): + raise ValueError("Text input cannot be empty.") + + key = api_key.strip() or os.environ.get("XI_API_KEY") + if not key: + raise ValueError("No API key provided.") + + url = f"https://api.elevenlabs.io/v1/text-to-speech/{voice_id}" + + headers = { + "xi-api-key": key, + "Content-Type": "application/json" + } + + if model_id == "eleven_v3": + allowed_stabilities = [0.0, 0.5, 1.0] + original_stability = stability + stability = min(allowed_stabilities, key=lambda x: abs(x - original_stability)) + if stability != original_stability: + print(f"For 'eleven_v3' model, stability value must be one of: [0.0, 0.5, 1.0] (0.0 = Creative, 0.5 = Natural, 1.0 = Robust). Rounding stability to {stability}.") + + voice_settings = { + "stability": stability, + "similarity_boost": similarity_boost, + "speed": speed, + "style": style, + "use_speaker_boost": use_speaker_boost + } + + data = { + "text": text, + "voice_settings": voice_settings, + "model_id": model_id, + "seed": seed, + "output_format": output_format + } + + if model_id == "eleven_v3": + if previous_text.strip() or next_text.strip(): + print("Providing previous_text or next_text is not yet supported with the 'eleven_v3' model. Ignoring these inputs.") + else: + if previous_text.strip(): + data["previous_text"] = previous_text + if next_text.strip(): + data["next_text"] = next_text + + response = requests.post(url, json=data, headers=headers) + + if response.status_code != 200: + raise Exception(f"ElevenLabs API Error: {response.status_code}, {response.text}") + + # Decode audio from memory + audio_buffer = io.BytesIO(response.content) + waveform, sample_rate = torchaudio.load(audio_buffer) + + # Return in ComfyUI audio format + return ({"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate},) + + @classmethod + def IS_CHANGED(cls, **kwargs): + return f"{kwargs.get('text', '')}-{kwargs.get('voice_id', '')}-{kwargs.get('seed', 40)}" + +NODE_CLASS_MAPPINGS = {"ElevenLabsTTSNode": ElevenLabsTTSNode} +NODE_DISPLAY_NAME_MAPPINGS = {"ElevenLabsTTSNode": "ElevenLabs TTS"} \ No newline at end of file diff --git a/gemini_node.py b/gemini_node.py index 24cabdc..cda95db 100644 --- a/gemini_node.py +++ b/gemini_node.py @@ -1,4 +1,3 @@ -import base64 import os import io import numpy as np diff --git a/gemini_tts.py b/gemini_tts.py index 05c08d9..bc3cd19 100644 --- a/gemini_tts.py +++ b/gemini_tts.py @@ -10,15 +10,15 @@ class GeminiTTSNode: def INPUT_TYPES(cls): return { "required": { - "text": ("STRING", {"multiline": True, "default": ""}), - "api_key": ("STRING", {"multiline": False, "default": ""}), - "model": ("STRING", {"default": "gemini-2.5-flash-preview-tts", "multiline": False}), - "voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],), - "seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}), - "temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "text": ("STRING", {"multiline": True, "default": ""}), + "api_key": ("STRING", {"multiline": False, "default": ""}), + "model": (["gemini-2.5-flash-preview-tts", "gemini-2.5-pro-preview-tts"],), + "voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],), + "seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}), + "temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), }, "optional": { - "system_prompt": ("STRING", {"multiline": True, "default": ""}), + "system_prompt": ("STRING", {"multiline": True, "default": ""}), } } diff --git a/gpt_image1.py b/gpt_image1.py index cff3ee6..2e30641 100644 --- a/gpt_image1.py +++ b/gpt_image1.py @@ -1,6 +1,5 @@ import requests import base64 -import json import torch import numpy as np from PIL import Image