Added ElevenLabs TTS

This commit is contained in:
Aryan
2025-10-13 16:56:14 +05:30
parent 7519c6b2a7
commit 15a05575e4
6 changed files with 180 additions and 12 deletions
+40 -1
View File
@@ -13,6 +13,8 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl
* **Google Imagen Generator & Edit:** Create and edit images with Google's Imagen models, with support for Vertex AI.
* **Nano Banana:** A creative image generation node using a specialized Gemini model.
* **Veo Text-to-Video:** Generate high-quality video clips from text prompts using Google's Veo model via Vertex AI.
* **ElevenLabs TTS:** Generate high-quality speech from text using ElevenLabs' diverse range of voices and models.
* **Gemini TTS:** Create speech from text using Google's Gemini models.
* **Seamless Integration:** All nodes are designed to work seamlessly with standard ComfyUI inputs (IMAGE, MASK, STRING) and outputs, allowing you to chain them into complex and creative workflows.
* **Secure & Simple:** Simply provide your API key in the node's input field to get started.
@@ -44,8 +46,9 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl
All nodes in this collection require API keys to function.
* **FLUX Nodes (Replicate):** You will need a [Replicate API Token](https://replicate.com/account/api-tokens).
* **Gemini, Imagen, and Nano Banana Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys).
* **Gemini, Imagen, Nano Banana, and Gemini TTS Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys).
* **GPT Image Edit Node:** You will need an [OpenAI API Key](https://platform.openai.com/api-keys).
* **ElevenLabs TTS Node:** You will need an [ElevenLabs API Key](https://elevenlabs.io/).
* **Vertex AI Nodes (Imagen Edit, Veo):** You will need a Google Cloud Project ID, a service account with appropriate permissions, and the location for the resources.
You can paste your key directly into the `api_key` field on the corresponding node. For Vertex AI nodes, you will need to provide the project ID, location, and path to your service account JSON file.
@@ -174,6 +177,42 @@ Generate short, high-quality video clips from a text description using Google's
---
### ElevenLabs TTS
Generate speech from text using the ElevenLabs API.
* **Category:** `audio/generation`
* **Inputs:**
* `text`: The text to convert to speech.
* `api_key`: Your API key from ElevenLabs.
* `voice_id`: The ID of the voice to use for generation.
* `model_id`: The ElevenLabs model to use.
* `output_format`: The desired output audio format.
* `stability`: Controls the stability and variability of the generated speech.
* `similarity_boost`: Enhances the similarity of the generated speech to the chosen voice.
* `speed`: Adjusts the speaking rate.
* `style`: Controls the expressiveness of the speech.
* `use_speaker_boost`: A boolean to enable or disable speaker boost.
* `seed`: A seed for ensuring reproducible results.
* **Output:**
* `audio`: The generated audio waveform and sample rate.
### Gemini TTS
Generate speech from text using Google's Gemini TTS models.
* **Category:** `audio/generation`
* **Inputs:**
* `text`: The text to be converted into speech.
* `api_key`: Your API key from Google AI Studio.
* `model`: The specific Gemini model to use for generation.
* `voice_id`: The prebuilt voice to use for the output.
* `temperature`: Controls the randomness and creativity of the output.
* `seed`: A seed for ensuring reproducible results.
* `system_prompt` (Optional): A system-level instruction to guide the model's behavior.
* **Output:**
* `audio`: The generated audio waveform and sample rate.
## Acknowledgements
+3 -2
View File
@@ -8,8 +8,9 @@ from .veo import NODE_CLASS_MAPPINGS as VEO_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
from .gemini_segment import NODE_CLASS_MAPPINGS as GEMINI_SEGMENT_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_SEGMENT_DISPLAY
from .nano_banana import NODE_CLASS_MAPPINGS as NANO_BANANA_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as NANO_BANANA_DISPLAY
from .gemini_tts import NODE_CLASS_MAPPINGS as GEMINI_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_TTS_DISPLAY
from .elevenlabs_tts import NODE_CLASS_MAPPINGS as ELEVENLABS_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as ELEVENLABS_TTS_DISPLAY
NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS}
NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY}
NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS, **ELEVENLABS_TTS_MAPPINGS}
NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY, **ELEVENLABS_TTS_DISPLAY}
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
+130
View File
@@ -0,0 +1,130 @@
import os
import io
import requests
import torchaudio
class ElevenLabsTTSNode:
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"text": ("STRING", {"multiline": True, "default": ""}),
"model_id": ([
"eleven_multilingual_v2",
"eleven_turbo_v2_5",
"eleven_flash_v2_5",
"eleven_flash_v2",
"eleven_turbo_v2",
"eleven_multilingual_v1",
"eleven_v3"
],),
"output_format": ([
"mp3_44100_128",
"mp3_22050_32",
"mp3_44100_32",
"mp3_44100_64",
"mp3_44100_96",
"mp3_44100_192",
"pcm_8000",
"pcm_16000",
"pcm_22050",
"pcm_24000",
"pcm_44100",
"pcm_48000",
"ulaw_8000",
"alaw_8000",
"opus_48000_32",
"opus_48000_64",
"opus_48000_96",
"opus_48000_128",
"opus_48000_192"
],),
"voice_id": ("STRING", {"multiline": False, "default": "oPM3trUCF4e0vTcsrMQr"}),
"stability": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
"similarity_boost": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
"speed": ("FLOAT", {"default": 1.0, "min": 0.25, "max": 2.0, "step": 0.01}),
"style": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
"use_speaker_boost": ("BOOLEAN", {"default": True}),
"seed": ("INT", {"default": 40, "min": 0, "max": 4294967294}),
"api_key": ("STRING", {"multiline": False, "default": ""}),
},
"optional": {
"previous_text": ("STRING", {"multiline": True, "default": ""}),
"next_text": ("STRING", {"multiline": True, "default": ""}),
}
}
RETURN_TYPES = ("AUDIO",)
RETURN_NAMES = ("audio",)
FUNCTION = "generate_speech"
CATEGORY = "audio/generation"
def generate_speech(self, text, api_key, voice_id, model_id, output_format,
stability, similarity_boost, speed, style, use_speaker_boost, seed,
previous_text="", next_text=""):
if not text.strip():
raise ValueError("Text input cannot be empty.")
key = api_key.strip() or os.environ.get("XI_API_KEY")
if not key:
raise ValueError("No API key provided.")
url = f"https://api.elevenlabs.io/v1/text-to-speech/{voice_id}"
headers = {
"xi-api-key": key,
"Content-Type": "application/json"
}
if model_id == "eleven_v3":
allowed_stabilities = [0.0, 0.5, 1.0]
original_stability = stability
stability = min(allowed_stabilities, key=lambda x: abs(x - original_stability))
if stability != original_stability:
print(f"For 'eleven_v3' model, stability value must be one of: [0.0, 0.5, 1.0] (0.0 = Creative, 0.5 = Natural, 1.0 = Robust). Rounding stability to {stability}.")
voice_settings = {
"stability": stability,
"similarity_boost": similarity_boost,
"speed": speed,
"style": style,
"use_speaker_boost": use_speaker_boost
}
data = {
"text": text,
"voice_settings": voice_settings,
"model_id": model_id,
"seed": seed,
"output_format": output_format
}
if model_id == "eleven_v3":
if previous_text.strip() or next_text.strip():
print("Providing previous_text or next_text is not yet supported with the 'eleven_v3' model. Ignoring these inputs.")
else:
if previous_text.strip():
data["previous_text"] = previous_text
if next_text.strip():
data["next_text"] = next_text
response = requests.post(url, json=data, headers=headers)
if response.status_code != 200:
raise Exception(f"ElevenLabs API Error: {response.status_code}, {response.text}")
# Decode audio from memory
audio_buffer = io.BytesIO(response.content)
waveform, sample_rate = torchaudio.load(audio_buffer)
# Return in ComfyUI audio format
return ({"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate},)
@classmethod
def IS_CHANGED(cls, **kwargs):
return f"{kwargs.get('text', '')}-{kwargs.get('voice_id', '')}-{kwargs.get('seed', 40)}"
NODE_CLASS_MAPPINGS = {"ElevenLabsTTSNode": ElevenLabsTTSNode}
NODE_DISPLAY_NAME_MAPPINGS = {"ElevenLabsTTSNode": "ElevenLabs TTS"}
-1
View File
@@ -1,4 +1,3 @@
import base64
import os
import io
import numpy as np
+7 -7
View File
@@ -10,15 +10,15 @@ class GeminiTTSNode:
def INPUT_TYPES(cls):
return {
"required": {
"text": ("STRING", {"multiline": True, "default": ""}),
"api_key": ("STRING", {"multiline": False, "default": ""}),
"model": ("STRING", {"default": "gemini-2.5-flash-preview-tts", "multiline": False}),
"voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],),
"seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}),
"temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
"text": ("STRING", {"multiline": True, "default": ""}),
"api_key": ("STRING", {"multiline": False, "default": ""}),
"model": (["gemini-2.5-flash-preview-tts", "gemini-2.5-pro-preview-tts"],),
"voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],),
"seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}),
"temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
},
"optional": {
"system_prompt": ("STRING", {"multiline": True, "default": ""}),
"system_prompt": ("STRING", {"multiline": True, "default": ""}),
}
}
-1
View File
@@ -1,6 +1,5 @@
import requests
import base64
import json
import torch
import numpy as np
from PIL import Image