Added ElevenLabs TTS
This commit is contained in:
@@ -13,6 +13,8 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl
|
||||
* **Google Imagen Generator & Edit:** Create and edit images with Google's Imagen models, with support for Vertex AI.
|
||||
* **Nano Banana:** A creative image generation node using a specialized Gemini model.
|
||||
* **Veo Text-to-Video:** Generate high-quality video clips from text prompts using Google's Veo model via Vertex AI.
|
||||
* **ElevenLabs TTS:** Generate high-quality speech from text using ElevenLabs' diverse range of voices and models.
|
||||
* **Gemini TTS:** Create speech from text using Google's Gemini models.
|
||||
* **Seamless Integration:** All nodes are designed to work seamlessly with standard ComfyUI inputs (IMAGE, MASK, STRING) and outputs, allowing you to chain them into complex and creative workflows.
|
||||
* **Secure & Simple:** Simply provide your API key in the node's input field to get started.
|
||||
|
||||
@@ -44,8 +46,9 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl
|
||||
All nodes in this collection require API keys to function.
|
||||
|
||||
* **FLUX Nodes (Replicate):** You will need a [Replicate API Token](https://replicate.com/account/api-tokens).
|
||||
* **Gemini, Imagen, and Nano Banana Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys).
|
||||
* **Gemini, Imagen, Nano Banana, and Gemini TTS Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys).
|
||||
* **GPT Image Edit Node:** You will need an [OpenAI API Key](https://platform.openai.com/api-keys).
|
||||
* **ElevenLabs TTS Node:** You will need an [ElevenLabs API Key](https://elevenlabs.io/).
|
||||
* **Vertex AI Nodes (Imagen Edit, Veo):** You will need a Google Cloud Project ID, a service account with appropriate permissions, and the location for the resources.
|
||||
|
||||
You can paste your key directly into the `api_key` field on the corresponding node. For Vertex AI nodes, you will need to provide the project ID, location, and path to your service account JSON file.
|
||||
@@ -174,6 +177,42 @@ Generate short, high-quality video clips from a text description using Google's
|
||||
|
||||
---
|
||||
|
||||
### ElevenLabs TTS
|
||||
|
||||
Generate speech from text using the ElevenLabs API.
|
||||
|
||||
* **Category:** `audio/generation`
|
||||
* **Inputs:**
|
||||
* `text`: The text to convert to speech.
|
||||
* `api_key`: Your API key from ElevenLabs.
|
||||
* `voice_id`: The ID of the voice to use for generation.
|
||||
* `model_id`: The ElevenLabs model to use.
|
||||
* `output_format`: The desired output audio format.
|
||||
* `stability`: Controls the stability and variability of the generated speech.
|
||||
* `similarity_boost`: Enhances the similarity of the generated speech to the chosen voice.
|
||||
* `speed`: Adjusts the speaking rate.
|
||||
* `style`: Controls the expressiveness of the speech.
|
||||
* `use_speaker_boost`: A boolean to enable or disable speaker boost.
|
||||
* `seed`: A seed for ensuring reproducible results.
|
||||
* **Output:**
|
||||
* `audio`: The generated audio waveform and sample rate.
|
||||
|
||||
### Gemini TTS
|
||||
|
||||
Generate speech from text using Google's Gemini TTS models.
|
||||
|
||||
* **Category:** `audio/generation`
|
||||
* **Inputs:**
|
||||
* `text`: The text to be converted into speech.
|
||||
* `api_key`: Your API key from Google AI Studio.
|
||||
* `model`: The specific Gemini model to use for generation.
|
||||
* `voice_id`: The prebuilt voice to use for the output.
|
||||
* `temperature`: Controls the randomness and creativity of the output.
|
||||
* `seed`: A seed for ensuring reproducible results.
|
||||
* `system_prompt` (Optional): A system-level instruction to guide the model's behavior.
|
||||
* **Output:**
|
||||
* `audio`: The generated audio waveform and sample rate.
|
||||
|
||||
|
||||
## Acknowledgements
|
||||
|
||||
|
||||
+3
-2
@@ -8,8 +8,9 @@ from .veo import NODE_CLASS_MAPPINGS as VEO_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
from .gemini_segment import NODE_CLASS_MAPPINGS as GEMINI_SEGMENT_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_SEGMENT_DISPLAY
|
||||
from .nano_banana import NODE_CLASS_MAPPINGS as NANO_BANANA_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as NANO_BANANA_DISPLAY
|
||||
from .gemini_tts import NODE_CLASS_MAPPINGS as GEMINI_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as GEMINI_TTS_DISPLAY
|
||||
from .elevenlabs_tts import NODE_CLASS_MAPPINGS as ELEVENLABS_TTS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as ELEVENLABS_TTS_DISPLAY
|
||||
|
||||
NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY}
|
||||
NODE_CLASS_MAPPINGS = {**PRO_MAPPINGS, **MAX_MAPPINGS, **GEMINI_MAPPINGS, **GPT_IMAGE_MAPPINGS, **IMAGEN_IMAGE_MAPPINGS, **IMAGEN_EDIT_MAPPINGS, **VEO_MAPPINGS, **GEMINI_SEGMENT_MAPPINGS, **NANO_BANANA_MAPPINGS, **GEMINI_TTS_MAPPINGS, **ELEVENLABS_TTS_MAPPINGS}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {**PRO_DISPLAY, **MAX_DISPLAY, **GEMINI_DISPLAY, **GPT_IMAGE_DISPLAY, **IMAGEN_IMAGE_DISPLAY, **IMAGEN_EDIT_DISPLAY, **VEO_DISPLAY, **GEMINI_SEGMENT_DISPLAY, **NANO_BANANA_DISPLAY, **GEMINI_TTS_DISPLAY, **ELEVENLABS_TTS_DISPLAY}
|
||||
|
||||
__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS']
|
||||
@@ -0,0 +1,130 @@
|
||||
import os
|
||||
import io
|
||||
import requests
|
||||
import torchaudio
|
||||
|
||||
class ElevenLabsTTSNode:
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"text": ("STRING", {"multiline": True, "default": ""}),
|
||||
"model_id": ([
|
||||
"eleven_multilingual_v2",
|
||||
"eleven_turbo_v2_5",
|
||||
"eleven_flash_v2_5",
|
||||
"eleven_flash_v2",
|
||||
"eleven_turbo_v2",
|
||||
"eleven_multilingual_v1",
|
||||
"eleven_v3"
|
||||
],),
|
||||
"output_format": ([
|
||||
"mp3_44100_128",
|
||||
"mp3_22050_32",
|
||||
"mp3_44100_32",
|
||||
"mp3_44100_64",
|
||||
"mp3_44100_96",
|
||||
"mp3_44100_192",
|
||||
"pcm_8000",
|
||||
"pcm_16000",
|
||||
"pcm_22050",
|
||||
"pcm_24000",
|
||||
"pcm_44100",
|
||||
"pcm_48000",
|
||||
"ulaw_8000",
|
||||
"alaw_8000",
|
||||
"opus_48000_32",
|
||||
"opus_48000_64",
|
||||
"opus_48000_96",
|
||||
"opus_48000_128",
|
||||
"opus_48000_192"
|
||||
],),
|
||||
"voice_id": ("STRING", {"multiline": False, "default": "oPM3trUCF4e0vTcsrMQr"}),
|
||||
"stability": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
"similarity_boost": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
"speed": ("FLOAT", {"default": 1.0, "min": 0.25, "max": 2.0, "step": 0.01}),
|
||||
"style": ("FLOAT", {"default": 0.50, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
"use_speaker_boost": ("BOOLEAN", {"default": True}),
|
||||
"seed": ("INT", {"default": 40, "min": 0, "max": 4294967294}),
|
||||
"api_key": ("STRING", {"multiline": False, "default": ""}),
|
||||
},
|
||||
"optional": {
|
||||
"previous_text": ("STRING", {"multiline": True, "default": ""}),
|
||||
"next_text": ("STRING", {"multiline": True, "default": ""}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("AUDIO",)
|
||||
RETURN_NAMES = ("audio",)
|
||||
FUNCTION = "generate_speech"
|
||||
CATEGORY = "audio/generation"
|
||||
|
||||
def generate_speech(self, text, api_key, voice_id, model_id, output_format,
|
||||
stability, similarity_boost, speed, style, use_speaker_boost, seed,
|
||||
previous_text="", next_text=""):
|
||||
|
||||
if not text.strip():
|
||||
raise ValueError("Text input cannot be empty.")
|
||||
|
||||
key = api_key.strip() or os.environ.get("XI_API_KEY")
|
||||
if not key:
|
||||
raise ValueError("No API key provided.")
|
||||
|
||||
url = f"https://api.elevenlabs.io/v1/text-to-speech/{voice_id}"
|
||||
|
||||
headers = {
|
||||
"xi-api-key": key,
|
||||
"Content-Type": "application/json"
|
||||
}
|
||||
|
||||
if model_id == "eleven_v3":
|
||||
allowed_stabilities = [0.0, 0.5, 1.0]
|
||||
original_stability = stability
|
||||
stability = min(allowed_stabilities, key=lambda x: abs(x - original_stability))
|
||||
if stability != original_stability:
|
||||
print(f"For 'eleven_v3' model, stability value must be one of: [0.0, 0.5, 1.0] (0.0 = Creative, 0.5 = Natural, 1.0 = Robust). Rounding stability to {stability}.")
|
||||
|
||||
voice_settings = {
|
||||
"stability": stability,
|
||||
"similarity_boost": similarity_boost,
|
||||
"speed": speed,
|
||||
"style": style,
|
||||
"use_speaker_boost": use_speaker_boost
|
||||
}
|
||||
|
||||
data = {
|
||||
"text": text,
|
||||
"voice_settings": voice_settings,
|
||||
"model_id": model_id,
|
||||
"seed": seed,
|
||||
"output_format": output_format
|
||||
}
|
||||
|
||||
if model_id == "eleven_v3":
|
||||
if previous_text.strip() or next_text.strip():
|
||||
print("Providing previous_text or next_text is not yet supported with the 'eleven_v3' model. Ignoring these inputs.")
|
||||
else:
|
||||
if previous_text.strip():
|
||||
data["previous_text"] = previous_text
|
||||
if next_text.strip():
|
||||
data["next_text"] = next_text
|
||||
|
||||
response = requests.post(url, json=data, headers=headers)
|
||||
|
||||
if response.status_code != 200:
|
||||
raise Exception(f"ElevenLabs API Error: {response.status_code}, {response.text}")
|
||||
|
||||
# Decode audio from memory
|
||||
audio_buffer = io.BytesIO(response.content)
|
||||
waveform, sample_rate = torchaudio.load(audio_buffer)
|
||||
|
||||
# Return in ComfyUI audio format
|
||||
return ({"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate},)
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(cls, **kwargs):
|
||||
return f"{kwargs.get('text', '')}-{kwargs.get('voice_id', '')}-{kwargs.get('seed', 40)}"
|
||||
|
||||
NODE_CLASS_MAPPINGS = {"ElevenLabsTTSNode": ElevenLabsTTSNode}
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {"ElevenLabsTTSNode": "ElevenLabs TTS"}
|
||||
@@ -1,4 +1,3 @@
|
||||
import base64
|
||||
import os
|
||||
import io
|
||||
import numpy as np
|
||||
|
||||
+7
-7
@@ -10,15 +10,15 @@ class GeminiTTSNode:
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"text": ("STRING", {"multiline": True, "default": ""}),
|
||||
"api_key": ("STRING", {"multiline": False, "default": ""}),
|
||||
"model": ("STRING", {"default": "gemini-2.5-flash-preview-tts", "multiline": False}),
|
||||
"voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],),
|
||||
"seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}),
|
||||
"temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
"text": ("STRING", {"multiline": True, "default": ""}),
|
||||
"api_key": ("STRING", {"multiline": False, "default": ""}),
|
||||
"model": (["gemini-2.5-flash-preview-tts", "gemini-2.5-pro-preview-tts"],),
|
||||
"voice_id": (["Zephyr", "Puck", "Charon", "Kore", "Fenrir", "Leda", "Orus", "Aoede", "Callirrhoe", "Autonoe", "Enceladus", "Iapetus", "Umbriel", "Algieba", "Despina", "Erinome", "Achernar", "Laomedeia", "Rasalgethi", "Algenib", "Achird", "Pulcherrima", "Gacrux", "Schedar", "Alnilam", "Sulafat", "Sadaltager", "Sadachbia", "Vindemiatrix", "Zubenelgenubi"],),
|
||||
"seed": ("INT", {"default": 69, "min": -1, "max": 2147483646, "step": 1}),
|
||||
"temperature": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}),
|
||||
},
|
||||
"optional": {
|
||||
"system_prompt": ("STRING", {"multiline": True, "default": ""}),
|
||||
"system_prompt": ("STRING", {"multiline": True, "default": ""}),
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
import requests
|
||||
import base64
|
||||
import json
|
||||
import torch
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
Reference in New Issue
Block a user