From b95c6851edef85a2cdd20004c1e724330da320bb Mon Sep 17 00:00:00 2001 From: Aryan185 Date: Sat, 18 Apr 2026 15:25:22 +0530 Subject: [PATCH] Added Groq Orpheus TTS node --- README.md | 24 +++++++++++++++----- __init__.py | 1 + groq_orpheus.py | 59 +++++++++++++++++++++++++++++++++++++++++++++++++ pyproject.toml | 2 +- 4 files changed, 80 insertions(+), 6 deletions(-) create mode 100644 groq_orpheus.py diff --git a/README.md b/README.md index afaa88b..5562590 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,7 @@ A collection of powerful custom nodes for ComfyUI that connect your local workfl * **GPT Image Edit:** OpenAI's `gpt-image-1` for prompt-based image editing and inpainting. Simply mask an area and describe the change you want to see. * **OpenAI LLM:** Access OpenAI's powerful language models (GPT-4, GPT-5, o1, etc.) for text generation and reasoning. * **Groq LLM:** Fast text generation using Groq-hosted models, with optional image input for supported vision models. +* **Groq Orpheus TTS:** Generate speech using Groq-hosted Orpheus text-to-speech models. * **OpenAI Text-to-Speech:** Generate high-quality speech using OpenAI's TTS models. * **Google Imagen Generator & Edit:** Create and edit images with Google's Imagen models, with support for Vertex AI. * **Nano Banana:** A creative image generation node using a specialized Gemini model. @@ -51,7 +52,7 @@ All nodes in this collection require API keys to function. * **FLUX Nodes (Replicate):** You will need a [Replicate API Token](https://replicate.com/account/api-tokens). * **Gemini, Imagen, Nano Banana, Gemini TTS, Gemini Diarization, and Veo (Gemini API) Nodes:** You will need a [Google AI Studio API Key](https://aistudio.google.com/app/api-keys). * **OpenAI Nodes (GPT Image Edit, OpenAI LLM, OpenAI TTS):** You will need an [OpenAI API Key](https://platform.openai.com/api-keys). -* **Groq LLM Node:** You will need a [Groq API Key](https://console.groq.com/keys). +* **Groq Nodes (LLM, Orpheus TTS):** You will need a [Groq API Key](https://console.groq.com/keys). * **ElevenLabs TTS Node:** You will need an [ElevenLabs API Key](https://elevenlabs.io/). * **Tripo Nodes (Text-to-3D, Image-to-3D):** You will need a [Tripo API Key](https://tripo3d.ai/). * **Vertex AI Nodes (Imagen Edit, Veo Vertex AI):** You will need a Google Cloud Project ID, a service account with appropriate permissions, and the location for the resources. @@ -146,6 +147,14 @@ Access Groq-hosted language models for fast text generation, with optional image * **Key Inputs:** `max_completion_tokens`, `reasoning_effort`, `system_instruction`, `image` (optional) * **Output:** `response` (text) +### Groq Orpheus TTS + +Generate speech using Groq-hosted Orpheus text-to-speech models. + +* **Category:** `audio/generation` +* **Key Inputs:** `text`, `voice`, `speed` +* **Output:** `audio` + ### OpenAI Text-to-Speech Generate speech using OpenAI's TTS models. @@ -229,8 +238,13 @@ Convert images into detailed 3D models using Tripo AI's advanced image-to-model * **Output:** `glb` (3D model file) -## Acknowledgements +## Acknowledgements -* The [ComfyUI](https://github.com/comfyanonymous/ComfyUI) team for creating such a flexible and powerful platform. -* [Google](https://deepmind.google/technologies/gemini/), [OpenAI](https://openai.com/), and [Black Forest Labs](https://www.blackforestlabs.ai/) for developing these incredible models. -* [Replicate](https://replicate.com/) for providing easy API access to a wide range of models. \ No newline at end of file +* The [ComfyUI](https://github.com/comfyanonymous/ComfyUI) team for building the workflow platform that makes these nodes possible. +* [Google](https://deepmind.google/technologies/gemini/) for Gemini, Imagen, and Veo. +* [OpenAI](https://openai.com/) for GPT models and image and speech APIs. +* [Groq](https://groq.com/) for fast inference and hosted text-to-speech support. +* [xAI](https://x.ai/) for Grok image generation and editing models. +* [Black Forest Labs](https://www.blackforestlabs.ai/) and [Replicate](https://replicate.com/) for FLUX model access. +* [ElevenLabs](https://elevenlabs.io/) for speech synthesis. +* [Tripo AI](https://www.tripo3d.ai/) for 3D generation APIs. \ No newline at end of file diff --git a/__init__.py b/__init__.py index 9ba7636..2c22660 100644 --- a/__init__.py +++ b/__init__.py @@ -25,6 +25,7 @@ modules = [ "tripoImageToModel", "grok", "groq_node", + "groq_orpheus", ] NODE_CLASS_MAPPINGS = {} diff --git a/groq_orpheus.py b/groq_orpheus.py new file mode 100644 index 0000000..eff053e --- /dev/null +++ b/groq_orpheus.py @@ -0,0 +1,59 @@ +import os +import io +import torch +import soundfile as sf +from openai import OpenAI + +class GroqOrpheusTTSNode: + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "text": ("STRING", {"multiline": True, "default": ""}), + "model": (["canopylabs/orpheus-v1-english", "canopylabs/orpheus-arabic-saudi"], {"default": "canopylabs/orpheus-v1-english"}), + "voice": (["autumn", "diana", "hannah", "austin", "daniel", "troy", "abdullah", "fahad", "sultan", "lulwa", "noura", "aisha"], {"default": "troy"}), + "speed": ("FLOAT", {"default": 1.0, "min": 0.5, "max": 5.0, "step": 0.1}), + "seed": ("INT", {"default": 42, "min": 0, "max": 2147483646, "step": 1}), + "api_key": ("STRING", {"multiline": False, "default": "", "tooltip": "Directly put Groq API key or .env variable name (GROQ_API_KEY)"}), + } + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio",) + FUNCTION = "generate_speech" + CATEGORY = "audio/generation" + + def generate_speech(self, text, model, voice, speed, seed, api_key): + + key = os.environ.get(api_key.strip(), api_key.strip()) or os.environ.get("GROQ_API_KEY") + if not key: + raise ValueError("No API key provided.") + + client = OpenAI(api_key=key, base_url="https://api.groq.com/openai/v1") + + response = client.audio.speech.create( + model=model, + voice=voice, + input=text, + response_format="wav", + speed=speed, + ) + + waveform, sample_rate = sf.read(io.BytesIO(response.content), dtype='float32') + waveform = torch.from_numpy(waveform) + + if waveform.dim() == 1: + waveform = waveform.unsqueeze(0) + else: + waveform = waveform.t() + + return ({"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate},) + + @classmethod + def IS_CHANGED(cls, seed, **kwargs): + return seed + + +NODE_CLASS_MAPPINGS = {"GroqOrpheusTTSNode": GroqOrpheusTTSNode} +NODE_DISPLAY_NAME_MAPPINGS = {"GroqOrpheusTTSNode": "Groq Orpheus TTS"} \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml index 4e506a0..83e04dd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "externalapi-helpers" description = "Various ComfyUI nodes for Gemini, Replicate and OpenAI" -version = "1.1.0" +version = "1.1.1" license = {file = "LICENSE"} # classifiers = [ # # For OS-independent nodes (works on all operating systems)