commit 71b2bfb9c2688a04716c4903e370a2dd2c469a6f Author: NumZ Date: Wed Apr 16 14:33:25 2025 +0200 Initial commit diff --git a/README.md b/README.md new file mode 100644 index 0000000..a2078e0 --- /dev/null +++ b/README.md @@ -0,0 +1,147 @@ +# ComfyUI-Orpheus Node + +2 custom nodes for ComfyUI that enables text-to-speech generation using the GGUF [Orpheus](https://github.com/canopyai/Orpheus-TTS) model with emotional speech capabilities. + + + +## Features + +- High-quality text-to-speech synthesis +- Multiple voice options (24 different voices, depend language used) in **English, French, Spanish, Italian, Chinese, Korean, German, Hindi** +- Emotional speech capabilities +- Seamless integration with ComfyUI workflow + +## Available Voices + +- **English Voices**: + + supported tags : chuckle, cough, gasp, groan, laugh, sigh, sniffle, yawn + + - `tara` - Female voice + - `leah` - Female voice + - `jess` - Female voice + - `leo` - Male voice + - `dan` - Male voice + - `mia` - Female voice + - `zac` - Male voice + - `zoe` - Female voice + +- **French Voices**: + + supported tags : chuckle, cough, gasp, groan, laugh, sigh, sniffle, whimper, yawn + + - `pierre` - Male voice + - `amelie` - Female voice + - `marie` - Female voice (doesn't works well) + +- **German Voices**: + + supported tags : chuckle, cough, gasp, groan, laugh, sigh, sniffle, yawn + + - `jana` - Female voice + - `thomas` - Male voice + - `max` - Male voice + +- **Korean Voices**: + + supported tags : 한숨, 헐, 헛기침, 훌쩍, 하품, 낄낄, 신음, 작은 웃음, 기침, 으르렁 + + - `유나` - ?^^ + - `준서` - ?^^ + +- **Chinese Voices**: + + supported tags : 嬉笑, 轻笑, 呻吟, 大笑, 咳嗽, 抽鼻子, 咳 + + - `长乐` - ?^^ + - `白芷` - ?^^ + +- **Hindi**: + + supported tags : unknow + + - `ऋतिका` - ? ^^ + +- **Spanish Voices**: + + supported tags : groan, chuckle, gasp, resoplido, laugh, yawn, cough + + - `javi` - Male voice + - `sergio` - Male voice + - `maria` - Female voice + +- **Italian Voices**: + + supported tags : sigh, laugh, cough, sniffle, groan, yawn, gemito, gasp + + - `pietro` - Male voice + - `giulia` - Female voice + - `carlo` - Male voice + +## Requirements + +- Last ComfyUI version with python 3.12.9 (may be works with older versions but I haven't test it) + +## Installation + +1. Clone this repository into your ComfyUI custom nodes directory: + +```bash +cd ComfyUI/custom_nodes +git clone https://github.com/your-repo/ComfyUI-Orpheus.git +``` + +2. Install the required dependencies: + +load venv and : + +```bash +pip install -r ComfyUI-Orpheus/requirements.txt +``` + +use python_embeded : + +```bash +python_embeded\python.exe -m pip install -r ComfyUI-Orpheus/requirements.txt +``` + +3. Download the required [GGUF model](https://huggingface.co/freddyaboulton) from [FreddyAboulton](https://huggingface.co/freddyaboulton) huggingface page, and place it in your ComfyUI models directory under `models/unet/`. (Sorry I don't know where to find Italian one) + + + +4. **GPU Support** + +On windows, default installation of llama-cpp-python doesn't take GPU support. If you want GPU Support you need +to locate `nvcc.exe` folder and: + +```bash +set CMAKE_ARGS="-DGGML_CUDA=on" +set CUDA_CXX="YOUR_CUDA_DIR\v12.6.3\bin\nvcc.exe" +python_embeded\python.exe -m pip install llama-cpp-python[server] --upgrade --force-reinstall --no-cache-dir +``` + +Be patient, it takes time... + +## Usage + +1. In ComfyUI, locate the "Orpheus ⛓️" node in the node menu. + +2. Configure the node parameters: + + - `model_name`: Select your GGUF model + - `voice`: Choose from available voices + - `prompt`: Enter the text you want to convert to speech, you can add emotive tags :\, \, \, \, \, \, \, \ + +3. Connect the node outputs: + - `audio`: Contains the generated audio waveform and sample rate + +## Limitations + +- Maximum text length determined by MAX_TOKENS +- Processing speed depends on GPU capabilities +- Requires CUDA support for optimal performance + +## Credits + +- Original [Orpheus](https://github.com/canopyai/Orpheus-TTS) implementation +- [Freddy Aboulton](https://huggingface.co/freddyaboulton) diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..b76d4cd --- /dev/null +++ b/__init__.py @@ -0,0 +1,4 @@ +from .node.orpheus import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS + +__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] + diff --git a/doc/demo.png b/doc/demo.png new file mode 100644 index 0000000..afb0497 Binary files /dev/null and b/doc/demo.png differ diff --git a/doc/models.png b/doc/models.png new file mode 100644 index 0000000..7fde7a5 Binary files /dev/null and b/doc/models.png differ diff --git a/node/__pycache__/decoder.cpython-311.pyc b/node/__pycache__/decoder.cpython-311.pyc new file mode 100644 index 0000000..c20a3d0 Binary files /dev/null and b/node/__pycache__/decoder.cpython-311.pyc differ diff --git a/node/__pycache__/decoder.cpython-312.pyc b/node/__pycache__/decoder.cpython-312.pyc new file mode 100644 index 0000000..c7e73e2 Binary files /dev/null and b/node/__pycache__/decoder.cpython-312.pyc differ diff --git a/node/__pycache__/orpheus.cpython-311.pyc b/node/__pycache__/orpheus.cpython-311.pyc new file mode 100644 index 0000000..7942393 Binary files /dev/null and b/node/__pycache__/orpheus.cpython-311.pyc differ diff --git a/node/__pycache__/orpheus.cpython-312.pyc b/node/__pycache__/orpheus.cpython-312.pyc new file mode 100644 index 0000000..f0ab2e0 Binary files /dev/null and b/node/__pycache__/orpheus.cpython-312.pyc differ diff --git a/node/decoder.py b/node/decoder.py new file mode 100644 index 0000000..c2fe979 --- /dev/null +++ b/node/decoder.py @@ -0,0 +1,140 @@ +from snac import SNAC +import numpy as np +import torch +import asyncio +import threading +import queue + + +model = SNAC.from_pretrained("hubertsiuzdak/snac_24khz").eval() + +snac_device = "cuda" +model = model.to(snac_device) + + +def convert_to_audio(multiframe, count): + frames = [] + if len(multiframe) < 7: + return + + codes_0 = torch.tensor([], device=snac_device, dtype=torch.int32) + codes_1 = torch.tensor([], device=snac_device, dtype=torch.int32) + codes_2 = torch.tensor([], device=snac_device, dtype=torch.int32) + + num_frames = len(multiframe) // 7 + frame = multiframe[:num_frames*7] + + for j in range(num_frames): + i = 7*j + if codes_0.shape[0] == 0: + codes_0 = torch.tensor([frame[i]], device=snac_device, dtype=torch.int32) + else: + codes_0 = torch.cat([codes_0, torch.tensor([frame[i]], device=snac_device, dtype=torch.int32)]) + + if codes_1.shape[0] == 0: + + codes_1 = torch.tensor([frame[i+1]], device=snac_device, dtype=torch.int32) + codes_1 = torch.cat([codes_1, torch.tensor([frame[i+4]], device=snac_device, dtype=torch.int32)]) + else: + codes_1 = torch.cat([codes_1, torch.tensor([frame[i+1]], device=snac_device, dtype=torch.int32)]) + codes_1 = torch.cat([codes_1, torch.tensor([frame[i+4]], device=snac_device, dtype=torch.int32)]) + + if codes_2.shape[0] == 0: + codes_2 = torch.tensor([frame[i+2]], device=snac_device, dtype=torch.int32) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+3]], device=snac_device, dtype=torch.int32)]) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+5]], device=snac_device, dtype=torch.int32)]) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+6]], device=snac_device, dtype=torch.int32)]) + else: + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+2]], device=snac_device, dtype=torch.int32)]) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+3]], device=snac_device, dtype=torch.int32)]) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+5]], device=snac_device, dtype=torch.int32)]) + codes_2 = torch.cat([codes_2, torch.tensor([frame[i+6]], device=snac_device, dtype=torch.int32)]) + + codes = [codes_0.unsqueeze(0), codes_1.unsqueeze(0), codes_2.unsqueeze(0)] + # check that all tokens are between 0 and 4096 otherwise return * + if torch.any(codes[0] < 0) or torch.any(codes[0] > 4096) or torch.any(codes[1] < 0) or torch.any(codes[1] > 4096) or torch.any(codes[2] < 0) or torch.any(codes[2] > 4096): + return + + with torch.inference_mode(): + audio_hat = model.decode(codes) + + audio_slice = audio_hat[:, :, 2048:4096] + detached_audio = audio_slice.detach().cpu() + audio_np = detached_audio.numpy() + audio_int16 = (audio_np * 32767).astype(np.int16) + audio_bytes = audio_int16.tobytes() + return audio_bytes + +def turn_token_into_id(token_string, index): + # Strip whitespace + token_string = token_string.strip() + + # Find the last token in the string + last_token_start = token_string.rfind(""): + try: + number_str = last_token[14:-1] + return int(number_str) - 10 - ((index % 7) * 4096) + except ValueError: + return None + else: + return None + + +async def tokens_decoder(token_gen): + buffer = [] + count = 0 + async for token_sim in token_gen: + token = turn_token_into_id(token_sim, count) + if token is None: + pass + else: + if token > 0: + buffer.append(token) + count += 1 + + if count % 7 == 0 and count > 27: + buffer_to_proc = buffer[-28:] + audio_samples = convert_to_audio(buffer_to_proc, count) + if audio_samples is not None: + yield audio_samples + + +# ------------------ Synchronous Tokens Decoder Wrapper ------------------ # +def tokens_decoder_sync(syn_token_gen): + + audio_queue = queue.Queue() + + # Convert the synchronous token generator into an async generator. + async def async_token_gen(): + for token in syn_token_gen: + yield token + + async def async_producer(): + # tokens_decoder.tokens_decoder is assumed to be an async generator that processes tokens. + async for audio_chunk in tokens_decoder(async_token_gen()): + audio_queue.put(audio_chunk) + audio_queue.put(None) # Sentinel + + def run_async(): + asyncio.run(async_producer()) + + thread = threading.Thread(target=run_async) + thread.start() + + while True: + audio = audio_queue.get() + if audio is None: + break + yield audio + + thread.join() \ No newline at end of file diff --git a/node/orpheus.py b/node/orpheus.py new file mode 100644 index 0000000..f146cdc --- /dev/null +++ b/node/orpheus.py @@ -0,0 +1,290 @@ +import os +import time +import logging + +import wave +import folder_paths +import hashlib +import torchaudio +from .decoder import convert_to_audio as orpheus_convert_to_audio + +from llama_cpp import Llama + + +def update_folder_names_and_paths(key, targets=[]): + # check for existing key + base = folder_paths.folder_names_and_paths.get(key, ([], {})) + base = base[0] if isinstance(base[0], (list, set, tuple)) else [] + # find base key & add w/ fallback, sanity check + warning + target = next((x for x in targets if x in folder_paths.folder_names_and_paths), targets[0]) + orig, _ = folder_paths.folder_names_and_paths.get(target, ([], {})) + folder_paths.folder_names_and_paths[key] = (orig or base, {".gguf"}) + if base and base != orig: + logging.warning(f"Unknown file list already present on key {key}: {base}") + +# Add a custom keys for files ending in .gguf +update_folder_names_and_paths("unet_gguf", ["diffusion_models", "unet"]) + + +BOOLEAN = ("BOOLEAN", {"default": True}) +STRING = ("STRING", {"default": ""}) + +# Model parameters +MAX_TOKENS = 8192 +TEMPERATURE = 0.6 +TOP_P = 0.9 +REPETITION_PENALTY = 1.1 +SAMPLE_RATE = 24000 # SNAC model uses 24kHz + +# Available voices based on the Orpheus-TTS repository +AVAILABLE_VOICES = ["tara", "leah", "jess", "leo", "dan", "mia", "zac", "zoe", "pierre", "amelie", "marie","jana", "thomas", "max", "유나", "준서", "长乐", "白芷" "javi", "sergio", "maria", "pietro", "giulia", "carlo"] +DEFAULT_VOICE = "pierre" # Best voice according to documentation + +CUSTOM_TOKEN_PREFIX = ""): + #print(f"Last token: {last_token}") + try: + number_str = last_token[14:-1] + # print(f"Number string: {number_str}") + token_id = int(number_str) - 10 - ((index % 7) * 4096) + # print(f"Token ID: {token_id}") + return token_id + except ValueError: + return None + else: + return None + +def convert_to_audio(multiframe, count): + """Convert token frames to audio.""" + # Import here to avoid circular imports + + return orpheus_convert_to_audio(multiframe, count) + +def my_tokens_decoder(token_gen): + """Asynchronous token decoder that converts token stream to audio stream.""" + buffer = [] + count = 0 + audio_samples_buffer = [] + #print(token_gen) + #tgen = extract_custom_tokens(token_gen) + for token_text in token_gen: + #print("Token text:", token_text) + token = turn_token_into_id(token_text, count) + #print("Token ID:", token) + if token is not None and token > 0: + buffer.append(token) + count += 1 + + # Convert to audio when we have enough tokens + if count % 7 == 0 and count > 27: + buffer_to_proc = buffer[-28:] + audio_samples = convert_to_audio(buffer_to_proc, count) + if audio_samples is not None: + audio_samples_buffer.append(audio_samples) + return audio_samples_buffer + +def extract_custom_tokens(token_string): + """ + Extract all custom tokens from a string and return them as an array. + Example: "" -> ["", ""] + """ + #print(token_string) + tokens = [] + i = 0 + while i < len(token_string): + # Find the start of a custom token + start_pos = token_string.find(CUSTOM_TOKEN_PREFIX, i) + if start_pos == -1: + break + + # Find the end of this token + end_pos = token_string.find(">", start_pos) + if end_pos == -1: + break + + # Extract the complete token + token = token_string[start_pos:end_pos+1] + tokens.append(token) + + # Move past this token + i = end_pos + 1 + + return tokens + +def tokens_decoder_sync(syn_token_gen, output_file=None): + wav_file = None + if output_file: + # Create directory if it doesn't exist + os.makedirs(os.path.dirname(os.path.abspath(output_file)), exist_ok=True) + wav_file = wave.open(output_file, "wb") + wav_file.setnchannels(1) + wav_file.setsampwidth(2) + wav_file.setframerate(SAMPLE_RATE) + + def my_producer(syn_token_gen): + return my_tokens_decoder(extract_custom_tokens(syn_token_gen)) + + tokens = my_producer(syn_token_gen) + + for audio_chunk in tokens: + wav_file.writeframes(audio_chunk) + + if wav_file: + wav_file.close() + + +def load_model(model_path): + llm = Llama( + model_path=model_path, + n_ctx=4096, # Context length to use + n_threads=12, # Number of CPU threads to use + n_gpu_layers=32, + verbose=False + ) + return llm + + +class orpheus: + def __init__(self): + pass + + @classmethod + def INPUT_TYPES(cls): + unet_names = [x for x in folder_paths.get_filename_list("unet_gguf")] + #print(unet_names) + return { + "required": { + "model_name": (unet_names, ), + "voice": (AVAILABLE_VOICES,), + "prompt": ("STRING", {"default": "Hello, I am Orpheus, an AI assistant with emotional speech capabilities.","multiline": True}) + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio",) + FUNCTION = "execute" + CATEGORY = "Orpheus ⛓️" + + @classmethod + def IS_CHANGED(s, model_name, voice,prompt, **kwargs): + m = hashlib.sha256() + m.update(model_name.encode() + voice.encode() + prompt.encode()) + return m.digest().hex() + + def execute(self, model_name, voice, prompt, **kwargs): + timestamp = time.strftime("%Y%m%d_%H%M%S") + output_file = os.path.join(folder_paths.get_output_directory(), f"orpheus_{voice}_{timestamp}.wav") + model_path = folder_paths.get_full_path("unet_gguf", model_name) + #print(model_path) + llm = load_model(model_path) + prompt = format_prompt(prompt, voice=voice) + generation_kwargs = { + "max_tokens": MAX_TOKENS, + "temperature": TEMPERATURE, + "top_p": TOP_P, + "repeat_penalty": REPETITION_PENALTY + } + + res = llm(prompt, **generation_kwargs) # Res is a dictionary + tokens_decoder_sync(res['choices'][0]['text'], output_file) + + # open file as a tensor + waveform, sample_rate = torchaudio.load(output_file) + audio = {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate} + return (audio, ) + + +class orpheusAdvanced: + def __init__(self): + pass + + @classmethod + def INPUT_TYPES(cls): + unet_names = [x for x in folder_paths.get_filename_list("unet_gguf")] + #print(unet_names) + return { + "required": { + "model_name": (unet_names, ), + "voice": (AVAILABLE_VOICES,), + "prompt": ("STRING", {"default": "Hello, I am Orpheus, an AI assistant with emotional speech capabilities.","multiline": True}), + "max_tokens" :("INT", {"default": 8192, "min": 4096, "max": 131072, "step": 1}), + "temperature" :("FLOAT", {"default": 0.6, "min": 0.0, "max": 1.0, "step": 0.01}), + "top_p" :("FLOAT", {"default": 0.9, "min": 0.0, "max": 1.0, "step": 0.01}), + "repeat_penalty" :("FLOAT", {"default": 1.1, "min": 0.0, "max": 1.5, "step": 0.01}), + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio",) + FUNCTION = "execute" + CATEGORY = "Orpheus ⛓️" + + @classmethod + def IS_CHANGED(s, model_name, voice,prompt, max_tokens, temperature, top_p, repeat_penalty, **kwargs): + m = hashlib.sha256() + m.update(model_name.encode() + voice.encode() + prompt.encode() + str(max_tokens).encode() + str(temperature).encode() + str(top_p).encode() + str(repeat_penalty).encode()) + return m.digest().hex() + + def execute(self, model_name, voice, prompt, max_tokens, temperature, top_p, repeat_penalty, **kwargs): + timestamp = time.strftime("%Y%m%d_%H%M%S") + output_file = os.path.join(folder_paths.get_output_directory(), f"orpheus_{voice}_{timestamp}.wav") + model_path = folder_paths.get_full_path("unet_gguf", model_name) + #print(model_path) + llm = load_model(model_path) + prompt = format_prompt(prompt, voice=voice) + generation_kwargs = { + "max_tokens": max_tokens, + "temperature": temperature, + "top_p": top_p, + "repeat_penalty": repeat_penalty + } + + res = llm(prompt, **generation_kwargs) # Res is a dictionary + tokens_decoder_sync(res['choices'][0]['text'], output_file) + + # open file as a tensor + waveform, sample_rate = torchaudio.load(output_file) + audio = {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate} + return (audio, ) + +NODE_CLASS_MAPPINGS = { + "orpheus": orpheus, + "orpheusAdvanced": orpheusAdvanced, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "orpheus": "Orpheus", + "orpheusAdvanced": "Orpheus Advanced", +} diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..f85fad2 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,2 @@ +sounddevice +snac \ No newline at end of file