diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..90bb4b8 --- /dev/null +++ b/.gitignore @@ -0,0 +1,39 @@ +# Python cache files +__pycache__/ +*.py[cod] +*$py.class +*.so +.Python +env/ +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +*.egg-info/ +.installed.cfg +*.egg + +# Virtual environments +venv/ +ENV/ +env/ + +# Generated audio files +outputs/*.wav + +# IDE files +.idea/ +.vscode/ +*.swp +*.swo + +# OS specific files +.DS_Store +Thumbs.db \ No newline at end of file diff --git a/README.md b/README.md index 6d1f786..eadf04c 100644 --- a/README.md +++ b/README.md @@ -1 +1,114 @@ -# ComfyUI_Fill-ChatterBox \ No newline at end of file +# ComfyUI_Fill-ChatterBox + +A custom node extension for ComfyUI that adds advanced text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library. + +## Features + +- **Text-to-Speech (TTS)**: Convert text to natural-sounding speech + - Adjustable emotion intensity, pace, and randomness + - Voice cloning from audio prompts + - GPU acceleration with CPU fallback + +- **Voice Conversion (VC)**: Transform voice characteristics between audio samples + - Preserve content while applying target voice style + - High-quality voice transformation + +- **ComfyUI Integration**: + - Custom styled nodes for easy identification (purple background with teal text) + - Compatible with ComfyUI's workflow system + - Connect with other audio and visual nodes + +## Installation + +### Important Note on Dependencies + +This extension uses ComfyUI's existing PyTorch installation to avoid conflicts. The installation process requires special handling for the chatterbox-tts package. + +### Installation Steps + +1. Clone this repository into your ComfyUI custom_nodes directory: + ```bash + cd /path/to/ComfyUI/custom_nodes + git clone https://github.com/yourusername/ComfyUI_Fill-ChatterBox.git + ``` + +2. Install the base dependencies: + ```bash + pip install -r ComfyUI_Fill-ChatterBox/requirements.txt + ``` + +3. **IMPORTANT**: Install chatterbox-tts WITHOUT its dependencies: + ```bash + pip install chatterbox-tts --no-deps + ``` + + ⚠️ The `--no-deps` flag is crucial to prevent conflicts with ComfyUI's PyTorch installation! + +## Usage + +### Text-to-Speech Node (FL Chatterbox TTS) + +The TTS node converts text input to speech with various customization options: + +1. Add the "FL Chatterbox TTS" node to your workflow +2. Configure the parameters: + - **text**: The text to convert to speech (supports multiline) + - **exaggeration**: Controls emotion intensity (0.25-2.0) + - **cfg_weight**: Controls pace/classifier-free guidance (0.2-1.0) + - **temperature**: Controls randomness in generation (0.05-5.0) + - **audio_prompt** (optional): Reference voice for TTS voice cloning + - **use_cpu** (optional): Force CPU usage even if CUDA is available + +3. Connect the output to other audio nodes or save as output + +### Voice Conversion Node (FL Chatterbox VC) + +The VC node transforms the voice characteristics of input audio to match a target voice: + +1. Add the "FL Chatterbox VC" node to your workflow +2. Configure the inputs: + - **input_audio**: The audio to convert + - **target_voice**: The voice to match + - **use_cpu** (optional): Force CPU usage even if CUDA is available + +3. Connect the output to other audio nodes or save as output + +## Technical Details + +### GPU/CPU Handling + +Both nodes intelligently handle GPU/CPU selection: +- Automatically detect CUDA availability +- Allow forcing CPU usage via the use_cpu parameter +- Graceful fallback to CPU if CUDA errors occur + +### Temporary File Management + +The nodes use temporary files for audio processing: +- Creates temporary WAV files for audio processing +- Ensures proper cleanup even if errors occur +- Provides detailed status messages about file operations + +## Troubleshooting + +### CUDA Out of Memory Errors + +If you encounter CUDA out of memory errors: +1. Try enabling the "use_cpu" option in the node settings +2. The nodes will automatically fall back to CPU if CUDA errors occur + +### Installation Issues + +If you encounter issues with conflicting PyTorch versions: +1. Ensure you installed chatterbox-tts with the `--no-deps` flag +2. Try uninstalling and reinstalling with the correct flags: + ```bash + pip uninstall -y chatterbox-tts + pip install chatterbox-tts --no-deps + ``` + +## Requirements + +- ComfyUI installation +- Python 3.8+ +- CUDA-compatible GPU recommended (but not required) \ No newline at end of file diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..eb06173 --- /dev/null +++ b/__init__.py @@ -0,0 +1,4 @@ +from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS + +WEB_DIRECTORY = "./web" +__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"] \ No newline at end of file diff --git a/appearance.js b/appearance.js new file mode 100644 index 0000000..b3c0dfc --- /dev/null +++ b/appearance.js @@ -0,0 +1,15 @@ +import { app } from "../../scripts/app.js"; + +app.registerExtension({ + name: "Fill-ChatterBox.appearance", // Extension name + async nodeCreated(node) { + // Check if the node's comfyClass starts with "FL_" + if (node.comfyClass.startsWith("FL_")) { + // Apply styling + node.color = "#16727c"; + node.bgcolor = "#4F0074"; + + + } + } +}); \ No newline at end of file diff --git a/chatterbox_node.py b/chatterbox_node.py new file mode 100644 index 0000000..8660b10 --- /dev/null +++ b/chatterbox_node.py @@ -0,0 +1,284 @@ +import os +import torch +import torchaudio +import numpy as np +from pathlib import Path +from typing import Optional + +# Import directly from the chatterbox package +from chatterbox.tts import ChatterboxTTS +from chatterbox.vc import ChatterboxVC + +# Monkey patch torch.load to always use CPU if needed +original_torch_load = torch.load +def patched_torch_load(*args, **kwargs): + if 'map_location' not in kwargs: + kwargs['map_location'] = torch.device('cpu') + return original_torch_load(*args, **kwargs) +torch.load = patched_torch_load + +class AudioNodeBase: + """Base class for audio nodes with common utilities.""" + + @staticmethod + def create_empty_tensor(audio, frame_rate, height, width, channels=None): + """Create an empty tensor with dimensions based on audio duration.""" + audio_duration = audio['waveform'].shape[-1] / audio['sample_rate'] + num_frames = int(audio_duration * frame_rate) + if channels is None: + return torch.zeros((num_frames, height, width), dtype=torch.float32) + else: + return torch.zeros((num_frames, height, width, channels), dtype=torch.float32) + +# Text-to-Speech node +class FL_ChatterboxTTSNode(AudioNodeBase): + """ + ComfyUI node for Chatterbox Text-to-Speech functionality. + """ + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "text": ("STRING", {"multiline": True, "default": "Hello, this is a test."}), + "exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}), + "cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}), + "temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}), + }, + "optional": { + "audio_prompt": ("AUDIO",), + "use_cpu": ("BOOLEAN", {"default": False}), + } + } + + RETURN_TYPES = ("AUDIO", "STRING") + RETURN_NAMES = ("audio", "message") + FUNCTION = "generate_speech" + CATEGORY = "ChatterBox" + + def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False): + """ + Generate speech from text. + + Args: + text: The text to convert to speech. + exaggeration: Controls emotion intensity (0.25-2.0). + cfg_weight: Controls pace/classifier-free guidance (0.2-1.0). + temperature: Controls randomness in generation (0.05-5.0). + audio_prompt: AUDIO object containing the reference voice for TTS voice cloning. + use_cpu: If True, forces CPU usage even if CUDA is available. + + Returns: + Tuple of (audio, message) + """ + # Determine device to use + device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "cpu") + if use_cpu: + message = "Using CPU for inference (CUDA disabled)" + else: + message = f"Using {device} for inference" + + # Create temporary files for any audio inputs + import tempfile + temp_files = [] + + # Create a temporary file for the audio prompt if provided + audio_prompt_path = None + if audio_prompt is not None: + try: + with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_prompt: + audio_prompt_path = temp_prompt.name + temp_files.append(audio_prompt_path) + + # Save the audio prompt to the temporary file + prompt_waveform = audio_prompt['waveform'].squeeze(0) + torchaudio.save(audio_prompt_path, prompt_waveform, audio_prompt['sample_rate']) + message += f"\nUsing provided audio prompt for voice cloning: {audio_prompt_path}" + + # Debug: Check if the file exists and has content + if os.path.exists(audio_prompt_path): + file_size = os.path.getsize(audio_prompt_path) + message += f"\nAudio prompt file created successfully: {file_size} bytes" + else: + message += f"\nWarning: Audio prompt file was not created properly" + except Exception as e: + message += f"\nError creating audio prompt file: {str(e)}" + audio_prompt_path = None + + try: + # Load the TTS model + message += f"\nLoading TTS model on {device}..." + tts_model = ChatterboxTTS.from_pretrained(device=device) + + # Generate speech + message += f"\nGenerating speech for: {text[:50]}..." if len(text) > 50 else f"\nGenerating speech for: {text}" + if audio_prompt_path: + message += f"\nUsing audio prompt: {audio_prompt_path}" + + wav = tts_model.generate( + text=text, + audio_prompt_path=audio_prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature, + ) + + except RuntimeError as e: + if "CUDA" in str(e) and device != "cpu": + message += "\nCUDA error detected during TTS. Falling back to CPU..." + # Try again with CPU + device = "cpu" + tts_model = ChatterboxTTS.from_pretrained(device=device) + wav = tts_model.generate( + text=text, + audio_prompt_path=audio_prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature, + ) + else: + # Re-raise if it's not a CUDA error or we're already on CPU + message += f"\nError during TTS: {str(e)}" + # Return empty audio data + empty_audio = {"waveform": torch.zeros((1, 2, 1)), "sample_rate": 16000} + # Clean up any temporary files + for temp_file in temp_files: + if os.path.exists(temp_file): + os.unlink(temp_file) + return (empty_audio, message) + finally: + # Clean up all temporary files + for temp_file in temp_files: + if os.path.exists(temp_file): + os.unlink(temp_file) + + # Create audio data structure for the output + audio_data = { + "waveform": wav.unsqueeze(0), # Add batch dimension + "sample_rate": tts_model.sr + } + + message += f"\nSpeech generated successfully" + + return (audio_data, message) + +# Voice Conversion node +class FL_ChatterboxVCNode(AudioNodeBase): + """ + ComfyUI node for Chatterbox Voice Conversion functionality. + """ + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "input_audio": ("AUDIO",), + "target_voice": ("AUDIO",), + }, + "optional": { + "use_cpu": ("BOOLEAN", {"default": False}), + } + } + + RETURN_TYPES = ("AUDIO", "STRING") + RETURN_NAMES = ("audio", "message") + FUNCTION = "convert_voice" + CATEGORY = "ChatterBox" + + def convert_voice(self, input_audio, target_voice, use_cpu=False): + """ + Convert the voice in an audio file to match a target voice. + + Args: + input_audio: AUDIO object containing the audio to convert. + target_voice: AUDIO object containing the target voice. + use_cpu: If True, forces CPU usage even if CUDA is available. + + Returns: + Tuple of (audio, message) + """ + # Determine device to use + device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "cpu") + if use_cpu: + message = "Using CPU for inference (CUDA disabled)" + else: + message = f"Using {device} for inference" + + # Create temporary files for the audio inputs + import tempfile + temp_files = [] + + # Create a temporary file for the input audio + with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_input: + input_audio_path = temp_input.name + temp_files.append(input_audio_path) + + # Save the input audio to the temporary file + input_waveform = input_audio['waveform'].squeeze(0) + torchaudio.save(input_audio_path, input_waveform, input_audio['sample_rate']) + + # Create a temporary file for the target voice + with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_target: + target_voice_path = temp_target.name + temp_files.append(target_voice_path) + + # Save the target voice to the temporary file + target_waveform = target_voice['waveform'].squeeze(0) + torchaudio.save(target_voice_path, target_waveform, target_voice['sample_rate']) + + try: + # Load the VC model + message += f"\nLoading VC model on {device}..." + vc_model = ChatterboxVC.from_pretrained(device=device) + + # Convert voice + message += f"\nConverting voice to match target voice" + + converted_wav = vc_model.generate( + audio=input_audio_path, + target_voice_path=target_voice_path, + ) + + except RuntimeError as e: + if "CUDA" in str(e) and device != "cpu": + message += "\nCUDA error detected during VC. Falling back to CPU..." + # Try again with CPU + device = "cpu" + vc_model = ChatterboxVC.from_pretrained(device=device) + converted_wav = vc_model.generate( + audio=input_audio_path, + target_voice_path=target_voice_path, + ) + else: + # Re-raise if it's not a CUDA error or we're already on CPU + message += f"\nError during VC: {str(e)}" + # Return the original audio + message += f"\nError: {str(e)}" + return (input_audio, message) + finally: + # Clean up all temporary files + for temp_file in temp_files: + if os.path.exists(temp_file): + os.unlink(temp_file) + + # Create audio data structure for the output + audio_data = { + "waveform": converted_wav.unsqueeze(0), # Add batch dimension + "sample_rate": vc_model.sr + } + + message += f"\nVoice converted successfully" + + return (audio_data, message) + +# Node mappings for ComfyUI +NODE_CLASS_MAPPINGS = { + "FL_ChatterboxTTS": FL_ChatterboxTTSNode, + "FL_ChatterboxVC": FL_ChatterboxVCNode, +} + +# Display names for the nodes +NODE_DISPLAY_NAME_MAPPINGS = { + "FL_ChatterboxTTS": "FL Chatterbox TTS", + "FL_ChatterboxVC": "FL Chatterbox VC", +} \ No newline at end of file diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..79a7814 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,15 @@ +[project] +name = "comfyui_fill-chatterbox" +description = "Voice Clone and TTS model." +version = "1.0.0" +license = "LICENSE" +dependencies = ["diffusers", "librosa", "sounddevice", "glitch_this", "PyOpenGL", "glfw", "scipy>=1.13.1", "requests", "aiohttp", "moviepy", "matplotlib", "reportlab", "openai", "PyPDF2", "pdf2image", "PyMuPDF", "reportlab", "PyPDF2", "ollama", "kornia", "opencv-python", "gdown", "open_clip_torch", "google-genai"] + +[project.urls] +Repository = "https://github.com/filliptm/ComfyUI_Fill-Nodes" +# Used by Comfy Registry https://comfyregistry.org + +[tool.comfy] +PublisherId = "machinedelusions" +DisplayName = "ComfyUI_Fill-Nodes" +Icon = "" diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..af080c2 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,13 @@ +# Core dependencies from chatterbox-tts without torch/torchaudio +numpy +resampy +librosa +s3tokenizer +transformers +diffusers +resemble-perth +omegaconf +conformer + +# NOTE: chatterbox-tts must be installed separately with: +# pip install chatterbox-tts --no-deps