This commit is contained in:
Fill
2025-05-29 21:54:50 +09:00
committed by GitHub
parent 2619b0945f
commit c8f00dc856
7 changed files with 484 additions and 1 deletions
+39
View File
@@ -0,0 +1,39 @@
# Python cache files
__pycache__/
*.py[cod]
*$py.class
*.so
.Python
env/
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
*.egg-info/
.installed.cfg
*.egg
# Virtual environments
venv/
ENV/
env/
# Generated audio files
outputs/*.wav
# IDE files
.idea/
.vscode/
*.swp
*.swo
# OS specific files
.DS_Store
Thumbs.db
+114 -1
View File
@@ -1 +1,114 @@
# ComfyUI_Fill-ChatterBox
# ComfyUI_Fill-ChatterBox
A custom node extension for ComfyUI that adds advanced text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library.
## Features
- **Text-to-Speech (TTS)**: Convert text to natural-sounding speech
- Adjustable emotion intensity, pace, and randomness
- Voice cloning from audio prompts
- GPU acceleration with CPU fallback
- **Voice Conversion (VC)**: Transform voice characteristics between audio samples
- Preserve content while applying target voice style
- High-quality voice transformation
- **ComfyUI Integration**:
- Custom styled nodes for easy identification (purple background with teal text)
- Compatible with ComfyUI's workflow system
- Connect with other audio and visual nodes
## Installation
### Important Note on Dependencies
This extension uses ComfyUI's existing PyTorch installation to avoid conflicts. The installation process requires special handling for the chatterbox-tts package.
### Installation Steps
1. Clone this repository into your ComfyUI custom_nodes directory:
```bash
cd /path/to/ComfyUI/custom_nodes
git clone https://github.com/yourusername/ComfyUI_Fill-ChatterBox.git
```
2. Install the base dependencies:
```bash
pip install -r ComfyUI_Fill-ChatterBox/requirements.txt
```
3. **IMPORTANT**: Install chatterbox-tts WITHOUT its dependencies:
```bash
pip install chatterbox-tts --no-deps
```
⚠️ The `--no-deps` flag is crucial to prevent conflicts with ComfyUI's PyTorch installation!
## Usage
### Text-to-Speech Node (FL Chatterbox TTS)
The TTS node converts text input to speech with various customization options:
1. Add the "FL Chatterbox TTS" node to your workflow
2. Configure the parameters:
- **text**: The text to convert to speech (supports multiline)
- **exaggeration**: Controls emotion intensity (0.25-2.0)
- **cfg_weight**: Controls pace/classifier-free guidance (0.2-1.0)
- **temperature**: Controls randomness in generation (0.05-5.0)
- **audio_prompt** (optional): Reference voice for TTS voice cloning
- **use_cpu** (optional): Force CPU usage even if CUDA is available
3. Connect the output to other audio nodes or save as output
### Voice Conversion Node (FL Chatterbox VC)
The VC node transforms the voice characteristics of input audio to match a target voice:
1. Add the "FL Chatterbox VC" node to your workflow
2. Configure the inputs:
- **input_audio**: The audio to convert
- **target_voice**: The voice to match
- **use_cpu** (optional): Force CPU usage even if CUDA is available
3. Connect the output to other audio nodes or save as output
## Technical Details
### GPU/CPU Handling
Both nodes intelligently handle GPU/CPU selection:
- Automatically detect CUDA availability
- Allow forcing CPU usage via the use_cpu parameter
- Graceful fallback to CPU if CUDA errors occur
### Temporary File Management
The nodes use temporary files for audio processing:
- Creates temporary WAV files for audio processing
- Ensures proper cleanup even if errors occur
- Provides detailed status messages about file operations
## Troubleshooting
### CUDA Out of Memory Errors
If you encounter CUDA out of memory errors:
1. Try enabling the "use_cpu" option in the node settings
2. The nodes will automatically fall back to CPU if CUDA errors occur
### Installation Issues
If you encounter issues with conflicting PyTorch versions:
1. Ensure you installed chatterbox-tts with the `--no-deps` flag
2. Try uninstalling and reinstalling with the correct flags:
```bash
pip uninstall -y chatterbox-tts
pip install chatterbox-tts --no-deps
```
## Requirements
- ComfyUI installation
- Python 3.8+
- CUDA-compatible GPU recommended (but not required)
+4
View File
@@ -0,0 +1,4 @@
from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
WEB_DIRECTORY = "./web"
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]
+15
View File
@@ -0,0 +1,15 @@
import { app } from "../../scripts/app.js";
app.registerExtension({
name: "Fill-ChatterBox.appearance", // Extension name
async nodeCreated(node) {
// Check if the node's comfyClass starts with "FL_"
if (node.comfyClass.startsWith("FL_")) {
// Apply styling
node.color = "#16727c";
node.bgcolor = "#4F0074";
}
}
});
+284
View File
@@ -0,0 +1,284 @@
import os
import torch
import torchaudio
import numpy as np
from pathlib import Path
from typing import Optional
# Import directly from the chatterbox package
from chatterbox.tts import ChatterboxTTS
from chatterbox.vc import ChatterboxVC
# Monkey patch torch.load to always use CPU if needed
original_torch_load = torch.load
def patched_torch_load(*args, **kwargs):
if 'map_location' not in kwargs:
kwargs['map_location'] = torch.device('cpu')
return original_torch_load(*args, **kwargs)
torch.load = patched_torch_load
class AudioNodeBase:
"""Base class for audio nodes with common utilities."""
@staticmethod
def create_empty_tensor(audio, frame_rate, height, width, channels=None):
"""Create an empty tensor with dimensions based on audio duration."""
audio_duration = audio['waveform'].shape[-1] / audio['sample_rate']
num_frames = int(audio_duration * frame_rate)
if channels is None:
return torch.zeros((num_frames, height, width), dtype=torch.float32)
else:
return torch.zeros((num_frames, height, width, channels), dtype=torch.float32)
# Text-to-Speech node
class FL_ChatterboxTTSNode(AudioNodeBase):
"""
ComfyUI node for Chatterbox Text-to-Speech functionality.
"""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"text": ("STRING", {"multiline": True, "default": "Hello, this is a test."}),
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
},
"optional": {
"audio_prompt": ("AUDIO",),
"use_cpu": ("BOOLEAN", {"default": False}),
}
}
RETURN_TYPES = ("AUDIO", "STRING")
RETURN_NAMES = ("audio", "message")
FUNCTION = "generate_speech"
CATEGORY = "ChatterBox"
def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False):
"""
Generate speech from text.
Args:
text: The text to convert to speech.
exaggeration: Controls emotion intensity (0.25-2.0).
cfg_weight: Controls pace/classifier-free guidance (0.2-1.0).
temperature: Controls randomness in generation (0.05-5.0).
audio_prompt: AUDIO object containing the reference voice for TTS voice cloning.
use_cpu: If True, forces CPU usage even if CUDA is available.
Returns:
Tuple of (audio, message)
"""
# Determine device to use
device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "cpu")
if use_cpu:
message = "Using CPU for inference (CUDA disabled)"
else:
message = f"Using {device} for inference"
# Create temporary files for any audio inputs
import tempfile
temp_files = []
# Create a temporary file for the audio prompt if provided
audio_prompt_path = None
if audio_prompt is not None:
try:
with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_prompt:
audio_prompt_path = temp_prompt.name
temp_files.append(audio_prompt_path)
# Save the audio prompt to the temporary file
prompt_waveform = audio_prompt['waveform'].squeeze(0)
torchaudio.save(audio_prompt_path, prompt_waveform, audio_prompt['sample_rate'])
message += f"\nUsing provided audio prompt for voice cloning: {audio_prompt_path}"
# Debug: Check if the file exists and has content
if os.path.exists(audio_prompt_path):
file_size = os.path.getsize(audio_prompt_path)
message += f"\nAudio prompt file created successfully: {file_size} bytes"
else:
message += f"\nWarning: Audio prompt file was not created properly"
except Exception as e:
message += f"\nError creating audio prompt file: {str(e)}"
audio_prompt_path = None
try:
# Load the TTS model
message += f"\nLoading TTS model on {device}..."
tts_model = ChatterboxTTS.from_pretrained(device=device)
# Generate speech
message += f"\nGenerating speech for: {text[:50]}..." if len(text) > 50 else f"\nGenerating speech for: {text}"
if audio_prompt_path:
message += f"\nUsing audio prompt: {audio_prompt_path}"
wav = tts_model.generate(
text=text,
audio_prompt_path=audio_prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature,
)
except RuntimeError as e:
if "CUDA" in str(e) and device != "cpu":
message += "\nCUDA error detected during TTS. Falling back to CPU..."
# Try again with CPU
device = "cpu"
tts_model = ChatterboxTTS.from_pretrained(device=device)
wav = tts_model.generate(
text=text,
audio_prompt_path=audio_prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature,
)
else:
# Re-raise if it's not a CUDA error or we're already on CPU
message += f"\nError during TTS: {str(e)}"
# Return empty audio data
empty_audio = {"waveform": torch.zeros((1, 2, 1)), "sample_rate": 16000}
# Clean up any temporary files
for temp_file in temp_files:
if os.path.exists(temp_file):
os.unlink(temp_file)
return (empty_audio, message)
finally:
# Clean up all temporary files
for temp_file in temp_files:
if os.path.exists(temp_file):
os.unlink(temp_file)
# Create audio data structure for the output
audio_data = {
"waveform": wav.unsqueeze(0), # Add batch dimension
"sample_rate": tts_model.sr
}
message += f"\nSpeech generated successfully"
return (audio_data, message)
# Voice Conversion node
class FL_ChatterboxVCNode(AudioNodeBase):
"""
ComfyUI node for Chatterbox Voice Conversion functionality.
"""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"input_audio": ("AUDIO",),
"target_voice": ("AUDIO",),
},
"optional": {
"use_cpu": ("BOOLEAN", {"default": False}),
}
}
RETURN_TYPES = ("AUDIO", "STRING")
RETURN_NAMES = ("audio", "message")
FUNCTION = "convert_voice"
CATEGORY = "ChatterBox"
def convert_voice(self, input_audio, target_voice, use_cpu=False):
"""
Convert the voice in an audio file to match a target voice.
Args:
input_audio: AUDIO object containing the audio to convert.
target_voice: AUDIO object containing the target voice.
use_cpu: If True, forces CPU usage even if CUDA is available.
Returns:
Tuple of (audio, message)
"""
# Determine device to use
device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "cpu")
if use_cpu:
message = "Using CPU for inference (CUDA disabled)"
else:
message = f"Using {device} for inference"
# Create temporary files for the audio inputs
import tempfile
temp_files = []
# Create a temporary file for the input audio
with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_input:
input_audio_path = temp_input.name
temp_files.append(input_audio_path)
# Save the input audio to the temporary file
input_waveform = input_audio['waveform'].squeeze(0)
torchaudio.save(input_audio_path, input_waveform, input_audio['sample_rate'])
# Create a temporary file for the target voice
with tempfile.NamedTemporaryFile(suffix='.wav', delete=False) as temp_target:
target_voice_path = temp_target.name
temp_files.append(target_voice_path)
# Save the target voice to the temporary file
target_waveform = target_voice['waveform'].squeeze(0)
torchaudio.save(target_voice_path, target_waveform, target_voice['sample_rate'])
try:
# Load the VC model
message += f"\nLoading VC model on {device}..."
vc_model = ChatterboxVC.from_pretrained(device=device)
# Convert voice
message += f"\nConverting voice to match target voice"
converted_wav = vc_model.generate(
audio=input_audio_path,
target_voice_path=target_voice_path,
)
except RuntimeError as e:
if "CUDA" in str(e) and device != "cpu":
message += "\nCUDA error detected during VC. Falling back to CPU..."
# Try again with CPU
device = "cpu"
vc_model = ChatterboxVC.from_pretrained(device=device)
converted_wav = vc_model.generate(
audio=input_audio_path,
target_voice_path=target_voice_path,
)
else:
# Re-raise if it's not a CUDA error or we're already on CPU
message += f"\nError during VC: {str(e)}"
# Return the original audio
message += f"\nError: {str(e)}"
return (input_audio, message)
finally:
# Clean up all temporary files
for temp_file in temp_files:
if os.path.exists(temp_file):
os.unlink(temp_file)
# Create audio data structure for the output
audio_data = {
"waveform": converted_wav.unsqueeze(0), # Add batch dimension
"sample_rate": vc_model.sr
}
message += f"\nVoice converted successfully"
return (audio_data, message)
# Node mappings for ComfyUI
NODE_CLASS_MAPPINGS = {
"FL_ChatterboxTTS": FL_ChatterboxTTSNode,
"FL_ChatterboxVC": FL_ChatterboxVCNode,
}
# Display names for the nodes
NODE_DISPLAY_NAME_MAPPINGS = {
"FL_ChatterboxTTS": "FL Chatterbox TTS",
"FL_ChatterboxVC": "FL Chatterbox VC",
}
+15
View File
@@ -0,0 +1,15 @@
[project]
name = "comfyui_fill-chatterbox"
description = "Voice Clone and TTS model."
version = "1.0.0"
license = "LICENSE"
dependencies = ["diffusers", "librosa", "sounddevice", "glitch_this", "PyOpenGL", "glfw", "scipy>=1.13.1", "requests", "aiohttp", "moviepy", "matplotlib", "reportlab", "openai", "PyPDF2", "pdf2image", "PyMuPDF", "reportlab", "PyPDF2", "ollama", "kornia", "opencv-python", "gdown", "open_clip_torch", "google-genai"]
[project.urls]
Repository = "https://github.com/filliptm/ComfyUI_Fill-Nodes"
# Used by Comfy Registry https://comfyregistry.org
[tool.comfy]
PublisherId = "machinedelusions"
DisplayName = "ComfyUI_Fill-Nodes"
Icon = ""
+13
View File
@@ -0,0 +1,13 @@
# Core dependencies from chatterbox-tts without torch/torchaudio
numpy
resampy
librosa
s3tokenizer
transformers
diffusers
resemble-perth
omegaconf
conformer
# NOTE: chatterbox-tts must be installed separately with:
# pip install chatterbox-tts --no-deps