diff --git a/README.md b/README.md index 054a5d0..6a28399 100644 --- a/README.md +++ b/README.md @@ -1,5 +1,12 @@ # ComfyUI_Fill-ChatterBox +If you enjoy this project, consider supporting me on Patreon! +

+ + Patreon + +

+ A custom node extension for ComfyUI that adds text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library. Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the model falls apart really badly with anything longer than that, so it remains. @@ -18,6 +25,12 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod pip install -r ComfyUI_Fill-ChatterBox/requirements.txt ``` +3. (Optional) Install watermarking support: + ```bash + pip install resemble-perth + ``` + **Note**: The `resemble-perth` package may have compatibility issues with Python 3.12+. If you encounter import errors, the nodes will still function without watermarking. + ## Usage @@ -31,8 +44,50 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod - Connect input audio and target voice - Both nodes support CPU fallback if CUDA errors occur +### Dialog TTS Node (FL Chatterbox Dialog TTS) +- Add the "FL Chatterbox Dialog TTS" node to your workflow. +- This node is designed to synthesize speech for dialogs with up to 4 distinct speakers (SPEAKER A, SPEAKER B, SPEAKER C, and SPEAKER D). +- **Inputs:** + - `dialog_text`: A multiline string where each line is prefixed by `SPEAKER A:`, `SPEAKER B:`, `SPEAKER C:`, or `SPEAKER D:`. For example: + ``` + SPEAKER A: Hello, how are you? + SPEAKER B: I am fine, thank you! + SPEAKER C: What about you, Speaker D? + SPEAKER D: I'm doing great as well! + SPEAKER A: That's wonderful to hear. + ``` + - `speaker_a_prompt`: An audio prompt (AUDIO type) for SPEAKER A's voice (required). + - `speaker_b_prompt`: An audio prompt (AUDIO type) for SPEAKER B's voice (required). + - `speaker_c_prompt`: An audio prompt (AUDIO type) for SPEAKER C's voice (optional). + - `speaker_d_prompt`: An audio prompt (AUDIO type) for SPEAKER D's voice (optional). + - `exaggeration`: Controls emotion intensity (0.25-2.0). + - `cfg_weight`: Controls pace/classifier-free guidance (0.2-1.0). + - `temperature`: Controls randomness in generation (0.05-5.0). + - `use_cpu` (optional): Boolean, defaults to False. Forces CPU usage. + - `keep_model_loaded` (optional): Boolean, defaults to False. Keeps the model loaded in memory. +- **Outputs:** + - `dialog_audio`: Combined audio with all speakers + - `speaker_a_audio`: Isolated track for Speaker A (with silence for other speakers) + - `speaker_b_audio`: Isolated track for Speaker B (with silence for other speakers) + - `speaker_c_audio`: Isolated track for Speaker C (with silence for other speakers) + - `speaker_d_audio`: Isolated track for Speaker D (with silence for other speakers) +- **Note:** SPEAKER C and SPEAKER D are optional. If their audio prompts are not provided, any dialog lines with "SPEAKER C:" or "SPEAKER D:" will be skipped. + ## Change Log +### 7/24/2025 +- Added Dialog TTS node that handles up to 4 speakers (A, B, C, D) for conversation-style audio creation +- Extended all nodes (TTS, VC, Dialog) with seed parameters for reproducible generation +- SPEAKER C and SPEAKER D are optional in Dialog node - allows flexible 2-4 speaker conversations +- Each speaker gets isolated audio track output for advanced audio editing workflows + +### 6/24/2025 +- Added seed parameter to both TTS and VC nodes for reproducible generation +- Seed range: 0 to 4,294,967,295 (32-bit integer) +- Enables consistent audio output for debugging and workflow control +- Made Perth watermarking optional to fix Python 3.12+ compatibility issues +- Nodes now function without watermarking if resemble-perth import fails + ### 5/31/2025 - Added Persistent model loading, and loading bar functionality - Added Mac support (needs to be tested so HMU) diff --git a/__init__.py b/__init__.py index eb06173..1925b56 100644 --- a/__init__.py +++ b/__init__.py @@ -1,4 +1,13 @@ -from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS +from .chatterbox_node import NODE_CLASS_MAPPINGS as BASE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as BASE_DISPLAY_NAME_MAPPINGS +from .chatterbox_dialog_node import NODE_CLASS_MAPPINGS as DIALOG_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as DIALOG_DISPLAY_NAME_MAPPINGS + +NODE_CLASS_MAPPINGS = {} +NODE_CLASS_MAPPINGS.update(BASE_CLASS_MAPPINGS) +NODE_CLASS_MAPPINGS.update(DIALOG_CLASS_MAPPINGS) + +NODE_DISPLAY_NAME_MAPPINGS = {} +NODE_DISPLAY_NAME_MAPPINGS.update(BASE_DISPLAY_NAME_MAPPINGS) +NODE_DISPLAY_NAME_MAPPINGS.update(DIALOG_DISPLAY_NAME_MAPPINGS) WEB_DIRECTORY = "./web" __all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"] \ No newline at end of file diff --git a/chatterbox_dialog_node.py b/chatterbox_dialog_node.py new file mode 100644 index 0000000..68ca53d --- /dev/null +++ b/chatterbox_dialog_node.py @@ -0,0 +1,197 @@ +import os +import torch +import torchaudio +import tempfile + +from .local_chatterbox.chatterbox.tts import ChatterboxTTS +from comfy.utils import ProgressBar + +class FL_ChatterboxDialogTTSNode: + """ + TTS Node that accepts dialog with speaker labels and generates audio using separate voice prompts. + """ + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "dialog_text": ("STRING", {"multiline": True, "default": "SPEAKER A: Test test\nSPEAKER B: 1 2 3"}), + "speaker_A_Audio": ("AUDIO",), + "speaker_B_Audio": ("AUDIO",), + "exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}), + "cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}), + "temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}), + "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}), + }, + "optional": { + "speaker_C_Audio": ("AUDIO",), + "speaker_D_Audio": ("AUDIO",), + "use_cpu": ("BOOLEAN", {"default": False}), + "keep_model_loaded": ("BOOLEAN", {"default": False}), + } + } + + RETURN_TYPES = ("AUDIO", "AUDIO", "AUDIO", "AUDIO", "AUDIO", "STRING") + RETURN_NAMES = ("dialog_audio", "speaker_a_audio", "speaker_b_audio", "speaker_c_audio", "speaker_d_audio", "message") + FUNCTION = "generate_dialog" + CATEGORY = "ChatterBox" + + _model = None + _device = None + + def generate_dialog(self, dialog_text, speaker_A_Audio, speaker_B_Audio, + exaggeration, cfg_weight, temperature, seed, + speaker_C_Audio=None, speaker_D_Audio=None, + use_cpu=False, keep_model_loaded=False): + # Set random seeds for reproducibility + torch.manual_seed(seed) + if torch.cuda.is_available(): + torch.cuda.manual_seed(seed) + torch.cuda.manual_seed_all(seed) + if torch.backends.mps.is_available(): + torch.mps.manual_seed(seed) + import numpy as np + import random + np.random.seed(seed) + random.seed(seed) + + device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "mps" if torch.backends.mps.is_available() else "cpu") + pbar = ProgressBar(100) + message = f"Running on {device}" + + def save_temp_audio(audio_data): + path = tempfile.NamedTemporaryFile(suffix='.wav', delete=False).name + torchaudio.save(path, audio_data['waveform'].squeeze(0), audio_data['sample_rate']) + return path + + prompt_a_path = save_temp_audio(speaker_A_Audio) + prompt_b_path = save_temp_audio(speaker_B_Audio) + temp_files = [prompt_a_path, prompt_b_path] + + # Handle optional speakers C and D + prompt_c_path = None + prompt_d_path = None + if speaker_C_Audio is not None: + prompt_c_path = save_temp_audio(speaker_C_Audio) + temp_files.append(prompt_c_path) + if speaker_D_Audio is not None: + prompt_d_path = save_temp_audio(speaker_D_Audio) + temp_files.append(prompt_d_path) + + if self._model is None or self._device != device: + self._model = ChatterboxTTS.from_pretrained(device=device) + self._device = device + tts = self._model + + lines = dialog_text.strip().splitlines() + speaker_a_waveforms = [] + speaker_b_waveforms = [] + speaker_c_waveforms = [] + speaker_d_waveforms = [] + combined_dialog_waveforms = [] + + for i, line in enumerate(lines): + wav = None + if line.startswith("SPEAKER A:"): + content = line[len("SPEAKER A:"):].strip() + prompt_path = prompt_a_path + pbar.update_absolute(int((i / len(lines)) * 80)) + current_speaker_wav = tts.generate( + text=content, + audio_prompt_path=prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature + ) + speaker_a_waveforms.append(current_speaker_wav) + combined_dialog_waveforms.append(current_speaker_wav) + # Add silence to other speakers' tracks + silence = torch.zeros_like(current_speaker_wav) + speaker_b_waveforms.append(silence) + speaker_c_waveforms.append(silence) + speaker_d_waveforms.append(silence) + elif line.startswith("SPEAKER B:"): + content = line[len("SPEAKER B:"):].strip() + prompt_path = prompt_b_path + pbar.update_absolute(int((i / len(lines)) * 80)) + current_speaker_wav = tts.generate( + text=content, + audio_prompt_path=prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature + ) + speaker_b_waveforms.append(current_speaker_wav) + combined_dialog_waveforms.append(current_speaker_wav) + # Add silence to other speakers' tracks + silence = torch.zeros_like(current_speaker_wav) + speaker_a_waveforms.append(silence) + speaker_c_waveforms.append(silence) + speaker_d_waveforms.append(silence) + elif line.startswith("SPEAKER C:") and prompt_c_path is not None: + content = line[len("SPEAKER C:"):].strip() + prompt_path = prompt_c_path + pbar.update_absolute(int((i / len(lines)) * 80)) + current_speaker_wav = tts.generate( + text=content, + audio_prompt_path=prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature + ) + speaker_c_waveforms.append(current_speaker_wav) + combined_dialog_waveforms.append(current_speaker_wav) + # Add silence to other speakers' tracks + silence = torch.zeros_like(current_speaker_wav) + speaker_a_waveforms.append(silence) + speaker_b_waveforms.append(silence) + speaker_d_waveforms.append(silence) + elif line.startswith("SPEAKER D:") and prompt_d_path is not None: + content = line[len("SPEAKER D:"):].strip() + prompt_path = prompt_d_path + pbar.update_absolute(int((i / len(lines)) * 80)) + current_speaker_wav = tts.generate( + text=content, + audio_prompt_path=prompt_path, + exaggeration=exaggeration, + cfg_weight=cfg_weight, + temperature=temperature + ) + speaker_d_waveforms.append(current_speaker_wav) + combined_dialog_waveforms.append(current_speaker_wav) + # Add silence to other speakers' tracks + silence = torch.zeros_like(current_speaker_wav) + speaker_a_waveforms.append(silence) + speaker_b_waveforms.append(silence) + speaker_c_waveforms.append(silence) + else: + continue # skip malformed line or missing prompt + + if not combined_dialog_waveforms: + empty_audio = {"waveform": torch.zeros((1, 1, 1)), "sample_rate": tts.sr if tts else 16000} + return (empty_audio, empty_audio, empty_audio, empty_audio, empty_audio, "No valid dialog lines found.") + + combined_waveform = torch.cat(combined_dialog_waveforms, dim=-1) + speaker_a_track = torch.cat(speaker_a_waveforms, dim=-1) + speaker_b_track = torch.cat(speaker_b_waveforms, dim=-1) + speaker_c_track = torch.cat(speaker_c_waveforms, dim=-1) + speaker_d_track = torch.cat(speaker_d_waveforms, dim=-1) + + dialog_audio = {"waveform": combined_waveform.unsqueeze(0), "sample_rate": tts.sr} + speaker_a_audio = {"waveform": speaker_a_track.unsqueeze(0), "sample_rate": tts.sr} + speaker_b_audio = {"waveform": speaker_b_track.unsqueeze(0), "sample_rate": tts.sr} + speaker_c_audio = {"waveform": speaker_c_track.unsqueeze(0), "sample_rate": tts.sr} + speaker_d_audio = {"waveform": speaker_d_track.unsqueeze(0), "sample_rate": tts.sr} + + for f in temp_files: + os.unlink(f) + + pbar.update_absolute(100) + return (dialog_audio, speaker_a_audio, speaker_b_audio, speaker_c_audio, speaker_d_audio, "Dialog synthesized successfully.") + +NODE_CLASS_MAPPINGS = { + "FL_ChatterboxDialogTTS": FL_ChatterboxDialogTTSNode, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "FL_ChatterboxDialogTTS": "FL Chatterbox Dialog TTS", +} \ No newline at end of file diff --git a/chatterbox_node.py b/chatterbox_node.py index aff07f0..88031a2 100644 --- a/chatterbox_node.py +++ b/chatterbox_node.py @@ -57,6 +57,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase): "exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}), "cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}), "temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}), + "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}), }, "optional": { "audio_prompt": ("AUDIO",), @@ -70,7 +71,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase): FUNCTION = "generate_speech" CATEGORY = "ChatterBox" - def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False, keep_model_loaded=False): + def generate_speech(self, text, exaggeration, cfg_weight, temperature, seed, audio_prompt=None, use_cpu=False, keep_model_loaded=False): """ Generate speech from text. @@ -79,6 +80,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase): exaggeration: Controls emotion intensity (0.25-2.0). cfg_weight: Controls pace/classifier-free guidance (0.2-1.0). temperature: Controls randomness in generation (0.05-5.0). + seed: Random seed for reproducible generation. audio_prompt: AUDIO object containing the reference voice for TTS voice cloning. use_cpu: If True, forces CPU usage even if CUDA is available. keep_model_loaded: If True, keeps the model loaded in memory after generation. @@ -86,6 +88,18 @@ class FL_ChatterboxTTSNode(AudioNodeBase): Returns: Tuple of (audio, message) """ + # Set random seeds for reproducibility + torch.manual_seed(seed) + if torch.cuda.is_available(): + torch.cuda.manual_seed(seed) + torch.cuda.manual_seed_all(seed) + if torch.backends.mps.is_available(): + torch.mps.manual_seed(seed) + import numpy as np + import random + np.random.seed(seed) + random.seed(seed) + # Determine device to use device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu")) if use_cpu: @@ -291,6 +305,7 @@ class FL_ChatterboxVCNode(AudioNodeBase): "required": { "input_audio": ("AUDIO",), "target_voice": ("AUDIO",), + "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}), }, "optional": { "use_cpu": ("BOOLEAN", {"default": False}), @@ -303,19 +318,32 @@ class FL_ChatterboxVCNode(AudioNodeBase): FUNCTION = "convert_voice" CATEGORY = "ChatterBox" - def convert_voice(self, input_audio, target_voice, use_cpu=False, keep_model_loaded=False): + def convert_voice(self, input_audio, target_voice, seed, use_cpu=False, keep_model_loaded=False): """ Convert the voice in an audio file to match a target voice. Args: input_audio: AUDIO object containing the audio to convert. target_voice: AUDIO object containing the target voice. + seed: Random seed for reproducible generation. use_cpu: If True, forces CPU usage even if CUDA is available. keep_model_loaded: If True, keeps the model loaded in memory after conversion. Returns: Tuple of (audio, message) """ + # Set random seeds for reproducibility + torch.manual_seed(seed) + if torch.cuda.is_available(): + torch.cuda.manual_seed(seed) + torch.cuda.manual_seed_all(seed) + if torch.backends.mps.is_available(): + torch.mps.manual_seed(seed) + import numpy as np + import random + np.random.seed(seed) + random.seed(seed) + # Determine device to use device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu")) if use_cpu: diff --git a/requirements.txt b/requirements.txt index 70519ea..dbf624d 100644 --- a/requirements.txt +++ b/requirements.txt @@ -4,7 +4,9 @@ librosa s3tokenizer transformers diffusers -resemble-perth +# Optional watermarking (may have Python 3.12+ compatibility issues) +# resemble-perth omegaconf conformer -safetensors \ No newline at end of file +safetensors +soundfile \ No newline at end of file diff --git a/web/image.png b/web/image.png index adfce04..586691e 100644 Binary files a/web/image.png and b/web/image.png differ