diff --git a/README.md b/README.md
index 054a5d0..6a28399 100644
--- a/README.md
+++ b/README.md
@@ -1,5 +1,12 @@
# ComfyUI_Fill-ChatterBox
+If you enjoy this project, consider supporting me on Patreon!
+
+
+
+
+
+
A custom node extension for ComfyUI that adds text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library.
Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the model falls apart really badly with anything longer than that, so it remains.
@@ -18,6 +25,12 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
pip install -r ComfyUI_Fill-ChatterBox/requirements.txt
```
+3. (Optional) Install watermarking support:
+ ```bash
+ pip install resemble-perth
+ ```
+ **Note**: The `resemble-perth` package may have compatibility issues with Python 3.12+. If you encounter import errors, the nodes will still function without watermarking.
+
## Usage
@@ -31,8 +44,50 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
- Connect input audio and target voice
- Both nodes support CPU fallback if CUDA errors occur
+### Dialog TTS Node (FL Chatterbox Dialog TTS)
+- Add the "FL Chatterbox Dialog TTS" node to your workflow.
+- This node is designed to synthesize speech for dialogs with up to 4 distinct speakers (SPEAKER A, SPEAKER B, SPEAKER C, and SPEAKER D).
+- **Inputs:**
+ - `dialog_text`: A multiline string where each line is prefixed by `SPEAKER A:`, `SPEAKER B:`, `SPEAKER C:`, or `SPEAKER D:`. For example:
+ ```
+ SPEAKER A: Hello, how are you?
+ SPEAKER B: I am fine, thank you!
+ SPEAKER C: What about you, Speaker D?
+ SPEAKER D: I'm doing great as well!
+ SPEAKER A: That's wonderful to hear.
+ ```
+ - `speaker_a_prompt`: An audio prompt (AUDIO type) for SPEAKER A's voice (required).
+ - `speaker_b_prompt`: An audio prompt (AUDIO type) for SPEAKER B's voice (required).
+ - `speaker_c_prompt`: An audio prompt (AUDIO type) for SPEAKER C's voice (optional).
+ - `speaker_d_prompt`: An audio prompt (AUDIO type) for SPEAKER D's voice (optional).
+ - `exaggeration`: Controls emotion intensity (0.25-2.0).
+ - `cfg_weight`: Controls pace/classifier-free guidance (0.2-1.0).
+ - `temperature`: Controls randomness in generation (0.05-5.0).
+ - `use_cpu` (optional): Boolean, defaults to False. Forces CPU usage.
+ - `keep_model_loaded` (optional): Boolean, defaults to False. Keeps the model loaded in memory.
+- **Outputs:**
+ - `dialog_audio`: Combined audio with all speakers
+ - `speaker_a_audio`: Isolated track for Speaker A (with silence for other speakers)
+ - `speaker_b_audio`: Isolated track for Speaker B (with silence for other speakers)
+ - `speaker_c_audio`: Isolated track for Speaker C (with silence for other speakers)
+ - `speaker_d_audio`: Isolated track for Speaker D (with silence for other speakers)
+- **Note:** SPEAKER C and SPEAKER D are optional. If their audio prompts are not provided, any dialog lines with "SPEAKER C:" or "SPEAKER D:" will be skipped.
+
## Change Log
+### 7/24/2025
+- Added Dialog TTS node that handles up to 4 speakers (A, B, C, D) for conversation-style audio creation
+- Extended all nodes (TTS, VC, Dialog) with seed parameters for reproducible generation
+- SPEAKER C and SPEAKER D are optional in Dialog node - allows flexible 2-4 speaker conversations
+- Each speaker gets isolated audio track output for advanced audio editing workflows
+
+### 6/24/2025
+- Added seed parameter to both TTS and VC nodes for reproducible generation
+- Seed range: 0 to 4,294,967,295 (32-bit integer)
+- Enables consistent audio output for debugging and workflow control
+- Made Perth watermarking optional to fix Python 3.12+ compatibility issues
+- Nodes now function without watermarking if resemble-perth import fails
+
### 5/31/2025
- Added Persistent model loading, and loading bar functionality
- Added Mac support (needs to be tested so HMU)
diff --git a/__init__.py b/__init__.py
index eb06173..1925b56 100644
--- a/__init__.py
+++ b/__init__.py
@@ -1,4 +1,13 @@
-from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
+from .chatterbox_node import NODE_CLASS_MAPPINGS as BASE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as BASE_DISPLAY_NAME_MAPPINGS
+from .chatterbox_dialog_node import NODE_CLASS_MAPPINGS as DIALOG_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as DIALOG_DISPLAY_NAME_MAPPINGS
+
+NODE_CLASS_MAPPINGS = {}
+NODE_CLASS_MAPPINGS.update(BASE_CLASS_MAPPINGS)
+NODE_CLASS_MAPPINGS.update(DIALOG_CLASS_MAPPINGS)
+
+NODE_DISPLAY_NAME_MAPPINGS = {}
+NODE_DISPLAY_NAME_MAPPINGS.update(BASE_DISPLAY_NAME_MAPPINGS)
+NODE_DISPLAY_NAME_MAPPINGS.update(DIALOG_DISPLAY_NAME_MAPPINGS)
WEB_DIRECTORY = "./web"
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]
\ No newline at end of file
diff --git a/chatterbox_dialog_node.py b/chatterbox_dialog_node.py
new file mode 100644
index 0000000..68ca53d
--- /dev/null
+++ b/chatterbox_dialog_node.py
@@ -0,0 +1,197 @@
+import os
+import torch
+import torchaudio
+import tempfile
+
+from .local_chatterbox.chatterbox.tts import ChatterboxTTS
+from comfy.utils import ProgressBar
+
+class FL_ChatterboxDialogTTSNode:
+ """
+ TTS Node that accepts dialog with speaker labels and generates audio using separate voice prompts.
+ """
+ @classmethod
+ def INPUT_TYPES(cls):
+ return {
+ "required": {
+ "dialog_text": ("STRING", {"multiline": True, "default": "SPEAKER A: Test test\nSPEAKER B: 1 2 3"}),
+ "speaker_A_Audio": ("AUDIO",),
+ "speaker_B_Audio": ("AUDIO",),
+ "exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
+ "cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
+ "temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
+ "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
+ },
+ "optional": {
+ "speaker_C_Audio": ("AUDIO",),
+ "speaker_D_Audio": ("AUDIO",),
+ "use_cpu": ("BOOLEAN", {"default": False}),
+ "keep_model_loaded": ("BOOLEAN", {"default": False}),
+ }
+ }
+
+ RETURN_TYPES = ("AUDIO", "AUDIO", "AUDIO", "AUDIO", "AUDIO", "STRING")
+ RETURN_NAMES = ("dialog_audio", "speaker_a_audio", "speaker_b_audio", "speaker_c_audio", "speaker_d_audio", "message")
+ FUNCTION = "generate_dialog"
+ CATEGORY = "ChatterBox"
+
+ _model = None
+ _device = None
+
+ def generate_dialog(self, dialog_text, speaker_A_Audio, speaker_B_Audio,
+ exaggeration, cfg_weight, temperature, seed,
+ speaker_C_Audio=None, speaker_D_Audio=None,
+ use_cpu=False, keep_model_loaded=False):
+ # Set random seeds for reproducibility
+ torch.manual_seed(seed)
+ if torch.cuda.is_available():
+ torch.cuda.manual_seed(seed)
+ torch.cuda.manual_seed_all(seed)
+ if torch.backends.mps.is_available():
+ torch.mps.manual_seed(seed)
+ import numpy as np
+ import random
+ np.random.seed(seed)
+ random.seed(seed)
+
+ device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "mps" if torch.backends.mps.is_available() else "cpu")
+ pbar = ProgressBar(100)
+ message = f"Running on {device}"
+
+ def save_temp_audio(audio_data):
+ path = tempfile.NamedTemporaryFile(suffix='.wav', delete=False).name
+ torchaudio.save(path, audio_data['waveform'].squeeze(0), audio_data['sample_rate'])
+ return path
+
+ prompt_a_path = save_temp_audio(speaker_A_Audio)
+ prompt_b_path = save_temp_audio(speaker_B_Audio)
+ temp_files = [prompt_a_path, prompt_b_path]
+
+ # Handle optional speakers C and D
+ prompt_c_path = None
+ prompt_d_path = None
+ if speaker_C_Audio is not None:
+ prompt_c_path = save_temp_audio(speaker_C_Audio)
+ temp_files.append(prompt_c_path)
+ if speaker_D_Audio is not None:
+ prompt_d_path = save_temp_audio(speaker_D_Audio)
+ temp_files.append(prompt_d_path)
+
+ if self._model is None or self._device != device:
+ self._model = ChatterboxTTS.from_pretrained(device=device)
+ self._device = device
+ tts = self._model
+
+ lines = dialog_text.strip().splitlines()
+ speaker_a_waveforms = []
+ speaker_b_waveforms = []
+ speaker_c_waveforms = []
+ speaker_d_waveforms = []
+ combined_dialog_waveforms = []
+
+ for i, line in enumerate(lines):
+ wav = None
+ if line.startswith("SPEAKER A:"):
+ content = line[len("SPEAKER A:"):].strip()
+ prompt_path = prompt_a_path
+ pbar.update_absolute(int((i / len(lines)) * 80))
+ current_speaker_wav = tts.generate(
+ text=content,
+ audio_prompt_path=prompt_path,
+ exaggeration=exaggeration,
+ cfg_weight=cfg_weight,
+ temperature=temperature
+ )
+ speaker_a_waveforms.append(current_speaker_wav)
+ combined_dialog_waveforms.append(current_speaker_wav)
+ # Add silence to other speakers' tracks
+ silence = torch.zeros_like(current_speaker_wav)
+ speaker_b_waveforms.append(silence)
+ speaker_c_waveforms.append(silence)
+ speaker_d_waveforms.append(silence)
+ elif line.startswith("SPEAKER B:"):
+ content = line[len("SPEAKER B:"):].strip()
+ prompt_path = prompt_b_path
+ pbar.update_absolute(int((i / len(lines)) * 80))
+ current_speaker_wav = tts.generate(
+ text=content,
+ audio_prompt_path=prompt_path,
+ exaggeration=exaggeration,
+ cfg_weight=cfg_weight,
+ temperature=temperature
+ )
+ speaker_b_waveforms.append(current_speaker_wav)
+ combined_dialog_waveforms.append(current_speaker_wav)
+ # Add silence to other speakers' tracks
+ silence = torch.zeros_like(current_speaker_wav)
+ speaker_a_waveforms.append(silence)
+ speaker_c_waveforms.append(silence)
+ speaker_d_waveforms.append(silence)
+ elif line.startswith("SPEAKER C:") and prompt_c_path is not None:
+ content = line[len("SPEAKER C:"):].strip()
+ prompt_path = prompt_c_path
+ pbar.update_absolute(int((i / len(lines)) * 80))
+ current_speaker_wav = tts.generate(
+ text=content,
+ audio_prompt_path=prompt_path,
+ exaggeration=exaggeration,
+ cfg_weight=cfg_weight,
+ temperature=temperature
+ )
+ speaker_c_waveforms.append(current_speaker_wav)
+ combined_dialog_waveforms.append(current_speaker_wav)
+ # Add silence to other speakers' tracks
+ silence = torch.zeros_like(current_speaker_wav)
+ speaker_a_waveforms.append(silence)
+ speaker_b_waveforms.append(silence)
+ speaker_d_waveforms.append(silence)
+ elif line.startswith("SPEAKER D:") and prompt_d_path is not None:
+ content = line[len("SPEAKER D:"):].strip()
+ prompt_path = prompt_d_path
+ pbar.update_absolute(int((i / len(lines)) * 80))
+ current_speaker_wav = tts.generate(
+ text=content,
+ audio_prompt_path=prompt_path,
+ exaggeration=exaggeration,
+ cfg_weight=cfg_weight,
+ temperature=temperature
+ )
+ speaker_d_waveforms.append(current_speaker_wav)
+ combined_dialog_waveforms.append(current_speaker_wav)
+ # Add silence to other speakers' tracks
+ silence = torch.zeros_like(current_speaker_wav)
+ speaker_a_waveforms.append(silence)
+ speaker_b_waveforms.append(silence)
+ speaker_c_waveforms.append(silence)
+ else:
+ continue # skip malformed line or missing prompt
+
+ if not combined_dialog_waveforms:
+ empty_audio = {"waveform": torch.zeros((1, 1, 1)), "sample_rate": tts.sr if tts else 16000}
+ return (empty_audio, empty_audio, empty_audio, empty_audio, empty_audio, "No valid dialog lines found.")
+
+ combined_waveform = torch.cat(combined_dialog_waveforms, dim=-1)
+ speaker_a_track = torch.cat(speaker_a_waveforms, dim=-1)
+ speaker_b_track = torch.cat(speaker_b_waveforms, dim=-1)
+ speaker_c_track = torch.cat(speaker_c_waveforms, dim=-1)
+ speaker_d_track = torch.cat(speaker_d_waveforms, dim=-1)
+
+ dialog_audio = {"waveform": combined_waveform.unsqueeze(0), "sample_rate": tts.sr}
+ speaker_a_audio = {"waveform": speaker_a_track.unsqueeze(0), "sample_rate": tts.sr}
+ speaker_b_audio = {"waveform": speaker_b_track.unsqueeze(0), "sample_rate": tts.sr}
+ speaker_c_audio = {"waveform": speaker_c_track.unsqueeze(0), "sample_rate": tts.sr}
+ speaker_d_audio = {"waveform": speaker_d_track.unsqueeze(0), "sample_rate": tts.sr}
+
+ for f in temp_files:
+ os.unlink(f)
+
+ pbar.update_absolute(100)
+ return (dialog_audio, speaker_a_audio, speaker_b_audio, speaker_c_audio, speaker_d_audio, "Dialog synthesized successfully.")
+
+NODE_CLASS_MAPPINGS = {
+ "FL_ChatterboxDialogTTS": FL_ChatterboxDialogTTSNode,
+}
+
+NODE_DISPLAY_NAME_MAPPINGS = {
+ "FL_ChatterboxDialogTTS": "FL Chatterbox Dialog TTS",
+}
\ No newline at end of file
diff --git a/chatterbox_node.py b/chatterbox_node.py
index aff07f0..88031a2 100644
--- a/chatterbox_node.py
+++ b/chatterbox_node.py
@@ -57,6 +57,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
+ "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
},
"optional": {
"audio_prompt": ("AUDIO",),
@@ -70,7 +71,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
FUNCTION = "generate_speech"
CATEGORY = "ChatterBox"
- def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
+ def generate_speech(self, text, exaggeration, cfg_weight, temperature, seed, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
"""
Generate speech from text.
@@ -79,6 +80,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
exaggeration: Controls emotion intensity (0.25-2.0).
cfg_weight: Controls pace/classifier-free guidance (0.2-1.0).
temperature: Controls randomness in generation (0.05-5.0).
+ seed: Random seed for reproducible generation.
audio_prompt: AUDIO object containing the reference voice for TTS voice cloning.
use_cpu: If True, forces CPU usage even if CUDA is available.
keep_model_loaded: If True, keeps the model loaded in memory after generation.
@@ -86,6 +88,18 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
Returns:
Tuple of (audio, message)
"""
+ # Set random seeds for reproducibility
+ torch.manual_seed(seed)
+ if torch.cuda.is_available():
+ torch.cuda.manual_seed(seed)
+ torch.cuda.manual_seed_all(seed)
+ if torch.backends.mps.is_available():
+ torch.mps.manual_seed(seed)
+ import numpy as np
+ import random
+ np.random.seed(seed)
+ random.seed(seed)
+
# Determine device to use
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
if use_cpu:
@@ -291,6 +305,7 @@ class FL_ChatterboxVCNode(AudioNodeBase):
"required": {
"input_audio": ("AUDIO",),
"target_voice": ("AUDIO",),
+ "seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
},
"optional": {
"use_cpu": ("BOOLEAN", {"default": False}),
@@ -303,19 +318,32 @@ class FL_ChatterboxVCNode(AudioNodeBase):
FUNCTION = "convert_voice"
CATEGORY = "ChatterBox"
- def convert_voice(self, input_audio, target_voice, use_cpu=False, keep_model_loaded=False):
+ def convert_voice(self, input_audio, target_voice, seed, use_cpu=False, keep_model_loaded=False):
"""
Convert the voice in an audio file to match a target voice.
Args:
input_audio: AUDIO object containing the audio to convert.
target_voice: AUDIO object containing the target voice.
+ seed: Random seed for reproducible generation.
use_cpu: If True, forces CPU usage even if CUDA is available.
keep_model_loaded: If True, keeps the model loaded in memory after conversion.
Returns:
Tuple of (audio, message)
"""
+ # Set random seeds for reproducibility
+ torch.manual_seed(seed)
+ if torch.cuda.is_available():
+ torch.cuda.manual_seed(seed)
+ torch.cuda.manual_seed_all(seed)
+ if torch.backends.mps.is_available():
+ torch.mps.manual_seed(seed)
+ import numpy as np
+ import random
+ np.random.seed(seed)
+ random.seed(seed)
+
# Determine device to use
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
if use_cpu:
diff --git a/requirements.txt b/requirements.txt
index 70519ea..dbf624d 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -4,7 +4,9 @@ librosa
s3tokenizer
transformers
diffusers
-resemble-perth
+# Optional watermarking (may have Python 3.12+ compatibility issues)
+# resemble-perth
omegaconf
conformer
-safetensors
\ No newline at end of file
+safetensors
+soundfile
\ No newline at end of file
diff --git a/web/image.png b/web/image.png
index adfce04..586691e 100644
Binary files a/web/image.png and b/web/image.png differ