added multi dialog node

This commit is contained in:
filliptm
2025-07-24 02:29:40 -07:00
parent d900801d38
commit 77903b3950
6 changed files with 296 additions and 5 deletions
+55
View File
@@ -1,5 +1,12 @@
# ComfyUI_Fill-ChatterBox
If you enjoy this project, consider supporting me on Patreon!
<p align="left">
<a href="https://www.patreon.com/c/Machinedelusions">
<img src="assets/Patreon.png" width="150px" alt="Patreon">
</a>
</p>
A custom node extension for ComfyUI that adds text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library.
Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the model falls apart really badly with anything longer than that, so it remains.
@@ -18,6 +25,12 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
pip install -r ComfyUI_Fill-ChatterBox/requirements.txt
```
3. (Optional) Install watermarking support:
```bash
pip install resemble-perth
```
**Note**: The `resemble-perth` package may have compatibility issues with Python 3.12+. If you encounter import errors, the nodes will still function without watermarking.
## Usage
@@ -31,8 +44,50 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
- Connect input audio and target voice
- Both nodes support CPU fallback if CUDA errors occur
### Dialog TTS Node (FL Chatterbox Dialog TTS)
- Add the "FL Chatterbox Dialog TTS" node to your workflow.
- This node is designed to synthesize speech for dialogs with up to 4 distinct speakers (SPEAKER A, SPEAKER B, SPEAKER C, and SPEAKER D).
- **Inputs:**
- `dialog_text`: A multiline string where each line is prefixed by `SPEAKER A:`, `SPEAKER B:`, `SPEAKER C:`, or `SPEAKER D:`. For example:
```
SPEAKER A: Hello, how are you?
SPEAKER B: I am fine, thank you!
SPEAKER C: What about you, Speaker D?
SPEAKER D: I'm doing great as well!
SPEAKER A: That's wonderful to hear.
```
- `speaker_a_prompt`: An audio prompt (AUDIO type) for SPEAKER A's voice (required).
- `speaker_b_prompt`: An audio prompt (AUDIO type) for SPEAKER B's voice (required).
- `speaker_c_prompt`: An audio prompt (AUDIO type) for SPEAKER C's voice (optional).
- `speaker_d_prompt`: An audio prompt (AUDIO type) for SPEAKER D's voice (optional).
- `exaggeration`: Controls emotion intensity (0.25-2.0).
- `cfg_weight`: Controls pace/classifier-free guidance (0.2-1.0).
- `temperature`: Controls randomness in generation (0.05-5.0).
- `use_cpu` (optional): Boolean, defaults to False. Forces CPU usage.
- `keep_model_loaded` (optional): Boolean, defaults to False. Keeps the model loaded in memory.
- **Outputs:**
- `dialog_audio`: Combined audio with all speakers
- `speaker_a_audio`: Isolated track for Speaker A (with silence for other speakers)
- `speaker_b_audio`: Isolated track for Speaker B (with silence for other speakers)
- `speaker_c_audio`: Isolated track for Speaker C (with silence for other speakers)
- `speaker_d_audio`: Isolated track for Speaker D (with silence for other speakers)
- **Note:** SPEAKER C and SPEAKER D are optional. If their audio prompts are not provided, any dialog lines with "SPEAKER C:" or "SPEAKER D:" will be skipped.
## Change Log
### 7/24/2025
- Added Dialog TTS node that handles up to 4 speakers (A, B, C, D) for conversation-style audio creation
- Extended all nodes (TTS, VC, Dialog) with seed parameters for reproducible generation
- SPEAKER C and SPEAKER D are optional in Dialog node - allows flexible 2-4 speaker conversations
- Each speaker gets isolated audio track output for advanced audio editing workflows
### 6/24/2025
- Added seed parameter to both TTS and VC nodes for reproducible generation
- Seed range: 0 to 4,294,967,295 (32-bit integer)
- Enables consistent audio output for debugging and workflow control
- Made Perth watermarking optional to fix Python 3.12+ compatibility issues
- Nodes now function without watermarking if resemble-perth import fails
### 5/31/2025
- Added Persistent model loading, and loading bar functionality
- Added Mac support (needs to be tested so HMU)
+10 -1
View File
@@ -1,4 +1,13 @@
from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
from .chatterbox_node import NODE_CLASS_MAPPINGS as BASE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as BASE_DISPLAY_NAME_MAPPINGS
from .chatterbox_dialog_node import NODE_CLASS_MAPPINGS as DIALOG_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as DIALOG_DISPLAY_NAME_MAPPINGS
NODE_CLASS_MAPPINGS = {}
NODE_CLASS_MAPPINGS.update(BASE_CLASS_MAPPINGS)
NODE_CLASS_MAPPINGS.update(DIALOG_CLASS_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS = {}
NODE_DISPLAY_NAME_MAPPINGS.update(BASE_DISPLAY_NAME_MAPPINGS)
NODE_DISPLAY_NAME_MAPPINGS.update(DIALOG_DISPLAY_NAME_MAPPINGS)
WEB_DIRECTORY = "./web"
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]
+197
View File
@@ -0,0 +1,197 @@
import os
import torch
import torchaudio
import tempfile
from .local_chatterbox.chatterbox.tts import ChatterboxTTS
from comfy.utils import ProgressBar
class FL_ChatterboxDialogTTSNode:
"""
TTS Node that accepts dialog with speaker labels and generates audio using separate voice prompts.
"""
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"dialog_text": ("STRING", {"multiline": True, "default": "SPEAKER A: Test test\nSPEAKER B: 1 2 3"}),
"speaker_A_Audio": ("AUDIO",),
"speaker_B_Audio": ("AUDIO",),
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
},
"optional": {
"speaker_C_Audio": ("AUDIO",),
"speaker_D_Audio": ("AUDIO",),
"use_cpu": ("BOOLEAN", {"default": False}),
"keep_model_loaded": ("BOOLEAN", {"default": False}),
}
}
RETURN_TYPES = ("AUDIO", "AUDIO", "AUDIO", "AUDIO", "AUDIO", "STRING")
RETURN_NAMES = ("dialog_audio", "speaker_a_audio", "speaker_b_audio", "speaker_c_audio", "speaker_d_audio", "message")
FUNCTION = "generate_dialog"
CATEGORY = "ChatterBox"
_model = None
_device = None
def generate_dialog(self, dialog_text, speaker_A_Audio, speaker_B_Audio,
exaggeration, cfg_weight, temperature, seed,
speaker_C_Audio=None, speaker_D_Audio=None,
use_cpu=False, keep_model_loaded=False):
# Set random seeds for reproducibility
torch.manual_seed(seed)
if torch.cuda.is_available():
torch.cuda.manual_seed(seed)
torch.cuda.manual_seed_all(seed)
if torch.backends.mps.is_available():
torch.mps.manual_seed(seed)
import numpy as np
import random
np.random.seed(seed)
random.seed(seed)
device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "mps" if torch.backends.mps.is_available() else "cpu")
pbar = ProgressBar(100)
message = f"Running on {device}"
def save_temp_audio(audio_data):
path = tempfile.NamedTemporaryFile(suffix='.wav', delete=False).name
torchaudio.save(path, audio_data['waveform'].squeeze(0), audio_data['sample_rate'])
return path
prompt_a_path = save_temp_audio(speaker_A_Audio)
prompt_b_path = save_temp_audio(speaker_B_Audio)
temp_files = [prompt_a_path, prompt_b_path]
# Handle optional speakers C and D
prompt_c_path = None
prompt_d_path = None
if speaker_C_Audio is not None:
prompt_c_path = save_temp_audio(speaker_C_Audio)
temp_files.append(prompt_c_path)
if speaker_D_Audio is not None:
prompt_d_path = save_temp_audio(speaker_D_Audio)
temp_files.append(prompt_d_path)
if self._model is None or self._device != device:
self._model = ChatterboxTTS.from_pretrained(device=device)
self._device = device
tts = self._model
lines = dialog_text.strip().splitlines()
speaker_a_waveforms = []
speaker_b_waveforms = []
speaker_c_waveforms = []
speaker_d_waveforms = []
combined_dialog_waveforms = []
for i, line in enumerate(lines):
wav = None
if line.startswith("SPEAKER A:"):
content = line[len("SPEAKER A:"):].strip()
prompt_path = prompt_a_path
pbar.update_absolute(int((i / len(lines)) * 80))
current_speaker_wav = tts.generate(
text=content,
audio_prompt_path=prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature
)
speaker_a_waveforms.append(current_speaker_wav)
combined_dialog_waveforms.append(current_speaker_wav)
# Add silence to other speakers' tracks
silence = torch.zeros_like(current_speaker_wav)
speaker_b_waveforms.append(silence)
speaker_c_waveforms.append(silence)
speaker_d_waveforms.append(silence)
elif line.startswith("SPEAKER B:"):
content = line[len("SPEAKER B:"):].strip()
prompt_path = prompt_b_path
pbar.update_absolute(int((i / len(lines)) * 80))
current_speaker_wav = tts.generate(
text=content,
audio_prompt_path=prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature
)
speaker_b_waveforms.append(current_speaker_wav)
combined_dialog_waveforms.append(current_speaker_wav)
# Add silence to other speakers' tracks
silence = torch.zeros_like(current_speaker_wav)
speaker_a_waveforms.append(silence)
speaker_c_waveforms.append(silence)
speaker_d_waveforms.append(silence)
elif line.startswith("SPEAKER C:") and prompt_c_path is not None:
content = line[len("SPEAKER C:"):].strip()
prompt_path = prompt_c_path
pbar.update_absolute(int((i / len(lines)) * 80))
current_speaker_wav = tts.generate(
text=content,
audio_prompt_path=prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature
)
speaker_c_waveforms.append(current_speaker_wav)
combined_dialog_waveforms.append(current_speaker_wav)
# Add silence to other speakers' tracks
silence = torch.zeros_like(current_speaker_wav)
speaker_a_waveforms.append(silence)
speaker_b_waveforms.append(silence)
speaker_d_waveforms.append(silence)
elif line.startswith("SPEAKER D:") and prompt_d_path is not None:
content = line[len("SPEAKER D:"):].strip()
prompt_path = prompt_d_path
pbar.update_absolute(int((i / len(lines)) * 80))
current_speaker_wav = tts.generate(
text=content,
audio_prompt_path=prompt_path,
exaggeration=exaggeration,
cfg_weight=cfg_weight,
temperature=temperature
)
speaker_d_waveforms.append(current_speaker_wav)
combined_dialog_waveforms.append(current_speaker_wav)
# Add silence to other speakers' tracks
silence = torch.zeros_like(current_speaker_wav)
speaker_a_waveforms.append(silence)
speaker_b_waveforms.append(silence)
speaker_c_waveforms.append(silence)
else:
continue # skip malformed line or missing prompt
if not combined_dialog_waveforms:
empty_audio = {"waveform": torch.zeros((1, 1, 1)), "sample_rate": tts.sr if tts else 16000}
return (empty_audio, empty_audio, empty_audio, empty_audio, empty_audio, "No valid dialog lines found.")
combined_waveform = torch.cat(combined_dialog_waveforms, dim=-1)
speaker_a_track = torch.cat(speaker_a_waveforms, dim=-1)
speaker_b_track = torch.cat(speaker_b_waveforms, dim=-1)
speaker_c_track = torch.cat(speaker_c_waveforms, dim=-1)
speaker_d_track = torch.cat(speaker_d_waveforms, dim=-1)
dialog_audio = {"waveform": combined_waveform.unsqueeze(0), "sample_rate": tts.sr}
speaker_a_audio = {"waveform": speaker_a_track.unsqueeze(0), "sample_rate": tts.sr}
speaker_b_audio = {"waveform": speaker_b_track.unsqueeze(0), "sample_rate": tts.sr}
speaker_c_audio = {"waveform": speaker_c_track.unsqueeze(0), "sample_rate": tts.sr}
speaker_d_audio = {"waveform": speaker_d_track.unsqueeze(0), "sample_rate": tts.sr}
for f in temp_files:
os.unlink(f)
pbar.update_absolute(100)
return (dialog_audio, speaker_a_audio, speaker_b_audio, speaker_c_audio, speaker_d_audio, "Dialog synthesized successfully.")
NODE_CLASS_MAPPINGS = {
"FL_ChatterboxDialogTTS": FL_ChatterboxDialogTTSNode,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"FL_ChatterboxDialogTTS": "FL Chatterbox Dialog TTS",
}
+30 -2
View File
@@ -57,6 +57,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
},
"optional": {
"audio_prompt": ("AUDIO",),
@@ -70,7 +71,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
FUNCTION = "generate_speech"
CATEGORY = "ChatterBox"
def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
def generate_speech(self, text, exaggeration, cfg_weight, temperature, seed, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
"""
Generate speech from text.
@@ -79,6 +80,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
exaggeration: Controls emotion intensity (0.25-2.0).
cfg_weight: Controls pace/classifier-free guidance (0.2-1.0).
temperature: Controls randomness in generation (0.05-5.0).
seed: Random seed for reproducible generation.
audio_prompt: AUDIO object containing the reference voice for TTS voice cloning.
use_cpu: If True, forces CPU usage even if CUDA is available.
keep_model_loaded: If True, keeps the model loaded in memory after generation.
@@ -86,6 +88,18 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
Returns:
Tuple of (audio, message)
"""
# Set random seeds for reproducibility
torch.manual_seed(seed)
if torch.cuda.is_available():
torch.cuda.manual_seed(seed)
torch.cuda.manual_seed_all(seed)
if torch.backends.mps.is_available():
torch.mps.manual_seed(seed)
import numpy as np
import random
np.random.seed(seed)
random.seed(seed)
# Determine device to use
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
if use_cpu:
@@ -291,6 +305,7 @@ class FL_ChatterboxVCNode(AudioNodeBase):
"required": {
"input_audio": ("AUDIO",),
"target_voice": ("AUDIO",),
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
},
"optional": {
"use_cpu": ("BOOLEAN", {"default": False}),
@@ -303,19 +318,32 @@ class FL_ChatterboxVCNode(AudioNodeBase):
FUNCTION = "convert_voice"
CATEGORY = "ChatterBox"
def convert_voice(self, input_audio, target_voice, use_cpu=False, keep_model_loaded=False):
def convert_voice(self, input_audio, target_voice, seed, use_cpu=False, keep_model_loaded=False):
"""
Convert the voice in an audio file to match a target voice.
Args:
input_audio: AUDIO object containing the audio to convert.
target_voice: AUDIO object containing the target voice.
seed: Random seed for reproducible generation.
use_cpu: If True, forces CPU usage even if CUDA is available.
keep_model_loaded: If True, keeps the model loaded in memory after conversion.
Returns:
Tuple of (audio, message)
"""
# Set random seeds for reproducibility
torch.manual_seed(seed)
if torch.cuda.is_available():
torch.cuda.manual_seed(seed)
torch.cuda.manual_seed_all(seed)
if torch.backends.mps.is_available():
torch.mps.manual_seed(seed)
import numpy as np
import random
np.random.seed(seed)
random.seed(seed)
# Determine device to use
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
if use_cpu:
+4 -2
View File
@@ -4,7 +4,9 @@ librosa
s3tokenizer
transformers
diffusers
resemble-perth
# Optional watermarking (may have Python 3.12+ compatibility issues)
# resemble-perth
omegaconf
conformer
safetensors
safetensors
soundfile
BIN
View File
Binary file not shown.

Before

Width:  |  Height:  |  Size: 291 KiB

After

Width:  |  Height:  |  Size: 816 KiB