added multi dialog node
This commit is contained in:
@@ -1,5 +1,12 @@
|
||||
# ComfyUI_Fill-ChatterBox
|
||||
|
||||
If you enjoy this project, consider supporting me on Patreon!
|
||||
<p align="left">
|
||||
<a href="https://www.patreon.com/c/Machinedelusions">
|
||||
<img src="assets/Patreon.png" width="150px" alt="Patreon">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
A custom node extension for ComfyUI that adds text-to-speech (TTS) and voice conversion (VC) capabilities using the Chatterbox library.
|
||||
Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the model falls apart really badly with anything longer than that, so it remains.
|
||||
|
||||
@@ -18,6 +25,12 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
|
||||
pip install -r ComfyUI_Fill-ChatterBox/requirements.txt
|
||||
```
|
||||
|
||||
3. (Optional) Install watermarking support:
|
||||
```bash
|
||||
pip install resemble-perth
|
||||
```
|
||||
**Note**: The `resemble-perth` package may have compatibility issues with Python 3.12+. If you encounter import errors, the nodes will still function without watermarking.
|
||||
|
||||
|
||||
## Usage
|
||||
|
||||
@@ -31,8 +44,50 @@ Supports a MAXIMUM of 40 seconds. Iv tried removing this limitation, but the mod
|
||||
- Connect input audio and target voice
|
||||
- Both nodes support CPU fallback if CUDA errors occur
|
||||
|
||||
### Dialog TTS Node (FL Chatterbox Dialog TTS)
|
||||
- Add the "FL Chatterbox Dialog TTS" node to your workflow.
|
||||
- This node is designed to synthesize speech for dialogs with up to 4 distinct speakers (SPEAKER A, SPEAKER B, SPEAKER C, and SPEAKER D).
|
||||
- **Inputs:**
|
||||
- `dialog_text`: A multiline string where each line is prefixed by `SPEAKER A:`, `SPEAKER B:`, `SPEAKER C:`, or `SPEAKER D:`. For example:
|
||||
```
|
||||
SPEAKER A: Hello, how are you?
|
||||
SPEAKER B: I am fine, thank you!
|
||||
SPEAKER C: What about you, Speaker D?
|
||||
SPEAKER D: I'm doing great as well!
|
||||
SPEAKER A: That's wonderful to hear.
|
||||
```
|
||||
- `speaker_a_prompt`: An audio prompt (AUDIO type) for SPEAKER A's voice (required).
|
||||
- `speaker_b_prompt`: An audio prompt (AUDIO type) for SPEAKER B's voice (required).
|
||||
- `speaker_c_prompt`: An audio prompt (AUDIO type) for SPEAKER C's voice (optional).
|
||||
- `speaker_d_prompt`: An audio prompt (AUDIO type) for SPEAKER D's voice (optional).
|
||||
- `exaggeration`: Controls emotion intensity (0.25-2.0).
|
||||
- `cfg_weight`: Controls pace/classifier-free guidance (0.2-1.0).
|
||||
- `temperature`: Controls randomness in generation (0.05-5.0).
|
||||
- `use_cpu` (optional): Boolean, defaults to False. Forces CPU usage.
|
||||
- `keep_model_loaded` (optional): Boolean, defaults to False. Keeps the model loaded in memory.
|
||||
- **Outputs:**
|
||||
- `dialog_audio`: Combined audio with all speakers
|
||||
- `speaker_a_audio`: Isolated track for Speaker A (with silence for other speakers)
|
||||
- `speaker_b_audio`: Isolated track for Speaker B (with silence for other speakers)
|
||||
- `speaker_c_audio`: Isolated track for Speaker C (with silence for other speakers)
|
||||
- `speaker_d_audio`: Isolated track for Speaker D (with silence for other speakers)
|
||||
- **Note:** SPEAKER C and SPEAKER D are optional. If their audio prompts are not provided, any dialog lines with "SPEAKER C:" or "SPEAKER D:" will be skipped.
|
||||
|
||||
## Change Log
|
||||
|
||||
### 7/24/2025
|
||||
- Added Dialog TTS node that handles up to 4 speakers (A, B, C, D) for conversation-style audio creation
|
||||
- Extended all nodes (TTS, VC, Dialog) with seed parameters for reproducible generation
|
||||
- SPEAKER C and SPEAKER D are optional in Dialog node - allows flexible 2-4 speaker conversations
|
||||
- Each speaker gets isolated audio track output for advanced audio editing workflows
|
||||
|
||||
### 6/24/2025
|
||||
- Added seed parameter to both TTS and VC nodes for reproducible generation
|
||||
- Seed range: 0 to 4,294,967,295 (32-bit integer)
|
||||
- Enables consistent audio output for debugging and workflow control
|
||||
- Made Perth watermarking optional to fix Python 3.12+ compatibility issues
|
||||
- Nodes now function without watermarking if resemble-perth import fails
|
||||
|
||||
### 5/31/2025
|
||||
- Added Persistent model loading, and loading bar functionality
|
||||
- Added Mac support (needs to be tested so HMU)
|
||||
|
||||
+10
-1
@@ -1,4 +1,13 @@
|
||||
from .chatterbox_node import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
from .chatterbox_node import NODE_CLASS_MAPPINGS as BASE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as BASE_DISPLAY_NAME_MAPPINGS
|
||||
from .chatterbox_dialog_node import NODE_CLASS_MAPPINGS as DIALOG_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as DIALOG_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
NODE_CLASS_MAPPINGS = {}
|
||||
NODE_CLASS_MAPPINGS.update(BASE_CLASS_MAPPINGS)
|
||||
NODE_CLASS_MAPPINGS.update(DIALOG_CLASS_MAPPINGS)
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {}
|
||||
NODE_DISPLAY_NAME_MAPPINGS.update(BASE_DISPLAY_NAME_MAPPINGS)
|
||||
NODE_DISPLAY_NAME_MAPPINGS.update(DIALOG_DISPLAY_NAME_MAPPINGS)
|
||||
|
||||
WEB_DIRECTORY = "./web"
|
||||
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS", "WEB_DIRECTORY"]
|
||||
@@ -0,0 +1,197 @@
|
||||
import os
|
||||
import torch
|
||||
import torchaudio
|
||||
import tempfile
|
||||
|
||||
from .local_chatterbox.chatterbox.tts import ChatterboxTTS
|
||||
from comfy.utils import ProgressBar
|
||||
|
||||
class FL_ChatterboxDialogTTSNode:
|
||||
"""
|
||||
TTS Node that accepts dialog with speaker labels and generates audio using separate voice prompts.
|
||||
"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"dialog_text": ("STRING", {"multiline": True, "default": "SPEAKER A: Test test\nSPEAKER B: 1 2 3"}),
|
||||
"speaker_A_Audio": ("AUDIO",),
|
||||
"speaker_B_Audio": ("AUDIO",),
|
||||
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
|
||||
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
|
||||
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
|
||||
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
|
||||
},
|
||||
"optional": {
|
||||
"speaker_C_Audio": ("AUDIO",),
|
||||
"speaker_D_Audio": ("AUDIO",),
|
||||
"use_cpu": ("BOOLEAN", {"default": False}),
|
||||
"keep_model_loaded": ("BOOLEAN", {"default": False}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("AUDIO", "AUDIO", "AUDIO", "AUDIO", "AUDIO", "STRING")
|
||||
RETURN_NAMES = ("dialog_audio", "speaker_a_audio", "speaker_b_audio", "speaker_c_audio", "speaker_d_audio", "message")
|
||||
FUNCTION = "generate_dialog"
|
||||
CATEGORY = "ChatterBox"
|
||||
|
||||
_model = None
|
||||
_device = None
|
||||
|
||||
def generate_dialog(self, dialog_text, speaker_A_Audio, speaker_B_Audio,
|
||||
exaggeration, cfg_weight, temperature, seed,
|
||||
speaker_C_Audio=None, speaker_D_Audio=None,
|
||||
use_cpu=False, keep_model_loaded=False):
|
||||
# Set random seeds for reproducibility
|
||||
torch.manual_seed(seed)
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.manual_seed(seed)
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
if torch.backends.mps.is_available():
|
||||
torch.mps.manual_seed(seed)
|
||||
import numpy as np
|
||||
import random
|
||||
np.random.seed(seed)
|
||||
random.seed(seed)
|
||||
|
||||
device = "cpu" if use_cpu else ("cuda" if torch.cuda.is_available() else "mps" if torch.backends.mps.is_available() else "cpu")
|
||||
pbar = ProgressBar(100)
|
||||
message = f"Running on {device}"
|
||||
|
||||
def save_temp_audio(audio_data):
|
||||
path = tempfile.NamedTemporaryFile(suffix='.wav', delete=False).name
|
||||
torchaudio.save(path, audio_data['waveform'].squeeze(0), audio_data['sample_rate'])
|
||||
return path
|
||||
|
||||
prompt_a_path = save_temp_audio(speaker_A_Audio)
|
||||
prompt_b_path = save_temp_audio(speaker_B_Audio)
|
||||
temp_files = [prompt_a_path, prompt_b_path]
|
||||
|
||||
# Handle optional speakers C and D
|
||||
prompt_c_path = None
|
||||
prompt_d_path = None
|
||||
if speaker_C_Audio is not None:
|
||||
prompt_c_path = save_temp_audio(speaker_C_Audio)
|
||||
temp_files.append(prompt_c_path)
|
||||
if speaker_D_Audio is not None:
|
||||
prompt_d_path = save_temp_audio(speaker_D_Audio)
|
||||
temp_files.append(prompt_d_path)
|
||||
|
||||
if self._model is None or self._device != device:
|
||||
self._model = ChatterboxTTS.from_pretrained(device=device)
|
||||
self._device = device
|
||||
tts = self._model
|
||||
|
||||
lines = dialog_text.strip().splitlines()
|
||||
speaker_a_waveforms = []
|
||||
speaker_b_waveforms = []
|
||||
speaker_c_waveforms = []
|
||||
speaker_d_waveforms = []
|
||||
combined_dialog_waveforms = []
|
||||
|
||||
for i, line in enumerate(lines):
|
||||
wav = None
|
||||
if line.startswith("SPEAKER A:"):
|
||||
content = line[len("SPEAKER A:"):].strip()
|
||||
prompt_path = prompt_a_path
|
||||
pbar.update_absolute(int((i / len(lines)) * 80))
|
||||
current_speaker_wav = tts.generate(
|
||||
text=content,
|
||||
audio_prompt_path=prompt_path,
|
||||
exaggeration=exaggeration,
|
||||
cfg_weight=cfg_weight,
|
||||
temperature=temperature
|
||||
)
|
||||
speaker_a_waveforms.append(current_speaker_wav)
|
||||
combined_dialog_waveforms.append(current_speaker_wav)
|
||||
# Add silence to other speakers' tracks
|
||||
silence = torch.zeros_like(current_speaker_wav)
|
||||
speaker_b_waveforms.append(silence)
|
||||
speaker_c_waveforms.append(silence)
|
||||
speaker_d_waveforms.append(silence)
|
||||
elif line.startswith("SPEAKER B:"):
|
||||
content = line[len("SPEAKER B:"):].strip()
|
||||
prompt_path = prompt_b_path
|
||||
pbar.update_absolute(int((i / len(lines)) * 80))
|
||||
current_speaker_wav = tts.generate(
|
||||
text=content,
|
||||
audio_prompt_path=prompt_path,
|
||||
exaggeration=exaggeration,
|
||||
cfg_weight=cfg_weight,
|
||||
temperature=temperature
|
||||
)
|
||||
speaker_b_waveforms.append(current_speaker_wav)
|
||||
combined_dialog_waveforms.append(current_speaker_wav)
|
||||
# Add silence to other speakers' tracks
|
||||
silence = torch.zeros_like(current_speaker_wav)
|
||||
speaker_a_waveforms.append(silence)
|
||||
speaker_c_waveforms.append(silence)
|
||||
speaker_d_waveforms.append(silence)
|
||||
elif line.startswith("SPEAKER C:") and prompt_c_path is not None:
|
||||
content = line[len("SPEAKER C:"):].strip()
|
||||
prompt_path = prompt_c_path
|
||||
pbar.update_absolute(int((i / len(lines)) * 80))
|
||||
current_speaker_wav = tts.generate(
|
||||
text=content,
|
||||
audio_prompt_path=prompt_path,
|
||||
exaggeration=exaggeration,
|
||||
cfg_weight=cfg_weight,
|
||||
temperature=temperature
|
||||
)
|
||||
speaker_c_waveforms.append(current_speaker_wav)
|
||||
combined_dialog_waveforms.append(current_speaker_wav)
|
||||
# Add silence to other speakers' tracks
|
||||
silence = torch.zeros_like(current_speaker_wav)
|
||||
speaker_a_waveforms.append(silence)
|
||||
speaker_b_waveforms.append(silence)
|
||||
speaker_d_waveforms.append(silence)
|
||||
elif line.startswith("SPEAKER D:") and prompt_d_path is not None:
|
||||
content = line[len("SPEAKER D:"):].strip()
|
||||
prompt_path = prompt_d_path
|
||||
pbar.update_absolute(int((i / len(lines)) * 80))
|
||||
current_speaker_wav = tts.generate(
|
||||
text=content,
|
||||
audio_prompt_path=prompt_path,
|
||||
exaggeration=exaggeration,
|
||||
cfg_weight=cfg_weight,
|
||||
temperature=temperature
|
||||
)
|
||||
speaker_d_waveforms.append(current_speaker_wav)
|
||||
combined_dialog_waveforms.append(current_speaker_wav)
|
||||
# Add silence to other speakers' tracks
|
||||
silence = torch.zeros_like(current_speaker_wav)
|
||||
speaker_a_waveforms.append(silence)
|
||||
speaker_b_waveforms.append(silence)
|
||||
speaker_c_waveforms.append(silence)
|
||||
else:
|
||||
continue # skip malformed line or missing prompt
|
||||
|
||||
if not combined_dialog_waveforms:
|
||||
empty_audio = {"waveform": torch.zeros((1, 1, 1)), "sample_rate": tts.sr if tts else 16000}
|
||||
return (empty_audio, empty_audio, empty_audio, empty_audio, empty_audio, "No valid dialog lines found.")
|
||||
|
||||
combined_waveform = torch.cat(combined_dialog_waveforms, dim=-1)
|
||||
speaker_a_track = torch.cat(speaker_a_waveforms, dim=-1)
|
||||
speaker_b_track = torch.cat(speaker_b_waveforms, dim=-1)
|
||||
speaker_c_track = torch.cat(speaker_c_waveforms, dim=-1)
|
||||
speaker_d_track = torch.cat(speaker_d_waveforms, dim=-1)
|
||||
|
||||
dialog_audio = {"waveform": combined_waveform.unsqueeze(0), "sample_rate": tts.sr}
|
||||
speaker_a_audio = {"waveform": speaker_a_track.unsqueeze(0), "sample_rate": tts.sr}
|
||||
speaker_b_audio = {"waveform": speaker_b_track.unsqueeze(0), "sample_rate": tts.sr}
|
||||
speaker_c_audio = {"waveform": speaker_c_track.unsqueeze(0), "sample_rate": tts.sr}
|
||||
speaker_d_audio = {"waveform": speaker_d_track.unsqueeze(0), "sample_rate": tts.sr}
|
||||
|
||||
for f in temp_files:
|
||||
os.unlink(f)
|
||||
|
||||
pbar.update_absolute(100)
|
||||
return (dialog_audio, speaker_a_audio, speaker_b_audio, speaker_c_audio, speaker_d_audio, "Dialog synthesized successfully.")
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"FL_ChatterboxDialogTTS": FL_ChatterboxDialogTTSNode,
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"FL_ChatterboxDialogTTS": "FL Chatterbox Dialog TTS",
|
||||
}
|
||||
+30
-2
@@ -57,6 +57,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
|
||||
"exaggeration": ("FLOAT", {"default": 0.5, "min": 0.25, "max": 2.0, "step": 0.05}),
|
||||
"cfg_weight": ("FLOAT", {"default": 0.5, "min": 0.2, "max": 1.0, "step": 0.05}),
|
||||
"temperature": ("FLOAT", {"default": 0.8, "min": 0.05, "max": 5.0, "step": 0.05}),
|
||||
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
|
||||
},
|
||||
"optional": {
|
||||
"audio_prompt": ("AUDIO",),
|
||||
@@ -70,7 +71,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
|
||||
FUNCTION = "generate_speech"
|
||||
CATEGORY = "ChatterBox"
|
||||
|
||||
def generate_speech(self, text, exaggeration, cfg_weight, temperature, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
|
||||
def generate_speech(self, text, exaggeration, cfg_weight, temperature, seed, audio_prompt=None, use_cpu=False, keep_model_loaded=False):
|
||||
"""
|
||||
Generate speech from text.
|
||||
|
||||
@@ -79,6 +80,7 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
|
||||
exaggeration: Controls emotion intensity (0.25-2.0).
|
||||
cfg_weight: Controls pace/classifier-free guidance (0.2-1.0).
|
||||
temperature: Controls randomness in generation (0.05-5.0).
|
||||
seed: Random seed for reproducible generation.
|
||||
audio_prompt: AUDIO object containing the reference voice for TTS voice cloning.
|
||||
use_cpu: If True, forces CPU usage even if CUDA is available.
|
||||
keep_model_loaded: If True, keeps the model loaded in memory after generation.
|
||||
@@ -86,6 +88,18 @@ class FL_ChatterboxTTSNode(AudioNodeBase):
|
||||
Returns:
|
||||
Tuple of (audio, message)
|
||||
"""
|
||||
# Set random seeds for reproducibility
|
||||
torch.manual_seed(seed)
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.manual_seed(seed)
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
if torch.backends.mps.is_available():
|
||||
torch.mps.manual_seed(seed)
|
||||
import numpy as np
|
||||
import random
|
||||
np.random.seed(seed)
|
||||
random.seed(seed)
|
||||
|
||||
# Determine device to use
|
||||
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
|
||||
if use_cpu:
|
||||
@@ -291,6 +305,7 @@ class FL_ChatterboxVCNode(AudioNodeBase):
|
||||
"required": {
|
||||
"input_audio": ("AUDIO",),
|
||||
"target_voice": ("AUDIO",),
|
||||
"seed": ("INT", {"default": 0, "min": 0, "max": 4294967295}),
|
||||
},
|
||||
"optional": {
|
||||
"use_cpu": ("BOOLEAN", {"default": False}),
|
||||
@@ -303,19 +318,32 @@ class FL_ChatterboxVCNode(AudioNodeBase):
|
||||
FUNCTION = "convert_voice"
|
||||
CATEGORY = "ChatterBox"
|
||||
|
||||
def convert_voice(self, input_audio, target_voice, use_cpu=False, keep_model_loaded=False):
|
||||
def convert_voice(self, input_audio, target_voice, seed, use_cpu=False, keep_model_loaded=False):
|
||||
"""
|
||||
Convert the voice in an audio file to match a target voice.
|
||||
|
||||
Args:
|
||||
input_audio: AUDIO object containing the audio to convert.
|
||||
target_voice: AUDIO object containing the target voice.
|
||||
seed: Random seed for reproducible generation.
|
||||
use_cpu: If True, forces CPU usage even if CUDA is available.
|
||||
keep_model_loaded: If True, keeps the model loaded in memory after conversion.
|
||||
|
||||
Returns:
|
||||
Tuple of (audio, message)
|
||||
"""
|
||||
# Set random seeds for reproducibility
|
||||
torch.manual_seed(seed)
|
||||
if torch.cuda.is_available():
|
||||
torch.cuda.manual_seed(seed)
|
||||
torch.cuda.manual_seed_all(seed)
|
||||
if torch.backends.mps.is_available():
|
||||
torch.mps.manual_seed(seed)
|
||||
import numpy as np
|
||||
import random
|
||||
np.random.seed(seed)
|
||||
random.seed(seed)
|
||||
|
||||
# Determine device to use
|
||||
device = "cpu" if use_cpu else ("mps" if torch.backends.mps.is_available() else ("cuda" if torch.cuda.is_available() else "cpu"))
|
||||
if use_cpu:
|
||||
|
||||
+4
-2
@@ -4,7 +4,9 @@ librosa
|
||||
s3tokenizer
|
||||
transformers
|
||||
diffusers
|
||||
resemble-perth
|
||||
# Optional watermarking (may have Python 3.12+ compatibility issues)
|
||||
# resemble-perth
|
||||
omegaconf
|
||||
conformer
|
||||
safetensors
|
||||
safetensors
|
||||
soundfile
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 291 KiB After Width: | Height: | Size: 816 KiB |
Reference in New Issue
Block a user