v1.5 supported

This commit is contained in:
billwuhao
2025-05-14 01:38:09 +08:00
parent c22ea93124
commit 88f8047e45
5 changed files with 131 additions and 2 deletions
+5
View File
@@ -8,6 +8,11 @@
## 📣 更新
[2025-05-14]⚒️: 支持 v1.5 版本. 模型下载并更名放到 `ComfyUI\models\TTS\Index-TTS` 路径下:
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bigvgan_generator.pth → `bigvgan_generator_v1_5.pth`
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bpe.model → `bpe_v1_5.model`
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/gpt.pth → `gpt_v1_5.pth`
[2025-05-02]⚒️: 可用 DeepSpeed 加速, 需要安装 DeepSpeed, Windows 详见 [DeepSpeed 安装](https://github.com/deepspeedai/DeepSpeed/blob/master/blogs/windows/08-2024/chinese/README.md). 加速不明显.
[2025-04-30]⚒️: 发布 v1.0.0.
+5
View File
@@ -6,6 +6,11 @@ High-quality voice cloning, very fast, supports Chinese and English, and allows
## 📣 Updates
[2025-05-14]⚒️: Supports v1.5. Download the models and rename they, placed in the `ComfyUI\models\TTS\Index-TTS` path.
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bigvgan_generator.pth → `bigvgan_generator_v1_5.pth`
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bpe.model → `bpe_v1_5.model`
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/gpt.pth → `gpt_v1_5.pth`
[2025-05-02] ⚒️: DeepSpeed acceleration is available, but DeepSpeed needs to be installed. For Windows, please refer to [DeepSpeed Installation](https://github.com/deepspeedai/DeepSpeed/blob/master/blogs/windows/08-2024/chinese/README.md). The acceleration is not obvious.
[2025-04-30] ⚒️: Released v1.0.0.
+112
View File
@@ -0,0 +1,112 @@
dataset:
bpe_model: bpe_v1_5.model
sample_rate: 24000
squeeze: false
mel:
sample_rate: 24000
n_fft: 1024
hop_length: 256
win_length: 1024
n_mels: 100
mel_fmin: 0
normalize: false
gpt:
model_dim: 1280
max_mel_tokens: 800
max_text_tokens: 600
heads: 20
use_mel_codes_as_input: true
mel_length_compression: 1024
layers: 24
number_text_tokens: 12000
number_mel_codes: 8194
start_mel_token: 8192
stop_mel_token: 8193
start_text_token: 0
stop_text_token: 1
train_solo_embeddings: false
condition_type: "conformer_perceiver"
condition_module:
output_size: 512
linear_units: 2048
attention_heads: 8
num_blocks: 6
input_layer: "conv2d2"
perceiver_mult: 2
vqvae:
channels: 100
num_tokens: 8192
hidden_dim: 512
num_resnet_blocks: 3
codebook_dim: 512
num_layers: 2
positional_dims: 1
kernel_size: 3
smooth_l1_loss: true
use_transposed_convs: false
bigvgan:
adam_b1: 0.8
adam_b2: 0.99
lr_decay: 0.999998
seed: 1234
resblock: "1"
upsample_rates: [4,4,4,4,2,2]
upsample_kernel_sizes: [8,8,4,4,4,4]
upsample_initial_channel: 1536
resblock_kernel_sizes: [3,7,11]
resblock_dilation_sizes: [[1,3,5], [1,3,5], [1,3,5]]
feat_upsample: false
speaker_embedding_dim: 512
cond_d_vector_in_each_upsampling_layer: true
gpt_dim: 1280
activation: "snakebeta"
snake_logscale: true
use_cqtd_instead_of_mrd: true
cqtd_filters: 128
cqtd_max_filters: 1024
cqtd_filters_scale: 1
cqtd_dilations: [1, 2, 4]
cqtd_hop_lengths: [512, 256, 256]
cqtd_n_octaves: [9, 9, 9]
cqtd_bins_per_octaves: [24, 36, 48]
resolutions: [[1024, 120, 600], [2048, 240, 1200], [512, 50, 240]]
mpd_reshapes: [2, 3, 5, 7, 11]
use_spectral_norm: false
discriminator_channel_mult: 1
use_multiscale_melloss: true
lambda_melloss: 15
clip_grad_norm: 1000
segment_size: 16384
num_mels: 100
num_freq: 1025
n_fft: 1024
hop_size: 256
win_size: 1024
sampling_rate: 24000
fmin: 0
fmax: null
fmax_for_loss: null
mel_type: "pytorch"
num_workers: 2
dist_config:
dist_backend: "nccl"
dist_url: "tcp://localhost:54321"
world_size: 1
dvae_checkpoint: dvae.pth
gpt_checkpoint: gpt_v1_5.pth
bigvgan_checkpoint: bigvgan_generator_v1_5.pth
+8 -1
View File
@@ -507,6 +507,7 @@ class IndexTTSRun:
def INPUT_TYPES(s):
return {
"required": {
"version":(["v1.5", "V1.0"], {"default": "v1.5"}),
"audio_prompt":("AUDIO",),
"text": ("STRING", {"forceInput": True}),
"text_language": (["zh", "en"], {"default": "zh"}),
@@ -526,6 +527,7 @@ class IndexTTSRun:
CATEGORY = "🎤MW/MW-IndexTTS"
def clone(self,
version,
audio_prompt,
text,
text_language,
@@ -537,8 +539,13 @@ class IndexTTSRun:
fast_inference=True,
unload_model=True
):
if version == "v1.5":
cfg_path=f"{current_dir}/checkpoints/config_v1_5.yaml"
else:
cfg_path=f"{current_dir}/checkpoints/config.yaml"
if self.index_tts is None:
self.index_tts = IndexTTS(text_language=text_language)
self.index_tts = IndexTTS(cfg_path=cfg_path, text_language=text_language)
if fast_inference:
res = self.index_tts.infer_fast(
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "indextts-mw"
description = "IndexTTS Voice Cloning Nodes for ComfyUI. High-quality voice cloning, very fast, supports Chinese and English, and allows custom voice styles."
version = "1.0.2"
version = "1.1.0"
license = {file = "LICENSE"}
dependencies = ["# accelerate==0.25.0", "# transformers==4.36.2", "# tokenizers==0.15.0", "# cn2an==0.5.22", "# ffmpeg-python==0.2.0", "# Cython==3.0.7", "# g2p-en==2.1.0", "# jieba==0.42.1", "# keras==2.9.0", "# numba==0.58.1", "# numpy==1.26.2", "# pandas==2.1.3", "# matplotlib==3.8.2", "# opencv-python==4.9.0.80", "# vocos==0.1.0", "# accelerate==0.25.0", "# tensorboard==2.9.1", "omegaconf", "sentencepiece", "librosa", "tqdm", "# deepspeeds # Use it to accelerate model inference"]