v1.5 supported
This commit is contained in:
@@ -8,6 +8,11 @@
|
||||
|
||||
## 📣 更新
|
||||
|
||||
[2025-05-14]⚒️: 支持 v1.5 版本. 模型下载并更名放到 `ComfyUI\models\TTS\Index-TTS` 路径下:
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bigvgan_generator.pth → `bigvgan_generator_v1_5.pth`
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bpe.model → `bpe_v1_5.model`
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/gpt.pth → `gpt_v1_5.pth`
|
||||
|
||||
[2025-05-02]⚒️: 可用 DeepSpeed 加速, 需要安装 DeepSpeed, Windows 详见 [DeepSpeed 安装](https://github.com/deepspeedai/DeepSpeed/blob/master/blogs/windows/08-2024/chinese/README.md). 加速不明显.
|
||||
|
||||
[2025-04-30]⚒️: 发布 v1.0.0.
|
||||
|
||||
@@ -6,6 +6,11 @@ High-quality voice cloning, very fast, supports Chinese and English, and allows
|
||||
|
||||
## 📣 Updates
|
||||
|
||||
[2025-05-14]⚒️: Supports v1.5. Download the models and rename they, placed in the `ComfyUI\models\TTS\Index-TTS` path.
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bigvgan_generator.pth → `bigvgan_generator_v1_5.pth`
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/bpe.model → `bpe_v1_5.model`
|
||||
- https://huggingface.co/IndexTeam/IndexTTS-1.5/blob/main/gpt.pth → `gpt_v1_5.pth`
|
||||
|
||||
[2025-05-02] ⚒️: DeepSpeed acceleration is available, but DeepSpeed needs to be installed. For Windows, please refer to [DeepSpeed Installation](https://github.com/deepspeedai/DeepSpeed/blob/master/blogs/windows/08-2024/chinese/README.md). The acceleration is not obvious.
|
||||
|
||||
[2025-04-30] ⚒️: Released v1.0.0.
|
||||
|
||||
@@ -0,0 +1,112 @@
|
||||
dataset:
|
||||
bpe_model: bpe_v1_5.model
|
||||
sample_rate: 24000
|
||||
squeeze: false
|
||||
mel:
|
||||
sample_rate: 24000
|
||||
n_fft: 1024
|
||||
hop_length: 256
|
||||
win_length: 1024
|
||||
n_mels: 100
|
||||
mel_fmin: 0
|
||||
normalize: false
|
||||
|
||||
gpt:
|
||||
model_dim: 1280
|
||||
max_mel_tokens: 800
|
||||
max_text_tokens: 600
|
||||
heads: 20
|
||||
use_mel_codes_as_input: true
|
||||
mel_length_compression: 1024
|
||||
layers: 24
|
||||
number_text_tokens: 12000
|
||||
number_mel_codes: 8194
|
||||
start_mel_token: 8192
|
||||
stop_mel_token: 8193
|
||||
start_text_token: 0
|
||||
stop_text_token: 1
|
||||
train_solo_embeddings: false
|
||||
condition_type: "conformer_perceiver"
|
||||
condition_module:
|
||||
output_size: 512
|
||||
linear_units: 2048
|
||||
attention_heads: 8
|
||||
num_blocks: 6
|
||||
input_layer: "conv2d2"
|
||||
perceiver_mult: 2
|
||||
|
||||
vqvae:
|
||||
channels: 100
|
||||
num_tokens: 8192
|
||||
hidden_dim: 512
|
||||
num_resnet_blocks: 3
|
||||
codebook_dim: 512
|
||||
num_layers: 2
|
||||
positional_dims: 1
|
||||
kernel_size: 3
|
||||
smooth_l1_loss: true
|
||||
use_transposed_convs: false
|
||||
|
||||
bigvgan:
|
||||
adam_b1: 0.8
|
||||
adam_b2: 0.99
|
||||
lr_decay: 0.999998
|
||||
seed: 1234
|
||||
|
||||
resblock: "1"
|
||||
upsample_rates: [4,4,4,4,2,2]
|
||||
upsample_kernel_sizes: [8,8,4,4,4,4]
|
||||
upsample_initial_channel: 1536
|
||||
resblock_kernel_sizes: [3,7,11]
|
||||
resblock_dilation_sizes: [[1,3,5], [1,3,5], [1,3,5]]
|
||||
feat_upsample: false
|
||||
speaker_embedding_dim: 512
|
||||
cond_d_vector_in_each_upsampling_layer: true
|
||||
|
||||
gpt_dim: 1280
|
||||
|
||||
activation: "snakebeta"
|
||||
snake_logscale: true
|
||||
|
||||
use_cqtd_instead_of_mrd: true
|
||||
cqtd_filters: 128
|
||||
cqtd_max_filters: 1024
|
||||
cqtd_filters_scale: 1
|
||||
cqtd_dilations: [1, 2, 4]
|
||||
cqtd_hop_lengths: [512, 256, 256]
|
||||
cqtd_n_octaves: [9, 9, 9]
|
||||
cqtd_bins_per_octaves: [24, 36, 48]
|
||||
|
||||
resolutions: [[1024, 120, 600], [2048, 240, 1200], [512, 50, 240]]
|
||||
mpd_reshapes: [2, 3, 5, 7, 11]
|
||||
use_spectral_norm: false
|
||||
discriminator_channel_mult: 1
|
||||
|
||||
use_multiscale_melloss: true
|
||||
lambda_melloss: 15
|
||||
|
||||
clip_grad_norm: 1000
|
||||
|
||||
segment_size: 16384
|
||||
num_mels: 100
|
||||
num_freq: 1025
|
||||
n_fft: 1024
|
||||
hop_size: 256
|
||||
win_size: 1024
|
||||
|
||||
sampling_rate: 24000
|
||||
|
||||
fmin: 0
|
||||
fmax: null
|
||||
fmax_for_loss: null
|
||||
mel_type: "pytorch"
|
||||
|
||||
num_workers: 2
|
||||
dist_config:
|
||||
dist_backend: "nccl"
|
||||
dist_url: "tcp://localhost:54321"
|
||||
world_size: 1
|
||||
|
||||
dvae_checkpoint: dvae.pth
|
||||
gpt_checkpoint: gpt_v1_5.pth
|
||||
bigvgan_checkpoint: bigvgan_generator_v1_5.pth
|
||||
+8
-1
@@ -507,6 +507,7 @@ class IndexTTSRun:
|
||||
def INPUT_TYPES(s):
|
||||
return {
|
||||
"required": {
|
||||
"version":(["v1.5", "V1.0"], {"default": "v1.5"}),
|
||||
"audio_prompt":("AUDIO",),
|
||||
"text": ("STRING", {"forceInput": True}),
|
||||
"text_language": (["zh", "en"], {"default": "zh"}),
|
||||
@@ -526,6 +527,7 @@ class IndexTTSRun:
|
||||
CATEGORY = "🎤MW/MW-IndexTTS"
|
||||
|
||||
def clone(self,
|
||||
version,
|
||||
audio_prompt,
|
||||
text,
|
||||
text_language,
|
||||
@@ -537,8 +539,13 @@ class IndexTTSRun:
|
||||
fast_inference=True,
|
||||
unload_model=True
|
||||
):
|
||||
if version == "v1.5":
|
||||
cfg_path=f"{current_dir}/checkpoints/config_v1_5.yaml"
|
||||
else:
|
||||
cfg_path=f"{current_dir}/checkpoints/config.yaml"
|
||||
|
||||
if self.index_tts is None:
|
||||
self.index_tts = IndexTTS(text_language=text_language)
|
||||
self.index_tts = IndexTTS(cfg_path=cfg_path, text_language=text_language)
|
||||
|
||||
if fast_inference:
|
||||
res = self.index_tts.infer_fast(
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "indextts-mw"
|
||||
description = "IndexTTS Voice Cloning Nodes for ComfyUI. High-quality voice cloning, very fast, supports Chinese and English, and allows custom voice styles."
|
||||
version = "1.0.2"
|
||||
version = "1.1.0"
|
||||
license = {file = "LICENSE"}
|
||||
dependencies = ["# accelerate==0.25.0", "# transformers==4.36.2", "# tokenizers==0.15.0", "# cn2an==0.5.22", "# ffmpeg-python==0.2.0", "# Cython==3.0.7", "# g2p-en==2.1.0", "# jieba==0.42.1", "# keras==2.9.0", "# numba==0.58.1", "# numpy==1.26.2", "# pandas==2.1.3", "# matplotlib==3.8.2", "# opencv-python==4.9.0.80", "# vocos==0.1.0", "# accelerate==0.25.0", "# tensorboard==2.9.1", "omegaconf", "sentencepiece", "librosa", "tqdm", "# deepspeeds # Use it to accelerate model inference"]
|
||||
|
||||
|
||||
Reference in New Issue
Block a user