update readme
This commit is contained in:
@@ -38,6 +38,18 @@ You can then freely customize speakers under the `ComfyUI\models\TTS\Step-Audio-
|
||||
|
||||
[2025-02-25]⚒️: Support custom speaker `custom_stpeaker`.
|
||||
|
||||
## Installation
|
||||
|
||||
```
|
||||
cd ComfyUI/custom_nodes
|
||||
git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git
|
||||
cd ComfyUI_StepAudioTTS
|
||||
pip install -r requirements.txt
|
||||
|
||||
# python_embeded
|
||||
./python_embeded/python.exe -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
## Model Download
|
||||
|
||||
Download to the `ComfyUI\models\TTS` folder
|
||||
|
||||
@@ -38,6 +38,17 @@ ComfyUI\models\TTS
|
||||
|
||||
[2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`.
|
||||
|
||||
## 安装
|
||||
|
||||
```
|
||||
cd ComfyUI/custom_nodes
|
||||
git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git
|
||||
cd ComfyUI_StepAudioTTS
|
||||
pip install -r requirements.txt
|
||||
|
||||
# python_embeded
|
||||
./python_embeded/python.exe -m pip install -r requirements.txt
|
||||
```
|
||||
|
||||
## 模型下载
|
||||
|
||||
|
||||
@@ -7,6 +7,10 @@ import numpy as np
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
from transformers.generation.logits_process import LogitsProcessor
|
||||
from transformers.generation.utils import LogitsProcessorList
|
||||
import sys
|
||||
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, current_dir)
|
||||
|
||||
from tokenizer import StepAudioTokenizer
|
||||
from cosyvoice.cli.cosyvoice import CosyVoice
|
||||
|
||||
+1
-7
@@ -1,9 +1,3 @@
|
||||
import sys
|
||||
import os
|
||||
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, current_dir)
|
||||
|
||||
from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
from .StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
|
||||
|
||||
@@ -20,11 +20,6 @@ from funasr_detach.utils.load_utils import load_audio_text_image_video
|
||||
from funasr_detach.utils.timestamp_tools import timestamp_sentence
|
||||
from funasr_detach.models.campplus.utils import sv_chunk, postprocess, distribute_spk
|
||||
|
||||
try:
|
||||
from funasr_detach.models.campplus.cluster_backend import ClusterBackend
|
||||
except:
|
||||
print("If you want to use the speaker diarization, please `pip install hdbscan`")
|
||||
|
||||
|
||||
def prepare_data_iterator(data_in, input_len=None, data_type=None, key=None):
|
||||
"""
|
||||
@@ -134,6 +129,12 @@ class AutoModel:
|
||||
# if spk_model is not None, build spk model else None
|
||||
spk_model = kwargs.get("spk_model", None)
|
||||
spk_kwargs = kwargs.get("spk_model_revision", None)
|
||||
|
||||
try:
|
||||
from funasr_detach.models.campplus.cluster_backend import ClusterBackend
|
||||
except:
|
||||
print("If you want to use the speaker diarization, please `pip install hdbscan`")
|
||||
|
||||
if spk_model is not None:
|
||||
logging.info("Building SPK model.")
|
||||
spk_kwargs = {
|
||||
|
||||
@@ -2,12 +2,6 @@ import torch
|
||||
import torch.nn as nn
|
||||
import torch.nn.functional as F
|
||||
|
||||
try:
|
||||
from rotary_embedding_torch import RotaryEmbedding
|
||||
except:
|
||||
print(
|
||||
"If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch"
|
||||
)
|
||||
from funasr_detach.models.transformer.layer_norm import (
|
||||
GlobalLayerNorm,
|
||||
CumulativeLayerNorm,
|
||||
@@ -57,6 +51,12 @@ class MossformerBlock(nn.Module):
|
||||
|
||||
self.group_size = group_size
|
||||
|
||||
try:
|
||||
from rotary_embedding_torch import RotaryEmbedding
|
||||
except:
|
||||
print(
|
||||
"If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch"
|
||||
)
|
||||
rotary_pos_emb = RotaryEmbedding(dim=min(32, query_key_dim))
|
||||
# max rotary embedding dimensions of 32, partial Rotary embeddings, from Wang et al - GPT-J
|
||||
self.layers = nn.ModuleList(
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "stepaudiotts_mw"
|
||||
description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice."
|
||||
version = "1.1.4"
|
||||
version = "1.1.5"
|
||||
license = {file = "LICENSE"}
|
||||
dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"]
|
||||
|
||||
|
||||
+11
-15
@@ -1,18 +1,14 @@
|
||||
# torch==2.3.1
|
||||
# torchaudio==2.3.1
|
||||
# torchvision==0.18.1
|
||||
# accelerate==1.3.0
|
||||
# onnxruntime-gpu==1.17.0
|
||||
# omegaconf==2.3.0
|
||||
# librosa==0.10.2.post1
|
||||
# modelscope
|
||||
# numpy==1.26.4
|
||||
# six==1.16.0
|
||||
# diffusers
|
||||
# pillow
|
||||
# sentencepiece
|
||||
# protobuf==5.29.3
|
||||
|
||||
accelerate
|
||||
onnxruntime-gpu
|
||||
omegaconf
|
||||
librosa>=0.10.2.post1
|
||||
modelscope
|
||||
numpy
|
||||
six
|
||||
diffusers
|
||||
pillow
|
||||
sentencepiece
|
||||
protobuf
|
||||
transformers
|
||||
openai-whisper
|
||||
sox
|
||||
|
||||
Reference in New Issue
Block a user