update readme

This commit is contained in:
billwuhao
2025-03-15 22:48:42 +08:00
parent a62d75f754
commit 6f5bcaa698
8 changed files with 52 additions and 34 deletions
+12
View File
@@ -38,6 +38,18 @@ You can then freely customize speakers under the `ComfyUI\models\TTS\Step-Audio-
[2025-02-25]⚒️: Support custom speaker `custom_stpeaker`.
## Installation
```
cd ComfyUI/custom_nodes
git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git
cd ComfyUI_StepAudioTTS
pip install -r requirements.txt
# python_embeded
./python_embeded/python.exe -m pip install -r requirements.txt
```
## Model Download
Download to the `ComfyUI\models\TTS` folder
+11
View File
@@ -38,6 +38,17 @@ ComfyUI\models\TTS
[2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`.
## 安装
```
cd ComfyUI/custom_nodes
git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git
cd ComfyUI_StepAudioTTS
pip install -r requirements.txt
# python_embeded
./python_embeded/python.exe -m pip install -r requirements.txt
```
## 模型下载
+4
View File
@@ -7,6 +7,10 @@ import numpy as np
from transformers import AutoModelForCausalLM, AutoTokenizer
from transformers.generation.logits_process import LogitsProcessor
from transformers.generation.utils import LogitsProcessorList
import sys
current_dir = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, current_dir)
from tokenizer import StepAudioTokenizer
from cosyvoice.cli.cosyvoice import CosyVoice
+1 -7
View File
@@ -1,9 +1,3 @@
import sys
import os
current_dir = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, current_dir)
from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
from .StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
+6 -5
View File
@@ -20,11 +20,6 @@ from funasr_detach.utils.load_utils import load_audio_text_image_video
from funasr_detach.utils.timestamp_tools import timestamp_sentence
from funasr_detach.models.campplus.utils import sv_chunk, postprocess, distribute_spk
try:
from funasr_detach.models.campplus.cluster_backend import ClusterBackend
except:
print("If you want to use the speaker diarization, please `pip install hdbscan`")
def prepare_data_iterator(data_in, input_len=None, data_type=None, key=None):
"""
@@ -134,6 +129,12 @@ class AutoModel:
# if spk_model is not None, build spk model else None
spk_model = kwargs.get("spk_model", None)
spk_kwargs = kwargs.get("spk_model_revision", None)
try:
from funasr_detach.models.campplus.cluster_backend import ClusterBackend
except:
print("If you want to use the speaker diarization, please `pip install hdbscan`")
if spk_model is not None:
logging.info("Building SPK model.")
spk_kwargs = {
@@ -2,12 +2,6 @@ import torch
import torch.nn as nn
import torch.nn.functional as F
try:
from rotary_embedding_torch import RotaryEmbedding
except:
print(
"If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch"
)
from funasr_detach.models.transformer.layer_norm import (
GlobalLayerNorm,
CumulativeLayerNorm,
@@ -57,6 +51,12 @@ class MossformerBlock(nn.Module):
self.group_size = group_size
try:
from rotary_embedding_torch import RotaryEmbedding
except:
print(
"If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch"
)
rotary_pos_emb = RotaryEmbedding(dim=min(32, query_key_dim))
# max rotary embedding dimensions of 32, partial Rotary embeddings, from Wang et al - GPT-J
self.layers = nn.ModuleList(
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "stepaudiotts_mw"
description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice."
version = "1.1.4"
version = "1.1.5"
license = {file = "LICENSE"}
dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"]
+11 -15
View File
@@ -1,18 +1,14 @@
# torch==2.3.1
# torchaudio==2.3.1
# torchvision==0.18.1
# accelerate==1.3.0
# onnxruntime-gpu==1.17.0
# omegaconf==2.3.0
# librosa==0.10.2.post1
# modelscope
# numpy==1.26.4
# six==1.16.0
# diffusers
# pillow
# sentencepiece
# protobuf==5.29.3
accelerate
onnxruntime-gpu
omegaconf
librosa>=0.10.2.post1
modelscope
numpy
six
diffusers
pillow
sentencepiece
protobuf
transformers
openai-whisper
sox