From 6f5bcaa69809a1faff5cf0f98711e489c204cc06 Mon Sep 17 00:00:00 2001 From: billwuhao Date: Sat, 15 Mar 2025 22:48:42 +0800 Subject: [PATCH] update readme --- README-en.md | 12 +++++++++ README.md | 11 ++++++++ StepAudioTTS.py | 4 +++ __init__.py | 8 +----- funasr_detach/auto/auto_model.py | 11 ++++---- .../models/mossformer/mossformer_encoder.py | 12 ++++----- pyproject.toml | 2 +- requirements.txt | 26 ++++++++----------- 8 files changed, 52 insertions(+), 34 deletions(-) diff --git a/README-en.md b/README-en.md index 4da53b6..2afab20 100644 --- a/README-en.md +++ b/README-en.md @@ -38,6 +38,18 @@ You can then freely customize speakers under the `ComfyUI\models\TTS\Step-Audio- [2025-02-25]⚒️: Support custom speaker `custom_stpeaker`. +## Installation + +``` +cd ComfyUI/custom_nodes +git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git +cd ComfyUI_StepAudioTTS +pip install -r requirements.txt + +# python_embeded +./python_embeded/python.exe -m pip install -r requirements.txt +``` + ## Model Download Download to the `ComfyUI\models\TTS` folder diff --git a/README.md b/README.md index 873c534..0825c4f 100644 --- a/README.md +++ b/README.md @@ -38,6 +38,17 @@ ComfyUI\models\TTS [2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`. +## 安装 + +``` +cd ComfyUI/custom_nodes +git clone https://github.com/billwuhao/ComfyUI_StepAudioTTS.git +cd ComfyUI_StepAudioTTS +pip install -r requirements.txt + +# python_embeded +./python_embeded/python.exe -m pip install -r requirements.txt +``` ## 模型下载 diff --git a/StepAudioTTS.py b/StepAudioTTS.py index b8fb3c8..113b50c 100644 --- a/StepAudioTTS.py +++ b/StepAudioTTS.py @@ -7,6 +7,10 @@ import numpy as np from transformers import AutoModelForCausalLM, AutoTokenizer from transformers.generation.logits_process import LogitsProcessor from transformers.generation.utils import LogitsProcessorList +import sys + +current_dir = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, current_dir) from tokenizer import StepAudioTokenizer from cosyvoice.cli.cosyvoice import CosyVoice diff --git a/__init__.py b/__init__.py index bfdf0d8..16c477d 100644 --- a/__init__.py +++ b/__init__.py @@ -1,9 +1,3 @@ -import sys -import os - -current_dir = os.path.dirname(os.path.abspath(__file__)) -sys.path.insert(0, current_dir) - -from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS +from .StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS __all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] diff --git a/funasr_detach/auto/auto_model.py b/funasr_detach/auto/auto_model.py index 9315da8..7b54892 100644 --- a/funasr_detach/auto/auto_model.py +++ b/funasr_detach/auto/auto_model.py @@ -20,11 +20,6 @@ from funasr_detach.utils.load_utils import load_audio_text_image_video from funasr_detach.utils.timestamp_tools import timestamp_sentence from funasr_detach.models.campplus.utils import sv_chunk, postprocess, distribute_spk -try: - from funasr_detach.models.campplus.cluster_backend import ClusterBackend -except: - print("If you want to use the speaker diarization, please `pip install hdbscan`") - def prepare_data_iterator(data_in, input_len=None, data_type=None, key=None): """ @@ -134,6 +129,12 @@ class AutoModel: # if spk_model is not None, build spk model else None spk_model = kwargs.get("spk_model", None) spk_kwargs = kwargs.get("spk_model_revision", None) + + try: + from funasr_detach.models.campplus.cluster_backend import ClusterBackend + except: + print("If you want to use the speaker diarization, please `pip install hdbscan`") + if spk_model is not None: logging.info("Building SPK model.") spk_kwargs = { diff --git a/funasr_detach/models/mossformer/mossformer_encoder.py b/funasr_detach/models/mossformer/mossformer_encoder.py index 5d960ff..880d745 100644 --- a/funasr_detach/models/mossformer/mossformer_encoder.py +++ b/funasr_detach/models/mossformer/mossformer_encoder.py @@ -2,12 +2,6 @@ import torch import torch.nn as nn import torch.nn.functional as F -try: - from rotary_embedding_torch import RotaryEmbedding -except: - print( - "If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch" - ) from funasr_detach.models.transformer.layer_norm import ( GlobalLayerNorm, CumulativeLayerNorm, @@ -57,6 +51,12 @@ class MossformerBlock(nn.Module): self.group_size = group_size + try: + from rotary_embedding_torch import RotaryEmbedding + except: + print( + "If you want use mossformer, please install rotary_embedding_torch by: \n pip install -U rotary_embedding_torch" + ) rotary_pos_emb = RotaryEmbedding(dim=min(32, query_key_dim)) # max rotary embedding dimensions of 32, partial Rotary embeddings, from Wang et al - GPT-J self.layers = nn.ModuleList( diff --git a/pyproject.toml b/pyproject.toml index 97196a3..4f753cb 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "stepaudiotts_mw" description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice." -version = "1.1.4" +version = "1.1.5" license = {file = "LICENSE"} dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"] diff --git a/requirements.txt b/requirements.txt index d3573fc..82b0fb7 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,18 +1,14 @@ -# torch==2.3.1 -# torchaudio==2.3.1 -# torchvision==0.18.1 -# accelerate==1.3.0 -# onnxruntime-gpu==1.17.0 -# omegaconf==2.3.0 -# librosa==0.10.2.post1 -# modelscope -# numpy==1.26.4 -# six==1.16.0 -# diffusers -# pillow -# sentencepiece -# protobuf==5.29.3 - +accelerate +onnxruntime-gpu +omegaconf +librosa>=0.10.2.post1 +modelscope +numpy +six +diffusers +pillow +sentencepiece +protobuf transformers openai-whisper sox