diff --git a/DiffRhythmNode.py b/DiffRhythmNode.py index 4a31bdc..69d4507 100644 --- a/DiffRhythmNode.py +++ b/DiffRhythmNode.py @@ -19,6 +19,11 @@ import librosa from mutagen.mp3 import MP3 import torch from einops import rearrange +import sys +import os + +current_dir = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, current_dir) from diffrhythm_utils import ( decode_audio, diff --git a/README-CN.md b/README-CN.md new file mode 100644 index 0000000..1764d5e --- /dev/null +++ b/README-CN.md @@ -0,0 +1,64 @@ +[中文](README.md) | [English](README-en.md) + +# DiffRhythm 的 ComfyUI 节点 + +快速而简单的端到端全长歌曲生成. + +![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-12_23-49-32.png) + + +## 📣 更新 + +[2025-03-13]⚒️: 发布版本 v1.0.0. + +- 所有参数均是可选的, 不提供任何参数随机生成音乐. + +## 安装 + +``` +cd ComfyUI/custom_nodes +git clone https://github.com/billwuhao/ComfyUI_DiffRhythm.git +cd ComfyUI_DiffRhythm +pip install -r requirements.txt + +# python_embeded +./python_embeded/python.exe -m pip install -r requirements.txt +``` + +## 模型下载 + +模型会自动下载到 `ComfyUI\models\TTS\DiffRhythm` 文件夹下. + +结构如下: + +![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-13_00-08-51.png) + +手动下载地址: + +https://huggingface.co/ASLP-lab/DiffRhythm-base/blob/main/cfm_model.pt +https://huggingface.co/ASLP-lab/DiffRhythm-vae/blob/main/vae_model.pt +https://huggingface.co/OpenMuQ/MuQ-MuLan-large/tree/main +https://huggingface.co/OpenMuQ/MuQ-large-msd-iter/tree/main +https://huggingface.co/FacebookAI/xlm-roberta-base/tree/main + +## 环境配置 + +Windows 系统做如下配置. + +下载安装最新版 [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases/tag/1.52.0) + +添加环境变量 `PHONEMIZER_ESPEAK_LIBRARY` 到系统中, 值是你安装的 espeak-ng 软件中 `libespeak-ng.dll` 文件的路径, 例如: `C:\Program Files\eSpeak NG\libespeak-ng.dll`. + +Linux 系统下, 需要安装 `espeak-ng` 软件包. 执行如下命令安装: + +`apt-get -qq -y install espeak-ng > /dev/null 2>&1` + +应该支持 Mac, 但尚未测试. + +享受音乐吧🎶 + +## 鸣谢 + +[DiffRhythm](https://github.com/ASLP-lab/DiffRhythm) + +感谢 DiffRhythm 团队的卓越的工作, 目前最强开源 音乐/歌曲 生成模型👍. \ No newline at end of file diff --git a/README-en.md b/README-en.md deleted file mode 100644 index a142ac5..0000000 --- a/README-en.md +++ /dev/null @@ -1,45 +0,0 @@ -[中文](README.md) | [English](README-en.md) - -# DiffRhythm Node for ComfyUI - -Blazingly Fast and Embarrassingly Simple End-to-End Full-Length Song Generation. - -![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-12_23-49-32.png) - -## 📣 update - -[2025-03-13]⚒️: Release version v1.0.0. - -- All parameters are optional; you can generate random music without providing any parameters. - -## Model Download - -Models will be automatically downloaded to the `ComfyUI\models\TTS\DiffRhythm` folder. - -The structure is as follows: - -![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-13_00-08-51.png) - -Manual Download Addresses: - -https://huggingface.co/ASLP-lab/DiffRhythm-base/blob/main/cfm_model.pt -https://huggingface.co/ASLP-lab/DiffRhythm-vae/blob/main/vae_model.pt -https://huggingface.co/OpenMuQ/MuQ-MuLan-large/tree/main -https://huggingface.co/OpenMuQ/MuQ-large-msd-iter/tree/main -https://huggingface.co/FacebookAI/xlm-roberta-base/tree/main - -## Environment Configuration - -Configure the following on Windows systems, other systems have not been tested. Should support Linux and Mac. - -Download and install the latest version of [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases/tag/1.52.0) - -Add the environment variable `PHONEMIZER_ESPEAK_LIBRARY` to your system. The value should be the path to the `libespeak-ng.dll` file in your espeak-ng installation, for example: `C:\Program Files\eSpeak NG\libespeak-ng.dll`. - -Enjoy the music! 🎶 - -## Acknowledgements - -[DiffRhythm](https://github.com/ASLP-lab/DiffRhythm) - -Thanks to the DiffRhythm team for their excellent work. Currently the strongest open-source music/song generation model 👍. diff --git a/README.md b/README.md index 5830634..90a00b9 100644 --- a/README.md +++ b/README.md @@ -1,27 +1,38 @@ [中文](README.md) | [English](README-en.md) -# DiffRhythm 的 ComfyUI 节点 +# DiffRhythm Node for ComfyUI -快速而简单的端到端全长歌曲生成. +Blazingly Fast and Embarrassingly Simple End-to-End Full-Length Song Generation. ![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-12_23-49-32.png) +## 📣 update -## 📣 更新 +[2025-03-13]⚒️: Release version v1.0.0. -[2025-03-13]⚒️: 发布版本 v1.0.0. +- All parameters are optional; you can generate random music without providing any parameters. -- 所有参数均是可选的, 不提供任何参数随机生成音乐. +## Installation -## 模型下载 +``` +cd ComfyUI/custom_nodes +git clone https://github.com/billwuhao/ComfyUI_DiffRhythm.git +cd ComfyUI_DiffRhythm +pip install -r requirements.txt -模型会自动下载到 `ComfyUI\models\TTS\DiffRhythm` 文件夹下. +# python_embeded +./python_embeded/python.exe -m pip install -r requirements.txt +``` -结构如下: +## Model Download + +Models will be automatically downloaded to the `ComfyUI\models\TTS\DiffRhythm` folder. + +The structure is as follows: ![](https://github.com/billwuhao/ComfyUI_DiffRhythm/blob/master/images/2025-03-13_00-08-51.png) -手动下载地址: +Manual Download Addresses: https://huggingface.co/ASLP-lab/DiffRhythm-base/blob/main/cfm_model.pt https://huggingface.co/ASLP-lab/DiffRhythm-vae/blob/main/vae_model.pt @@ -29,18 +40,24 @@ https://huggingface.co/OpenMuQ/MuQ-MuLan-large/tree/main https://huggingface.co/OpenMuQ/MuQ-large-msd-iter/tree/main https://huggingface.co/FacebookAI/xlm-roberta-base/tree/main -## 环境配置 +## Environment Configuration -Windows 系统做如下配置, 其他系统未测试. 应该支持 Linux, Mac. +- Configure the following on Windows systems: -下载安装最新版 [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases/tag/1.52.0) +Download and install the latest version of [espeak-ng](https://github.com/espeak-ng/espeak-ng/releases/tag/1.52.0) -添加环境变量 `PHONEMIZER_ESPEAK_LIBRARY` 到系统中, 值是你安装的 espeak-ng 软件中 `libespeak-ng.dll` 文件的路径, 例如: `C:\Program Files\eSpeak NG\libespeak-ng.dll`. +Add the environment variable `PHONEMIZER_ESPEAK_LIBRARY` to your system. The value should be the path to the `libespeak-ng.dll` file in your espeak-ng installation, for example: `C:\Program Files\eSpeak NG\libespeak-ng.dll`. -享受音乐吧🎶 +- On Linux systems, you need to install the `espeak-ng` package. Execute the following command to install: -## 鸣谢 +`apt-get -qq -y install espeak-ng > /dev/null 2>&1` + +It should support Mac, but has not been tested. + +Enjoy the music! 🎶 + +## Acknowledgements [DiffRhythm](https://github.com/ASLP-lab/DiffRhythm) -感谢 DiffRhythm 团队的卓越的工作, 目前最强开源 音乐/歌曲 生成模型👍. \ No newline at end of file +Thanks to the DiffRhythm team for their excellent work. Currently the strongest open-source music/song generation model 👍. diff --git a/__init__.py b/__init__.py index f8548d8..4488452 100644 --- a/__init__.py +++ b/__init__.py @@ -1,8 +1,3 @@ -import sys -import os +from .DiffRhythmNode import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS -current_dir = os.path.dirname(os.path.abspath(__file__)) -sys.path.insert(0, current_dir) - -from DiffRhythmNode import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS __all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] diff --git a/pyproject.toml b/pyproject.toml index f7f81e8..70d8f98 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "diffrhythm_mw" description = "Blazingly Fast and Embarrassingly Simple End-to-End Full-Length Song Generation. A node for ComfyUI." -version = "1.0.1" +version = "1.0.2" license = {file = "LICENSE"} dependencies = ["# accelerate==1.4.0", "# torchdiffeq==0.2.5", "# torchaudio==2.6.0", "# transformers==4.49.0", "# librosa==0.10.2.post1", "# pyarrow==19.0.1", "# pandas==2.2.3", "# bitsandbytes", "# jieba==0.42.1", "# cn2an==0.5.23", "# pypinyin==0.53.0", "# onnxruntime", "LangSegment", "x-transformers", "pylance", "ema-pytorch", "prefigure", "muq", "mutagen", "pyopenjtalk", "pykakasi", "Unidecode", "phonemizer"] diff --git a/requirements.txt b/requirements.txt index 3bbb0a9..e5ce39a 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,15 +1,13 @@ -# accelerate==1.4.0 -# torchdiffeq==0.2.5 -# torchaudio==2.6.0 -# transformers==4.49.0 -# librosa==0.10.2.post1 -# pyarrow==19.0.1 -# pandas==2.2.3 -# bitsandbytes -# jieba==0.42.1 -# cn2an==0.5.23 -# pypinyin==0.53.0 -# onnxruntime +accelerate +torchdiffeq +librosa>=0.10.2.post1 +pyarrow +pandas +bitsandbytes +jieba +cn2an +pypinyin +onnxruntime LangSegment x-transformers pylance