From 792fd8f4f6594404e73d0d3258c8be6d58baa78c Mon Sep 17 00:00:00 2001 From: BobRandomNumber Date: Tue, 8 Jul 2025 22:27:00 -0400 Subject: [PATCH] Update * Update CITATION.cff * Update README.md * Delete CITATION.cff --- CITATION.cff | 9 --------- README.md | 15 +++++++++++++-- 2 files changed, 13 insertions(+), 11 deletions(-) delete mode 100644 CITATION.cff diff --git a/CITATION.cff b/CITATION.cff deleted file mode 100644 index fedba81..0000000 --- a/CITATION.cff +++ /dev/null @@ -1,9 +0,0 @@ -@techreport{kyutai2024moshi, - author = {Alexandre D\'efossez and Laurent Mazar\'e and Manu Orsini and Am\'elie Royer and - Patrick P\'erez and Herv\'e J\'egou and Edouard Grave and Neil Zeghidour}, - title = {Moshi: a speech-text foundation model for real-time dialogue}, - institution = {Kyutai}, - year={2024}, - month={September}, - url={http://kyutai.org/Moshi.pdf}, -} \ No newline at end of file diff --git a/README.md b/README.md index 708f4a3..e995815 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ A custom node for ComfyUI that allows TTS generation with the [Kyutai TTS 1.6b en_fr model](https://huggingface.co/kyutai/tts-1.6b-en_fr) using [Kyutai offered voice models.](https://huggingface.co/kyutai/tts-voices) -The model's intended use is [https://github.com/kyutai-labs/delayed-streams-modeling](https://github.com/kyutai-labs/delayed-streams-modeling)which is not implemented here. +The model's intended use is [https://github.com/kyutai-labs/delayed-streams-modeling](https://github.com/kyutai-labs/delayed-streams-modeling) which is not implemented here. I made this version as it can generate large ammounts quickly and at acceptable quality for my use cases. The model outputs at 24000Hz, some post processing can improve it if needed. @@ -78,4 +78,15 @@ This custom node utilizes the Kyutai TTS model. * **Original Moshi Source:** [https://github.com/kyutai-labs/moshi/tree/main/moshi](https://github.com/kyutai-labs/moshi/tree/main/moshi) * **Kyutai TTS Model (1.6B en_fr):** [https://huggingface.co/kyutai/tts-1.6b-en_fr](https://huggingface.co/kyutai/tts-1.6b-en_fr) -* **Kyutai TTS Voice Models:** [https://huggingface.co/kyutai/tts-voices](https://huggingface.co/kyutai/tts-voices) \ No newline at end of file +* **Kyutai TTS Voice Models:** [https://huggingface.co/kyutai/tts-voices](https://huggingface.co/kyutai/tts-voices) + +## Citation + +@techreport{kyutai2024moshi, + author = {Alexandre D\'efossez and Laurent Mazar\'e and Manu Orsini and Am\'elie Royer and Patrick P\'erez and Herv\'e J\'egou and Edouard Grave and Neil Zeghidour}, + title = {Moshi: a speech-text foundation model for real-time dialogue}, + institution = {Kyutai}, + year={2024}, + month={September}, + url={http://kyutai.org/Moshi.pdf}, +}