From e484835e95801b3ea3199ffd59655e4985dccd26 Mon Sep 17 00:00:00 2001
From: AIFSH <1509359472@qq.com>
Date: Thu, 18 Jul 2024 21:45:20 +0800
Subject: [PATCH] add speed control
---
README.md | 11 +
nodes.py | 116 +++++++--
requirements.txt | 2 +-
...ne_workflow.json => 3sclone_workflow.json} | 135 +++++-----
workflows/base_workflow.json | 91 ++++---
...gual_workflow.json => cross_workflow.json} | 204 +++++++--------
...3s_workflow.json => 3sclone_workflow.json} | 242 +++++++++---------
..._workflow.json => base_dubb_workflow.json} | 162 ++++++------
workflows/instruct_workflow.json | 221 ++++++++--------
9 files changed, 629 insertions(+), 555 deletions(-)
rename workflows/{3s_clone_workflow.json => 3sclone_workflow.json} (89%)
rename workflows/{cross_lingual_workflow.json => cross_workflow.json} (89%)
rename workflows/dubbing/{3s_workflow.json => 3sclone_workflow.json} (92%)
rename workflows/dubbing/{base_workflow.json => base_dubb_workflow.json} (93%)
diff --git a/README.md b/README.md
index 03c587d..405cdc5 100644
--- a/README.md
+++ b/README.md
@@ -1,6 +1,17 @@
# CosyVoice-ComfyUI
a comfyui custom node for [CosyVoice](https://github.com/FunAudioLLM/CosyVoice),you can find workflow in [workflows](./workflows/)
+## new Feature
+suport `srt` file to single voice or mutiple voice clone
+
+input
+- [tts_srt](./workflows/dubbing/zh_test.srt)
+- [prompt_wav](./workflows/dubbing/test.mp3)
+- [prompt_srt](./workflows/dubbing/en_test.srt)(optional)
+
+output
+
+
## Example
test on 2080ti 11GB torch==2.3.0+cu121 python 3.10.8
diff --git a/nodes.py b/nodes.py
index 0ddf0bb..9dc3d66 100644
--- a/nodes.py
+++ b/nodes.py
@@ -14,12 +14,9 @@ pretrained_models = os.path.join(now_dir,"pretrained_models")
from modelscope import snapshot_download
+import ffmpeg
import audiosegment
from srt import parse as SrtPare
-from audiosegment import AudioSegment
-import audiotsm
-from audiotsm.io.wav import WavReader, WavWriter
-
from cosyvoice.cli.cosyvoice import CosyVoice
sft_spk_list = ['中文女', '中文男', '日语男', '粤语女', '英文女', '英文男', '韩语女']
@@ -44,6 +41,32 @@ def postprocess(speech, top_db=60, hop_length=220, win_length=440):
speech = torch.concat([speech, torch.zeros(1, int(target_sr * 0.2))], dim=1)
return speech
+def speed_change(input_audio, speed, sr):
+ # 检查输入数据类型和声道数
+ if input_audio.dtype != np.int16:
+ raise ValueError("输入音频数据类型必须为 np.int16")
+
+
+ # 转换为字节流
+ raw_audio = input_audio.astype(np.int16).tobytes()
+
+ # 设置 ffmpeg 输入流
+ input_stream = ffmpeg.input('pipe:', format='s16le', acodec='pcm_s16le', ar=str(sr), ac=1)
+
+ # 变速处理
+ output_stream = input_stream.filter('atempo', speed)
+
+ # 输出流到管道
+ out, _ = (
+ output_stream.output('pipe:', format='s16le', acodec='pcm_s16le')
+ .run(input=raw_audio, capture_stdout=True, capture_stderr=True)
+ )
+
+ # 将管道输出解码为 NumPy 数组
+ processed_audio = np.frombuffer(out, np.int16)
+
+ return processed_audio
+
class TextNode:
@classmethod
def INPUT_TYPES(s):
@@ -66,6 +89,9 @@ class CosyVoiceNode:
return {
"required":{
"tts_text":("TEXT",),
+ "speed":("FLOAT",{
+ "default": 1.0
+ }),
"inference_mode":(inference_mode_list,{
"default": "预训练音色"
}),
@@ -91,7 +117,7 @@ class CosyVoiceNode:
CATEGORY = "AIFSH_CosyVoice"
- def generate(self,tts_text,inference_mode,sft_dropdown,seed,
+ def generate(self,tts_text,speed,inference_mode,sft_dropdown,seed,
prompt_text=None,prompt_wav=None,instruct_text=None):
if inference_mode == '自然语言控制':
model_dir = os.path.join(pretrained_models,"CosyVoice-300M-Instruct")
@@ -140,7 +166,11 @@ class CosyVoiceNode:
set_all_random_seed(seed)
print(self.model_dir)
output = self.cosyvoice.inference_instruct(tts_text, sft_dropdown, instruct_text)
- audio = {"waveform": [output['tts_speech']],"sample_rate":target_sr}
+
+ output_numpy = output['tts_speech'].squeeze(0).numpy() * 32768
+ output_numpy = output_numpy.astype(np.int16)
+ output_numpy = speed_change(output_numpy,speed,target_sr)
+ audio = {"waveform": [torch.Tensor(output_numpy/32768).unsqueeze(0)],"sample_rate":target_sr}
return (audio,)
class CosyVoiceDubbingNode:
@@ -154,6 +184,9 @@ class CosyVoiceDubbingNode:
"tts_srt":("SRT",),
"prompt_wav": ("AUDIO",),
"language":(["<|zh|>","<|en|>","<|jp|>","<|yue|>","<|ko|>"],),
+ "if_single":("BOOLEAN",{
+ "default": True
+ }),
"seed":("INT",{
"default": 42
})
@@ -171,7 +204,7 @@ class CosyVoiceDubbingNode:
CATEGORY = "AIFSH_CosyVoice"
- def generate(self,tts_srt,prompt_wav,language,seed,prompt_srt=None):
+ def generate(self,tts_srt,prompt_wav,language,if_single,seed,prompt_srt=None):
model_dir = os.path.join(pretrained_models,"CosyVoice-300M")
snapshot_download(model_id="iic/CosyVoice-300M",local_dir=model_dir)
set_all_random_seed(seed)
@@ -195,6 +228,7 @@ class CosyVoiceDubbingNode:
speech_numpy = speech.squeeze(0).numpy() * 32768
speech_numpy = speech_numpy.astype(np.int16)
audio_seg = audiosegment.from_numpy_array(speech_numpy,prompt_sr)
+ assert audio_seg.duration_seconds > 3, "prompt wav should be > 3s"
# audio_seg.export(os.path.join(output_dir,"test.mp3"),format="mp3")
new_audio_seg = audiosegment.silent(0,target_sr)
for i,text_sub in enumerate(text_subtitles):
@@ -202,16 +236,53 @@ class CosyVoiceDubbingNode:
end_time = text_sub.end.total_seconds() * 1000
if i == 0:
new_audio_seg += audio_seg[:start_time]
-
- curr_tts_text = language + text_sub.content
+
+ if if_single:
+ curr_tts_text = language + text_sub.content
+ else:
+ curr_tts_text = language + text_sub.content[1:]
+ speaker_id = text_sub.content[0]
prompt_wav_seg = audio_seg[start_time:end_time]
- # prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav")
+ if prompt_srt:
+ prompt_text_list = [prompt_subtitles[i].content]
+ while prompt_wav_seg.duration_seconds < 30:
+ for j in range(i+1,len(text_subtitles)):
+ j_start = text_subtitles[j].start.total_seconds() * 1000
+ j_end = text_subtitles[j].end.total_seconds() * 1000
+ if if_single:
+ prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
+ if prompt_srt:
+ prompt_text_list.append(prompt_subtitles[j].content)
+ else:
+ if text_subtitles[j].content[0] == speaker_id:
+ prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
+ if prompt_srt:
+ prompt_text_list.append(prompt_subtitles[j].content)
+ for j in range(0,i):
+ j_start = text_subtitles[j].start.total_seconds() * 1000
+ j_end = text_subtitles[j].end.total_seconds() * 1000
+ if if_single:
+ prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
+ if prompt_srt:
+ prompt_text_list.append(prompt_subtitles[j].content)
+ else:
+ if text_subtitles[j].content[0] == speaker_id:
+ prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
+ if prompt_srt:
+ prompt_text_list.append(prompt_subtitles[j].content)
+
+ if prompt_wav_seg.duration_seconds > 3:
+ break
+ print(f"prompt_wav {prompt_wav_seg.duration_seconds}s")
+ prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav")
prompt_wav_seg_numpy = prompt_wav_seg.to_numpy_array() / 32768
# print(prompt_wav_seg_numpy.shape)
prompt_speech_16k = postprocess(torch.Tensor(prompt_wav_seg_numpy).unsqueeze(0))
if prompt_srt:
- prompt_text = prompt_subtitles[i].content
+ # prompt_text = prompt_subtitles[i].content
+ prompt_text = ','.join(prompt_text_list)
+ print(f"prompt_text:{prompt_text}")
curr_output = self.cosyvoice.inference_zero_shot(curr_tts_text,prompt_text,prompt_speech_16k)
else:
curr_output = self.cosyvoice.inference_cross_lingual(curr_tts_text, prompt_speech_16k)
@@ -233,31 +304,24 @@ class CosyVoiceDubbingNode:
ratio = text_audio_dur_time / dur_time
if text_audio_dur_time > dur_time:
- tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav")
+ tmp_numpy = speed_change(curr_output_numpy,ratio,target_sr)
+ tmp_audio = audiosegment.from_numpy_array(tmp_numpy,target_sr)
+ # tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav")
tmp_audio += audiosegment.silent(dur_time - tmp_audio.duration_seconds*1000,target_sr)
else:
tmp_audio = text_audio + audiosegment.silent(dur_time - text_audio_dur_time,target_sr)
new_audio_seg += tmp_audio
+
+ if i == len(text_subtitles) - 1:
+ new_audio_seg += audio_seg[end_time:]
+
output_numpy = new_audio_seg.to_numpy_array() / 32768
# print(output_numpy.shape)
audio = {"waveform": [torch.Tensor(output_numpy).unsqueeze(0)],"sample_rate":target_sr}
return (audio,)
-
- def map_vocal(self,audio,ratio,dur_time,wav_name):
- os.makedirs(output_dir, exist_ok=True)
- tmp_path = f"{output_dir}/{wav_name}"
- audio.export(tmp_path, format="wav")
-
- clone_path = f"{output_dir}/speed_{wav_name}"
-
- with WavReader(tmp_path) as reader:
- with WavWriter(clone_path, reader.channels, reader.samplerate) as writer:
- tsm = audiotsm.phasevocoder(reader.channels, speed=ratio)
- tsm.run(reader, writer)
- audio_extended = audiosegment.from_file(clone_path)
- return audio_extended[:dur_time]
+
class LoadSRT:
@classmethod
def INPUT_TYPES(s):
diff --git a/requirements.txt b/requirements.txt
index 016638c..01a5b20 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -28,4 +28,4 @@ pypinyin
pydub
audiosegment
srt
-audiotsm
\ No newline at end of file
+ffmpeg-python
\ No newline at end of file
diff --git a/workflows/3s_clone_workflow.json b/workflows/3sclone_workflow.json
similarity index 89%
rename from workflows/3s_clone_workflow.json
rename to workflows/3sclone_workflow.json
index ec31346..93ae456 100644
--- a/workflows/3s_clone_workflow.json
+++ b/workflows/3sclone_workflow.json
@@ -1,36 +1,38 @@
{
- "last_node_id": 14,
- "last_link_id": 11,
+ "last_node_id": 15,
+ "last_link_id": 15,
"nodes": [
{
- "id": 12,
- "type": "TextNode",
+ "id": 13,
+ "type": "LoadAudio",
"pos": [
- 47,
- 398
+ 502,
+ 432
],
"size": {
- "0": 400,
- "1": 200
+ "0": 315,
+ "1": 124
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
- "name": "TEXT",
- "type": "TEXT",
+ "name": "AUDIO",
+ "type": "AUDIO",
"links": [
- 9
+ 14
],
"shape": 3
}
],
"properties": {
- "Node name for S&R": "TextNode"
+ "Node name for S&R": "LoadAudio"
},
"widgets_values": [
- "希望你以后能够做的比我还好呦。"
+ "zero_shot_prompt.wav",
+ null,
+ ""
]
},
{
@@ -52,7 +54,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
- 5
+ 12
],
"shape": 3,
"slot_index": 0
@@ -66,15 +68,47 @@
]
},
{
- "id": 9,
+ "id": 12,
+ "type": "TextNode",
+ "pos": [
+ 47,
+ 398
+ ],
+ "size": {
+ "0": 400,
+ "1": 200
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "TEXT",
+ "type": "TEXT",
+ "links": [
+ 13
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "TextNode"
+ },
+ "widgets_values": [
+ "希望你以后能够做的比我还好呦。"
+ ]
+ },
+ {
+ "id": 15,
"type": "CosyVoiceNode",
"pos": [
- 553,
- 127
+ 559,
+ 115
],
"size": {
"0": 315,
- "1": 190
+ "1": 214
},
"flags": {},
"order": 3,
@@ -83,18 +117,17 @@
{
"name": "tts_text",
"type": "TEXT",
- "link": 5
+ "link": 12
},
{
"name": "prompt_text",
"type": "TEXT",
- "link": 9,
- "slot_index": 1
+ "link": 13
},
{
"name": "prompt_wav",
"type": "AUDIO",
- "link": 10,
+ "link": 14,
"slot_index": 2
},
{
@@ -108,7 +141,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
- 11
+ 15
],
"shape": 3,
"slot_index": 0
@@ -118,45 +151,13 @@
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
- "3s极速复刻",
+ 1.5,
+ "预训练音色",
"中文女",
- 1873,
+ 2002,
"randomize"
]
},
- {
- "id": 13,
- "type": "LoadAudio",
- "pos": [
- 502,
- 432
- ],
- "size": {
- "0": 315,
- "1": 124
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "outputs": [
- {
- "name": "AUDIO",
- "type": "AUDIO",
- "links": [
- 10
- ],
- "shape": 3
- }
- ],
- "properties": {
- "Node name for S&R": "LoadAudio"
- },
- "widgets_values": [
- "zero_shot_prompt.wav",
- null,
- ""
- ]
- },
{
"id": 14,
"type": "SaveAudio",
@@ -175,7 +176,7 @@
{
"name": "audio",
"type": "AUDIO",
- "link": 11
+ "link": 15
}
],
"properties": {
@@ -189,32 +190,32 @@
],
"links": [
[
- 5,
+ 12,
2,
0,
- 9,
+ 15,
0,
"TEXT"
],
[
- 9,
+ 13,
12,
0,
- 9,
+ 15,
1,
"TEXT"
],
[
- 10,
+ 14,
13,
0,
- 9,
+ 15,
2,
"AUDIO"
],
[
- 11,
- 9,
+ 15,
+ 15,
0,
14,
0,
diff --git a/workflows/base_workflow.json b/workflows/base_workflow.json
index c758af9..f91e968 100644
--- a/workflows/base_workflow.json
+++ b/workflows/base_workflow.json
@@ -1,35 +1,7 @@
{
- "last_node_id": 11,
- "last_link_id": 8,
+ "last_node_id": 12,
+ "last_link_id": 10,
"nodes": [
- {
- "id": 3,
- "type": "PreviewAudio",
- "pos": [
- 970,
- 118
- ],
- "size": {
- "0": 315,
- "1": 76
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [
- {
- "name": "audio",
- "type": "AUDIO",
- "link": 6
- }
- ],
- "properties": {
- "Node name for S&R": "PreviewAudio"
- },
- "widgets_values": [
- null
- ]
- },
{
"id": 2,
"type": "TextNode",
@@ -49,7 +21,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
- 5
+ 9
],
"shape": 3,
"slot_index": 0
@@ -63,15 +35,15 @@
]
},
{
- "id": 9,
+ "id": 12,
"type": "CosyVoiceNode",
"pos": [
- 553,
- 127
+ 578,
+ 128
],
"size": {
"0": 315,
- "1": 190
+ "1": 214
},
"flags": {},
"order": 1,
@@ -80,19 +52,17 @@
{
"name": "tts_text",
"type": "TEXT",
- "link": 5
+ "link": 9
},
{
"name": "prompt_text",
"type": "TEXT",
- "link": null,
- "slot_index": 1
+ "link": null
},
{
"name": "prompt_wav",
"type": "AUDIO",
- "link": null,
- "slot_index": 2
+ "link": null
},
{
"name": "instruct_text",
@@ -105,7 +75,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
- 6
+ 10
],
"shape": 3,
"slot_index": 0
@@ -115,25 +85,54 @@
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
+ 1.2000000000000002,
"预训练音色",
"中文女",
- 1712,
+ 1365,
"randomize"
]
+ },
+ {
+ "id": 3,
+ "type": "PreviewAudio",
+ "pos": [
+ 970,
+ 118
+ ],
+ "size": {
+ "0": 315,
+ "1": 76
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "audio",
+ "type": "AUDIO",
+ "link": 10
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "PreviewAudio"
+ },
+ "widgets_values": [
+ null
+ ]
}
],
"links": [
[
- 5,
+ 9,
2,
0,
- 9,
+ 12,
0,
"TEXT"
],
[
- 6,
- 9,
+ 10,
+ 12,
0,
3,
0,
diff --git a/workflows/cross_lingual_workflow.json b/workflows/cross_workflow.json
similarity index 89%
rename from workflows/cross_lingual_workflow.json
rename to workflows/cross_workflow.json
index 82e860d..4ced739 100644
--- a/workflows/cross_lingual_workflow.json
+++ b/workflows/cross_workflow.json
@@ -1,98 +1,7 @@
{
- "last_node_id": 14,
- "last_link_id": 11,
+ "last_node_id": 15,
+ "last_link_id": 14,
"nodes": [
- {
- "id": 9,
- "type": "CosyVoiceNode",
- "pos": [
- 553,
- 127
- ],
- "size": {
- "0": 315,
- "1": 190
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [
- {
- "name": "tts_text",
- "type": "TEXT",
- "link": 5
- },
- {
- "name": "prompt_text",
- "type": "TEXT",
- "link": null,
- "slot_index": 1
- },
- {
- "name": "prompt_wav",
- "type": "AUDIO",
- "link": 10,
- "slot_index": 2
- },
- {
- "name": "instruct_text",
- "type": "TEXT",
- "link": null
- }
- ],
- "outputs": [
- {
- "name": "AUDIO",
- "type": "AUDIO",
- "links": [
- 11
- ],
- "shape": 3,
- "slot_index": 0
- }
- ],
- "properties": {
- "Node name for S&R": "CosyVoiceNode"
- },
- "widgets_values": [
- "跨语种复刻",
- "中文女",
- 1883,
- "randomize"
- ]
- },
- {
- "id": 2,
- "type": "TextNode",
- "pos": [
- 77,
- 137
- ],
- "size": {
- "0": 400,
- "1": 200
- },
- "flags": {},
- "order": 0,
- "mode": 0,
- "outputs": [
- {
- "name": "TEXT",
- "type": "TEXT",
- "links": [
- 5
- ],
- "shape": 3,
- "slot_index": 0
- }
- ],
- "properties": {
- "Node name for S&R": "TextNode"
- },
- "widgets_values": [
- "<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing."
- ]
- },
{
"id": 13,
"type": "LoadAudio",
@@ -105,14 +14,14 @@
"1": 124
},
"flags": {},
- "order": 1,
+ "order": 0,
"mode": 0,
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
- 10
+ 13
],
"shape": 3
}
@@ -144,7 +53,7 @@
{
"name": "audio",
"type": "AUDIO",
- "link": 11
+ "link": 14
}
],
"properties": {
@@ -154,28 +63,119 @@
"audio/ComfyUI",
null
]
+ },
+ {
+ "id": 2,
+ "type": "TextNode",
+ "pos": [
+ 77,
+ 137
+ ],
+ "size": {
+ "0": 400,
+ "1": 200
+ },
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "TEXT",
+ "type": "TEXT",
+ "links": [
+ 12
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "TextNode"
+ },
+ "widgets_values": [
+ "<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing."
+ ]
+ },
+ {
+ "id": 15,
+ "type": "CosyVoiceNode",
+ "pos": [
+ 568,
+ 105
+ ],
+ "size": {
+ "0": 315,
+ "1": 214
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "tts_text",
+ "type": "TEXT",
+ "link": 12
+ },
+ {
+ "name": "prompt_text",
+ "type": "TEXT",
+ "link": null
+ },
+ {
+ "name": "prompt_wav",
+ "type": "AUDIO",
+ "link": 13,
+ "slot_index": 2
+ },
+ {
+ "name": "instruct_text",
+ "type": "TEXT",
+ "link": null
+ }
+ ],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 14
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "CosyVoiceNode"
+ },
+ "widgets_values": [
+ 1,
+ "预训练音色",
+ "中文女",
+ 42,
+ "randomize"
+ ]
}
],
"links": [
[
- 5,
+ 12,
2,
0,
- 9,
+ 15,
0,
"TEXT"
],
[
- 10,
+ 13,
13,
0,
- 9,
+ 15,
2,
"AUDIO"
],
[
- 11,
- 9,
+ 14,
+ 15,
0,
14,
0,
diff --git a/workflows/dubbing/3s_workflow.json b/workflows/dubbing/3sclone_workflow.json
similarity index 92%
rename from workflows/dubbing/3s_workflow.json
rename to workflows/dubbing/3sclone_workflow.json
index 345c3bf..691a377 100644
--- a/workflows/dubbing/3s_workflow.json
+++ b/workflows/dubbing/3sclone_workflow.json
@@ -1,6 +1,6 @@
{
- "last_node_id": 8,
- "last_link_id": 8,
+ "last_node_id": 9,
+ "last_link_id": 11,
"nodes": [
{
"id": 2,
@@ -36,89 +36,6 @@
""
]
},
- {
- "id": 3,
- "type": "VocalSeparationNode",
- "pos": [
- 527,
- 62
- ],
- "size": {
- "0": 315,
- "1": 126
- },
- "flags": {},
- "order": 3,
- "mode": 0,
- "inputs": [
- {
- "name": "music",
- "type": "AUDIO",
- "link": 2
- }
- ],
- "outputs": [
- {
- "name": "vocals_AUDIO",
- "type": "AUDIO",
- "links": [
- 3,
- 4
- ],
- "shape": 3,
- "slot_index": 0
- },
- {
- "name": "instrumental_AUDIO",
- "type": "AUDIO",
- "links": [
- 5
- ],
- "shape": 3,
- "slot_index": 1
- }
- ],
- "properties": {
- "Node name for S&R": "VocalSeparationNode"
- },
- "widgets_values": [
- "bs_roformer",
- 4,
- true
- ]
- },
- {
- "id": 7,
- "type": "LoadSRT",
- "pos": [
- 70,
- 133
- ],
- "size": {
- "0": 315,
- "1": 82
- },
- "flags": {},
- "order": 1,
- "mode": 0,
- "outputs": [
- {
- "name": "SRT",
- "type": "SRT",
- "links": [
- 7
- ],
- "shape": 3
- }
- ],
- "properties": {
- "Node name for S&R": "LoadSRT"
- },
- "widgets_values": [
- "zh_test.srt",
- "Audio"
- ]
- },
{
"id": 4,
"type": "PreviewAudio",
@@ -131,7 +48,7 @@
"1": 76
},
"flags": {},
- "order": 5,
+ "order": 4,
"mode": 0,
"inputs": [
{
@@ -193,7 +110,7 @@
{
"name": "audio",
"type": "AUDIO",
- "link": 6
+ "link": 10
}
],
"properties": {
@@ -204,36 +121,118 @@
]
},
{
- "id": 1,
- "type": "CosyVoiceDubbingNode",
+ "id": 3,
+ "type": "VocalSeparationNode",
"pos": [
- 550,
- 374
+ 527,
+ 62
],
"size": {
"0": 315,
- "1": 146
+ "1": 126
},
"flags": {},
- "order": 4,
+ "order": 3,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "music",
+ "type": "AUDIO",
+ "link": 2
+ }
+ ],
+ "outputs": [
+ {
+ "name": "vocals_AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 4,
+ 8
+ ],
+ "shape": 3,
+ "slot_index": 0
+ },
+ {
+ "name": "instrumental_AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 5
+ ],
+ "shape": 3,
+ "slot_index": 1
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "VocalSeparationNode"
+ },
+ "widgets_values": [
+ "bs_roformer",
+ 4,
+ true
+ ]
+ },
+ {
+ "id": 7,
+ "type": "LoadSRT",
+ "pos": [
+ 70,
+ 133
+ ],
+ "size": {
+ "0": 315,
+ "1": 82
+ },
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "SRT",
+ "type": "SRT",
+ "links": [
+ 9
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "LoadSRT"
+ },
+ "widgets_values": [
+ "zh_test.srt",
+ "Audio"
+ ]
+ },
+ {
+ "id": 8,
+ "type": "CosyVoiceDubbingNode",
+ "pos": [
+ 532,
+ 365
+ ],
+ "size": {
+ "0": 315,
+ "1": 170
+ },
+ "flags": {},
+ "order": 5,
"mode": 0,
"inputs": [
{
"name": "tts_srt",
"type": "SRT",
- "link": 7,
- "slot_index": 0
+ "link": 9
},
{
"name": "prompt_wav",
"type": "AUDIO",
- "link": 3,
- "slot_index": 1
+ "link": 8
},
{
"name": "prompt_srt",
"type": "SRT",
- "link": 8,
+ "link": 11,
"slot_index": 2
}
],
@@ -242,7 +241,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
- 6
+ 10
],
"shape": 3,
"slot_index": 0
@@ -253,16 +252,17 @@
},
"widgets_values": [
"<|zh|>",
- 1085,
+ true,
+ 42,
"randomize"
]
},
{
- "id": 8,
+ "id": 9,
"type": "LoadSRT",
"pos": [
- 118,
- 535
+ 115,
+ 549
],
"size": {
"0": 315,
@@ -276,7 +276,7 @@
"name": "SRT",
"type": "SRT",
"links": [
- 8
+ 11
],
"shape": 3
}
@@ -285,7 +285,7 @@
"Node name for S&R": "LoadSRT"
},
"widgets_values": [
- "zh_test.srt",
+ "en_test.srt",
"Audio"
]
}
@@ -299,14 +299,6 @@
0,
"AUDIO"
],
- [
- 3,
- 3,
- 0,
- 1,
- 1,
- "AUDIO"
- ],
[
4,
3,
@@ -324,26 +316,34 @@
"AUDIO"
],
[
- 6,
+ 8,
+ 3,
+ 0,
+ 8,
1,
- 0,
- 6,
- 0,
"AUDIO"
],
[
- 7,
+ 9,
7,
0,
- 1,
+ 8,
0,
"SRT"
],
[
- 8,
+ 10,
8,
0,
- 1,
+ 6,
+ 0,
+ "AUDIO"
+ ],
+ [
+ 11,
+ 9,
+ 0,
+ 8,
2,
"SRT"
]
diff --git a/workflows/dubbing/base_workflow.json b/workflows/dubbing/base_dubb_workflow.json
similarity index 93%
rename from workflows/dubbing/base_workflow.json
rename to workflows/dubbing/base_dubb_workflow.json
index 51b2e13..7bd6745 100644
--- a/workflows/dubbing/base_workflow.json
+++ b/workflows/dubbing/base_dubb_workflow.json
@@ -1,6 +1,6 @@
{
- "last_node_id": 7,
- "last_link_id": 7,
+ "last_node_id": 8,
+ "last_link_id": 10,
"nodes": [
{
"id": 2,
@@ -36,57 +36,6 @@
""
]
},
- {
- "id": 3,
- "type": "VocalSeparationNode",
- "pos": [
- 527,
- 62
- ],
- "size": {
- "0": 315,
- "1": 126
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [
- {
- "name": "music",
- "type": "AUDIO",
- "link": 2
- }
- ],
- "outputs": [
- {
- "name": "vocals_AUDIO",
- "type": "AUDIO",
- "links": [
- 3,
- 4
- ],
- "shape": 3,
- "slot_index": 0
- },
- {
- "name": "instrumental_AUDIO",
- "type": "AUDIO",
- "links": [
- 5
- ],
- "shape": 3,
- "slot_index": 1
- }
- ],
- "properties": {
- "Node name for S&R": "VocalSeparationNode"
- },
- "widgets_values": [
- "bs_roformer",
- 4,
- true
- ]
- },
{
"id": 4,
"type": "PreviewAudio",
@@ -99,7 +48,7 @@
"1": 76
},
"flags": {},
- "order": 4,
+ "order": 3,
"mode": 0,
"inputs": [
{
@@ -161,7 +110,7 @@
{
"name": "audio",
"type": "AUDIO",
- "link": 6
+ "link": 10
}
],
"properties": {
@@ -171,6 +120,57 @@
null
]
},
+ {
+ "id": 3,
+ "type": "VocalSeparationNode",
+ "pos": [
+ 527,
+ 62
+ ],
+ "size": {
+ "0": 315,
+ "1": 126
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "music",
+ "type": "AUDIO",
+ "link": 2
+ }
+ ],
+ "outputs": [
+ {
+ "name": "vocals_AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 4,
+ 8
+ ],
+ "shape": 3,
+ "slot_index": 0
+ },
+ {
+ "name": "instrumental_AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 5
+ ],
+ "shape": 3,
+ "slot_index": 1
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "VocalSeparationNode"
+ },
+ "widgets_values": [
+ "bs_roformer",
+ 4,
+ true
+ ]
+ },
{
"id": 7,
"type": "LoadSRT",
@@ -190,9 +190,10 @@
"name": "SRT",
"type": "SRT",
"links": [
- 7
+ 9
],
- "shape": 3
+ "shape": 3,
+ "slot_index": 0
}
],
"properties": {
@@ -204,31 +205,29 @@
]
},
{
- "id": 1,
+ "id": 8,
"type": "CosyVoiceDubbingNode",
"pos": [
- 550,
- 374
+ 532,
+ 365
],
"size": {
"0": 315,
- "1": 146
+ "1": 170
},
"flags": {},
- "order": 3,
+ "order": 4,
"mode": 0,
"inputs": [
{
"name": "tts_srt",
"type": "SRT",
- "link": 7,
- "slot_index": 0
+ "link": 9
},
{
"name": "prompt_wav",
"type": "AUDIO",
- "link": 3,
- "slot_index": 1
+ "link": 8
},
{
"name": "prompt_srt",
@@ -241,7 +240,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
- 6
+ 10
],
"shape": 3,
"slot_index": 0
@@ -252,6 +251,7 @@
},
"widgets_values": [
"<|zh|>",
+ true,
42,
"randomize"
]
@@ -266,14 +266,6 @@
0,
"AUDIO"
],
- [
- 3,
- 3,
- 0,
- 1,
- 1,
- "AUDIO"
- ],
[
4,
3,
@@ -291,20 +283,28 @@
"AUDIO"
],
[
- 6,
+ 8,
+ 3,
+ 0,
+ 8,
1,
- 0,
- 6,
- 0,
"AUDIO"
],
[
- 7,
+ 9,
7,
0,
- 1,
+ 8,
0,
"SRT"
+ ],
+ [
+ 10,
+ 8,
+ 0,
+ 6,
+ 0,
+ "AUDIO"
]
],
"groups": [],
diff --git a/workflows/instruct_workflow.json b/workflows/instruct_workflow.json
index d1ad21f..070936a 100644
--- a/workflows/instruct_workflow.json
+++ b/workflows/instruct_workflow.json
@@ -1,13 +1,13 @@
{
- "last_node_id": 15,
- "last_link_id": 12,
+ "last_node_id": 16,
+ "last_link_id": 15,
"nodes": [
{
- "id": 2,
+ "id": 15,
"type": "TextNode",
"pos": [
- 77,
- 137
+ 126,
+ 426
],
"size": {
"0": 400,
@@ -21,99 +21,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
- 5
- ],
- "shape": 3,
- "slot_index": 0
- }
- ],
- "properties": {
- "Node name for S&R": "TextNode"
- },
- "widgets_values": [
- "在面对挑战时,他展现了非凡的勇气与智慧。'"
- ]
- },
- {
- "id": 9,
- "type": "CosyVoiceNode",
- "pos": [
- 553,
- 127
- ],
- "size": {
- "0": 315,
- "1": 190
- },
- "flags": {},
- "order": 2,
- "mode": 0,
- "inputs": [
- {
- "name": "tts_text",
- "type": "TEXT",
- "link": 5
- },
- {
- "name": "prompt_text",
- "type": "TEXT",
- "link": null,
- "slot_index": 1
- },
- {
- "name": "prompt_wav",
- "type": "AUDIO",
- "link": null,
- "slot_index": 2
- },
- {
- "name": "instruct_text",
- "type": "TEXT",
- "link": 12,
- "slot_index": 3
- }
- ],
- "outputs": [
- {
- "name": "AUDIO",
- "type": "AUDIO",
- "links": [
- 11
- ],
- "shape": 3,
- "slot_index": 0
- }
- ],
- "properties": {
- "Node name for S&R": "CosyVoiceNode"
- },
- "widgets_values": [
- "自然语言控制",
- "中文女",
- 208,
- "randomize"
- ]
- },
- {
- "id": 15,
- "type": "TextNode",
- "pos": [
- 126,
- 426
- ],
- "size": {
- "0": 400,
- "1": 200
- },
- "flags": {},
- "order": 1,
- "mode": 0,
- "outputs": [
- {
- "name": "TEXT",
- "type": "TEXT",
- "links": [
- 12
+ 14
],
"shape": 3
}
@@ -143,7 +51,7 @@
{
"name": "audio",
"type": "AUDIO",
- "link": 11
+ "link": 15
}
],
"properties": {
@@ -153,32 +61,123 @@
"audio/ComfyUI",
null
]
+ },
+ {
+ "id": 2,
+ "type": "TextNode",
+ "pos": [
+ 77,
+ 137
+ ],
+ "size": {
+ "0": 400,
+ "1": 200
+ },
+ "flags": {},
+ "order": 1,
+ "mode": 0,
+ "outputs": [
+ {
+ "name": "TEXT",
+ "type": "TEXT",
+ "links": [
+ 13
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "TextNode"
+ },
+ "widgets_values": [
+ "在面对挑战时,他展现了非凡的勇气与智慧。'"
+ ]
+ },
+ {
+ "id": 16,
+ "type": "CosyVoiceNode",
+ "pos": [
+ 569,
+ 117
+ ],
+ "size": {
+ "0": 315,
+ "1": 214
+ },
+ "flags": {},
+ "order": 2,
+ "mode": 0,
+ "inputs": [
+ {
+ "name": "tts_text",
+ "type": "TEXT",
+ "link": 13
+ },
+ {
+ "name": "prompt_text",
+ "type": "TEXT",
+ "link": null
+ },
+ {
+ "name": "prompt_wav",
+ "type": "AUDIO",
+ "link": null
+ },
+ {
+ "name": "instruct_text",
+ "type": "TEXT",
+ "link": 14,
+ "slot_index": 3
+ }
+ ],
+ "outputs": [
+ {
+ "name": "AUDIO",
+ "type": "AUDIO",
+ "links": [
+ 15
+ ],
+ "shape": 3,
+ "slot_index": 0
+ }
+ ],
+ "properties": {
+ "Node name for S&R": "CosyVoiceNode"
+ },
+ "widgets_values": [
+ 1,
+ "预训练音色",
+ "中文女",
+ 368,
+ "randomize"
+ ]
}
],
"links": [
[
- 5,
+ 13,
2,
0,
- 9,
+ 16,
0,
"TEXT"
],
[
- 11,
- 9,
+ 14,
+ 15,
+ 0,
+ 16,
+ 3,
+ "TEXT"
+ ],
+ [
+ 15,
+ 16,
0,
14,
0,
"AUDIO"
- ],
- [
- 12,
- 15,
- 0,
- 9,
- 3,
- "TEXT"
]
],
"groups": [],