add speed control

This commit is contained in:
AIFSH
2024-07-18 21:45:20 +08:00
parent 50289a9960
commit e484835e95
9 changed files with 629 additions and 555 deletions
+11
View File
@@ -1,6 +1,17 @@
# CosyVoice-ComfyUI
a comfyui custom node for [CosyVoice](https://github.com/FunAudioLLM/CosyVoice),you can find workflow in [workflows](./workflows/)
## new Feature
suport `srt` file to single voice or mutiple voice clone
input
- [tts_srt](./workflows/dubbing/zh_test.srt)
- [prompt_wav](./workflows/dubbing/test.mp3)
- [prompt_srt](./workflows/dubbing/en_test.srt)(optional)
output
## Example
test on 2080ti 11GB torch==2.3.0+cu121 python 3.10.8
+90 -26
View File
@@ -14,12 +14,9 @@ pretrained_models = os.path.join(now_dir,"pretrained_models")
from modelscope import snapshot_download
import ffmpeg
import audiosegment
from srt import parse as SrtPare
from audiosegment import AudioSegment
import audiotsm
from audiotsm.io.wav import WavReader, WavWriter
from cosyvoice.cli.cosyvoice import CosyVoice
sft_spk_list = ['中文女', '中文男', '日语男', '粤语女', '英文女', '英文男', '韩语女']
@@ -44,6 +41,32 @@ def postprocess(speech, top_db=60, hop_length=220, win_length=440):
speech = torch.concat([speech, torch.zeros(1, int(target_sr * 0.2))], dim=1)
return speech
def speed_change(input_audio, speed, sr):
# 检查输入数据类型和声道数
if input_audio.dtype != np.int16:
raise ValueError("输入音频数据类型必须为 np.int16")
# 转换为字节流
raw_audio = input_audio.astype(np.int16).tobytes()
# 设置 ffmpeg 输入流
input_stream = ffmpeg.input('pipe:', format='s16le', acodec='pcm_s16le', ar=str(sr), ac=1)
# 变速处理
output_stream = input_stream.filter('atempo', speed)
# 输出流到管道
out, _ = (
output_stream.output('pipe:', format='s16le', acodec='pcm_s16le')
.run(input=raw_audio, capture_stdout=True, capture_stderr=True)
)
# 将管道输出解码为 NumPy 数组
processed_audio = np.frombuffer(out, np.int16)
return processed_audio
class TextNode:
@classmethod
def INPUT_TYPES(s):
@@ -66,6 +89,9 @@ class CosyVoiceNode:
return {
"required":{
"tts_text":("TEXT",),
"speed":("FLOAT",{
"default": 1.0
}),
"inference_mode":(inference_mode_list,{
"default": "预训练音色"
}),
@@ -91,7 +117,7 @@ class CosyVoiceNode:
CATEGORY = "AIFSH_CosyVoice"
def generate(self,tts_text,inference_mode,sft_dropdown,seed,
def generate(self,tts_text,speed,inference_mode,sft_dropdown,seed,
prompt_text=None,prompt_wav=None,instruct_text=None):
if inference_mode == '自然语言控制':
model_dir = os.path.join(pretrained_models,"CosyVoice-300M-Instruct")
@@ -140,7 +166,11 @@ class CosyVoiceNode:
set_all_random_seed(seed)
print(self.model_dir)
output = self.cosyvoice.inference_instruct(tts_text, sft_dropdown, instruct_text)
audio = {"waveform": [output['tts_speech']],"sample_rate":target_sr}
output_numpy = output['tts_speech'].squeeze(0).numpy() * 32768
output_numpy = output_numpy.astype(np.int16)
output_numpy = speed_change(output_numpy,speed,target_sr)
audio = {"waveform": [torch.Tensor(output_numpy/32768).unsqueeze(0)],"sample_rate":target_sr}
return (audio,)
class CosyVoiceDubbingNode:
@@ -154,6 +184,9 @@ class CosyVoiceDubbingNode:
"tts_srt":("SRT",),
"prompt_wav": ("AUDIO",),
"language":(["<|zh|>","<|en|>","<|jp|>","<|yue|>","<|ko|>"],),
"if_single":("BOOLEAN",{
"default": True
}),
"seed":("INT",{
"default": 42
})
@@ -171,7 +204,7 @@ class CosyVoiceDubbingNode:
CATEGORY = "AIFSH_CosyVoice"
def generate(self,tts_srt,prompt_wav,language,seed,prompt_srt=None):
def generate(self,tts_srt,prompt_wav,language,if_single,seed,prompt_srt=None):
model_dir = os.path.join(pretrained_models,"CosyVoice-300M")
snapshot_download(model_id="iic/CosyVoice-300M",local_dir=model_dir)
set_all_random_seed(seed)
@@ -195,6 +228,7 @@ class CosyVoiceDubbingNode:
speech_numpy = speech.squeeze(0).numpy() * 32768
speech_numpy = speech_numpy.astype(np.int16)
audio_seg = audiosegment.from_numpy_array(speech_numpy,prompt_sr)
assert audio_seg.duration_seconds > 3, "prompt wav should be > 3s"
# audio_seg.export(os.path.join(output_dir,"test.mp3"),format="mp3")
new_audio_seg = audiosegment.silent(0,target_sr)
for i,text_sub in enumerate(text_subtitles):
@@ -202,16 +236,53 @@ class CosyVoiceDubbingNode:
end_time = text_sub.end.total_seconds() * 1000
if i == 0:
new_audio_seg += audio_seg[:start_time]
curr_tts_text = language + text_sub.content
if if_single:
curr_tts_text = language + text_sub.content
else:
curr_tts_text = language + text_sub.content[1:]
speaker_id = text_sub.content[0]
prompt_wav_seg = audio_seg[start_time:end_time]
# prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav")
if prompt_srt:
prompt_text_list = [prompt_subtitles[i].content]
while prompt_wav_seg.duration_seconds < 30:
for j in range(i+1,len(text_subtitles)):
j_start = text_subtitles[j].start.total_seconds() * 1000
j_end = text_subtitles[j].end.total_seconds() * 1000
if if_single:
prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
if prompt_srt:
prompt_text_list.append(prompt_subtitles[j].content)
else:
if text_subtitles[j].content[0] == speaker_id:
prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
if prompt_srt:
prompt_text_list.append(prompt_subtitles[j].content)
for j in range(0,i):
j_start = text_subtitles[j].start.total_seconds() * 1000
j_end = text_subtitles[j].end.total_seconds() * 1000
if if_single:
prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
if prompt_srt:
prompt_text_list.append(prompt_subtitles[j].content)
else:
if text_subtitles[j].content[0] == speaker_id:
prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end])
if prompt_srt:
prompt_text_list.append(prompt_subtitles[j].content)
if prompt_wav_seg.duration_seconds > 3:
break
print(f"prompt_wav {prompt_wav_seg.duration_seconds}s")
prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav")
prompt_wav_seg_numpy = prompt_wav_seg.to_numpy_array() / 32768
# print(prompt_wav_seg_numpy.shape)
prompt_speech_16k = postprocess(torch.Tensor(prompt_wav_seg_numpy).unsqueeze(0))
if prompt_srt:
prompt_text = prompt_subtitles[i].content
# prompt_text = prompt_subtitles[i].content
prompt_text = ','.join(prompt_text_list)
print(f"prompt_text:{prompt_text}")
curr_output = self.cosyvoice.inference_zero_shot(curr_tts_text,prompt_text,prompt_speech_16k)
else:
curr_output = self.cosyvoice.inference_cross_lingual(curr_tts_text, prompt_speech_16k)
@@ -233,31 +304,24 @@ class CosyVoiceDubbingNode:
ratio = text_audio_dur_time / dur_time
if text_audio_dur_time > dur_time:
tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav")
tmp_numpy = speed_change(curr_output_numpy,ratio,target_sr)
tmp_audio = audiosegment.from_numpy_array(tmp_numpy,target_sr)
# tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav")
tmp_audio += audiosegment.silent(dur_time - tmp_audio.duration_seconds*1000,target_sr)
else:
tmp_audio = text_audio + audiosegment.silent(dur_time - text_audio_dur_time,target_sr)
new_audio_seg += tmp_audio
if i == len(text_subtitles) - 1:
new_audio_seg += audio_seg[end_time:]
output_numpy = new_audio_seg.to_numpy_array() / 32768
# print(output_numpy.shape)
audio = {"waveform": [torch.Tensor(output_numpy).unsqueeze(0)],"sample_rate":target_sr}
return (audio,)
def map_vocal(self,audio,ratio,dur_time,wav_name):
os.makedirs(output_dir, exist_ok=True)
tmp_path = f"{output_dir}/{wav_name}"
audio.export(tmp_path, format="wav")
clone_path = f"{output_dir}/speed_{wav_name}"
with WavReader(tmp_path) as reader:
with WavWriter(clone_path, reader.channels, reader.samplerate) as writer:
tsm = audiotsm.phasevocoder(reader.channels, speed=ratio)
tsm.run(reader, writer)
audio_extended = audiosegment.from_file(clone_path)
return audio_extended[:dur_time]
class LoadSRT:
@classmethod
def INPUT_TYPES(s):
+1 -1
View File
@@ -28,4 +28,4 @@ pypinyin
pydub
audiosegment
srt
audiotsm
ffmpeg-python
@@ -1,36 +1,38 @@
{
"last_node_id": 14,
"last_link_id": 11,
"last_node_id": 15,
"last_link_id": 15,
"nodes": [
{
"id": 12,
"type": "TextNode",
"id": 13,
"type": "LoadAudio",
"pos": [
47,
398
502,
432
],
"size": {
"0": 400,
"1": 200
"0": 315,
"1": 124
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"name": "AUDIO",
"type": "AUDIO",
"links": [
9
14
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "TextNode"
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"希望你以后能够做的比我还好呦。"
"zero_shot_prompt.wav",
null,
""
]
},
{
@@ -52,7 +54,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
5
12
],
"shape": 3,
"slot_index": 0
@@ -66,15 +68,47 @@
]
},
{
"id": 9,
"id": 12,
"type": "TextNode",
"pos": [
47,
398
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
13
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "TextNode"
},
"widgets_values": [
"希望你以后能够做的比我还好呦。"
]
},
{
"id": 15,
"type": "CosyVoiceNode",
"pos": [
553,
127
559,
115
],
"size": {
"0": 315,
"1": 190
"1": 214
},
"flags": {},
"order": 3,
@@ -83,18 +117,17 @@
{
"name": "tts_text",
"type": "TEXT",
"link": 5
"link": 12
},
{
"name": "prompt_text",
"type": "TEXT",
"link": 9,
"slot_index": 1
"link": 13
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": 10,
"link": 14,
"slot_index": 2
},
{
@@ -108,7 +141,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
11
15
],
"shape": 3,
"slot_index": 0
@@ -118,45 +151,13 @@
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
"3s极速复刻",
1.5,
"预训练音色",
"中文女",
1873,
2002,
"randomize"
]
},
{
"id": 13,
"type": "LoadAudio",
"pos": [
502,
432
],
"size": {
"0": 315,
"1": 124
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
10
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"zero_shot_prompt.wav",
null,
""
]
},
{
"id": 14,
"type": "SaveAudio",
@@ -175,7 +176,7 @@
{
"name": "audio",
"type": "AUDIO",
"link": 11
"link": 15
}
],
"properties": {
@@ -189,32 +190,32 @@
],
"links": [
[
5,
12,
2,
0,
9,
15,
0,
"TEXT"
],
[
9,
13,
12,
0,
9,
15,
1,
"TEXT"
],
[
10,
14,
13,
0,
9,
15,
2,
"AUDIO"
],
[
11,
9,
15,
15,
0,
14,
0,
+45 -46
View File
@@ -1,35 +1,7 @@
{
"last_node_id": 11,
"last_link_id": 8,
"last_node_id": 12,
"last_link_id": 10,
"nodes": [
{
"id": 3,
"type": "PreviewAudio",
"pos": [
970,
118
],
"size": {
"0": 315,
"1": 76
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "AUDIO",
"link": 6
}
],
"properties": {
"Node name for S&R": "PreviewAudio"
},
"widgets_values": [
null
]
},
{
"id": 2,
"type": "TextNode",
@@ -49,7 +21,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
5
9
],
"shape": 3,
"slot_index": 0
@@ -63,15 +35,15 @@
]
},
{
"id": 9,
"id": 12,
"type": "CosyVoiceNode",
"pos": [
553,
127
578,
128
],
"size": {
"0": 315,
"1": 190
"1": 214
},
"flags": {},
"order": 1,
@@ -80,19 +52,17 @@
{
"name": "tts_text",
"type": "TEXT",
"link": 5
"link": 9
},
{
"name": "prompt_text",
"type": "TEXT",
"link": null,
"slot_index": 1
"link": null
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": null,
"slot_index": 2
"link": null
},
{
"name": "instruct_text",
@@ -105,7 +75,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
6
10
],
"shape": 3,
"slot_index": 0
@@ -115,25 +85,54 @@
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
1.2000000000000002,
"预训练音色",
"中文女",
1712,
1365,
"randomize"
]
},
{
"id": 3,
"type": "PreviewAudio",
"pos": [
970,
118
],
"size": {
"0": 315,
"1": 76
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "AUDIO",
"link": 10
}
],
"properties": {
"Node name for S&R": "PreviewAudio"
},
"widgets_values": [
null
]
}
],
"links": [
[
5,
9,
2,
0,
9,
12,
0,
"TEXT"
],
[
6,
9,
10,
12,
0,
3,
0,
@@ -1,98 +1,7 @@
{
"last_node_id": 14,
"last_link_id": 11,
"last_node_id": 15,
"last_link_id": 14,
"nodes": [
{
"id": 9,
"type": "CosyVoiceNode",
"pos": [
553,
127
],
"size": {
"0": 315,
"1": 190
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "tts_text",
"type": "TEXT",
"link": 5
},
{
"name": "prompt_text",
"type": "TEXT",
"link": null,
"slot_index": 1
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": 10,
"slot_index": 2
},
{
"name": "instruct_text",
"type": "TEXT",
"link": null
}
],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
11
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
"跨语种复刻",
"中文女",
1883,
"randomize"
]
},
{
"id": 2,
"type": "TextNode",
"pos": [
77,
137
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
5
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "TextNode"
},
"widgets_values": [
"<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing."
]
},
{
"id": 13,
"type": "LoadAudio",
@@ -105,14 +14,14 @@
"1": 124
},
"flags": {},
"order": 1,
"order": 0,
"mode": 0,
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
10
13
],
"shape": 3
}
@@ -144,7 +53,7 @@
{
"name": "audio",
"type": "AUDIO",
"link": 11
"link": 14
}
],
"properties": {
@@ -154,28 +63,119 @@
"audio/ComfyUI",
null
]
},
{
"id": 2,
"type": "TextNode",
"pos": [
77,
137
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
12
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "TextNode"
},
"widgets_values": [
"<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing."
]
},
{
"id": 15,
"type": "CosyVoiceNode",
"pos": [
568,
105
],
"size": {
"0": 315,
"1": 214
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "tts_text",
"type": "TEXT",
"link": 12
},
{
"name": "prompt_text",
"type": "TEXT",
"link": null
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": 13,
"slot_index": 2
},
{
"name": "instruct_text",
"type": "TEXT",
"link": null
}
],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
14
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
1,
"预训练音色",
"中文女",
42,
"randomize"
]
}
],
"links": [
[
5,
12,
2,
0,
9,
15,
0,
"TEXT"
],
[
10,
13,
13,
0,
9,
15,
2,
"AUDIO"
],
[
11,
9,
14,
15,
0,
14,
0,
@@ -1,6 +1,6 @@
{
"last_node_id": 8,
"last_link_id": 8,
"last_node_id": 9,
"last_link_id": 11,
"nodes": [
{
"id": 2,
@@ -36,89 +36,6 @@
""
]
},
{
"id": 3,
"type": "VocalSeparationNode",
"pos": [
527,
62
],
"size": {
"0": 315,
"1": 126
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "music",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "vocals_AUDIO",
"type": "AUDIO",
"links": [
3,
4
],
"shape": 3,
"slot_index": 0
},
{
"name": "instrumental_AUDIO",
"type": "AUDIO",
"links": [
5
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "VocalSeparationNode"
},
"widgets_values": [
"bs_roformer",
4,
true
]
},
{
"id": 7,
"type": "LoadSRT",
"pos": [
70,
133
],
"size": {
"0": 315,
"1": 82
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "SRT",
"type": "SRT",
"links": [
7
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadSRT"
},
"widgets_values": [
"zh_test.srt",
"Audio"
]
},
{
"id": 4,
"type": "PreviewAudio",
@@ -131,7 +48,7 @@
"1": 76
},
"flags": {},
"order": 5,
"order": 4,
"mode": 0,
"inputs": [
{
@@ -193,7 +110,7 @@
{
"name": "audio",
"type": "AUDIO",
"link": 6
"link": 10
}
],
"properties": {
@@ -204,36 +121,118 @@
]
},
{
"id": 1,
"type": "CosyVoiceDubbingNode",
"id": 3,
"type": "VocalSeparationNode",
"pos": [
550,
374
527,
62
],
"size": {
"0": 315,
"1": 146
"1": 126
},
"flags": {},
"order": 4,
"order": 3,
"mode": 0,
"inputs": [
{
"name": "music",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "vocals_AUDIO",
"type": "AUDIO",
"links": [
4,
8
],
"shape": 3,
"slot_index": 0
},
{
"name": "instrumental_AUDIO",
"type": "AUDIO",
"links": [
5
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "VocalSeparationNode"
},
"widgets_values": [
"bs_roformer",
4,
true
]
},
{
"id": 7,
"type": "LoadSRT",
"pos": [
70,
133
],
"size": {
"0": 315,
"1": 82
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "SRT",
"type": "SRT",
"links": [
9
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadSRT"
},
"widgets_values": [
"zh_test.srt",
"Audio"
]
},
{
"id": 8,
"type": "CosyVoiceDubbingNode",
"pos": [
532,
365
],
"size": {
"0": 315,
"1": 170
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "tts_srt",
"type": "SRT",
"link": 7,
"slot_index": 0
"link": 9
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": 3,
"slot_index": 1
"link": 8
},
{
"name": "prompt_srt",
"type": "SRT",
"link": 8,
"link": 11,
"slot_index": 2
}
],
@@ -242,7 +241,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
6
10
],
"shape": 3,
"slot_index": 0
@@ -253,16 +252,17 @@
},
"widgets_values": [
"<|zh|>",
1085,
true,
42,
"randomize"
]
},
{
"id": 8,
"id": 9,
"type": "LoadSRT",
"pos": [
118,
535
115,
549
],
"size": {
"0": 315,
@@ -276,7 +276,7 @@
"name": "SRT",
"type": "SRT",
"links": [
8
11
],
"shape": 3
}
@@ -285,7 +285,7 @@
"Node name for S&R": "LoadSRT"
},
"widgets_values": [
"zh_test.srt",
"en_test.srt",
"Audio"
]
}
@@ -299,14 +299,6 @@
0,
"AUDIO"
],
[
3,
3,
0,
1,
1,
"AUDIO"
],
[
4,
3,
@@ -324,26 +316,34 @@
"AUDIO"
],
[
6,
8,
3,
0,
8,
1,
0,
6,
0,
"AUDIO"
],
[
7,
9,
7,
0,
1,
8,
0,
"SRT"
],
[
8,
10,
8,
0,
1,
6,
0,
"AUDIO"
],
[
11,
9,
0,
8,
2,
"SRT"
]
@@ -1,6 +1,6 @@
{
"last_node_id": 7,
"last_link_id": 7,
"last_node_id": 8,
"last_link_id": 10,
"nodes": [
{
"id": 2,
@@ -36,57 +36,6 @@
""
]
},
{
"id": 3,
"type": "VocalSeparationNode",
"pos": [
527,
62
],
"size": {
"0": 315,
"1": 126
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "music",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "vocals_AUDIO",
"type": "AUDIO",
"links": [
3,
4
],
"shape": 3,
"slot_index": 0
},
{
"name": "instrumental_AUDIO",
"type": "AUDIO",
"links": [
5
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "VocalSeparationNode"
},
"widgets_values": [
"bs_roformer",
4,
true
]
},
{
"id": 4,
"type": "PreviewAudio",
@@ -99,7 +48,7 @@
"1": 76
},
"flags": {},
"order": 4,
"order": 3,
"mode": 0,
"inputs": [
{
@@ -161,7 +110,7 @@
{
"name": "audio",
"type": "AUDIO",
"link": 6
"link": 10
}
],
"properties": {
@@ -171,6 +120,57 @@
null
]
},
{
"id": 3,
"type": "VocalSeparationNode",
"pos": [
527,
62
],
"size": {
"0": 315,
"1": 126
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "music",
"type": "AUDIO",
"link": 2
}
],
"outputs": [
{
"name": "vocals_AUDIO",
"type": "AUDIO",
"links": [
4,
8
],
"shape": 3,
"slot_index": 0
},
{
"name": "instrumental_AUDIO",
"type": "AUDIO",
"links": [
5
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "VocalSeparationNode"
},
"widgets_values": [
"bs_roformer",
4,
true
]
},
{
"id": 7,
"type": "LoadSRT",
@@ -190,9 +190,10 @@
"name": "SRT",
"type": "SRT",
"links": [
7
9
],
"shape": 3
"shape": 3,
"slot_index": 0
}
],
"properties": {
@@ -204,31 +205,29 @@
]
},
{
"id": 1,
"id": 8,
"type": "CosyVoiceDubbingNode",
"pos": [
550,
374
532,
365
],
"size": {
"0": 315,
"1": 146
"1": 170
},
"flags": {},
"order": 3,
"order": 4,
"mode": 0,
"inputs": [
{
"name": "tts_srt",
"type": "SRT",
"link": 7,
"slot_index": 0
"link": 9
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": 3,
"slot_index": 1
"link": 8
},
{
"name": "prompt_srt",
@@ -241,7 +240,7 @@
"name": "AUDIO",
"type": "AUDIO",
"links": [
6
10
],
"shape": 3,
"slot_index": 0
@@ -252,6 +251,7 @@
},
"widgets_values": [
"<|zh|>",
true,
42,
"randomize"
]
@@ -266,14 +266,6 @@
0,
"AUDIO"
],
[
3,
3,
0,
1,
1,
"AUDIO"
],
[
4,
3,
@@ -291,20 +283,28 @@
"AUDIO"
],
[
6,
8,
3,
0,
8,
1,
0,
6,
0,
"AUDIO"
],
[
7,
9,
7,
0,
1,
8,
0,
"SRT"
],
[
10,
8,
0,
6,
0,
"AUDIO"
]
],
"groups": [],
+110 -111
View File
@@ -1,13 +1,13 @@
{
"last_node_id": 15,
"last_link_id": 12,
"last_node_id": 16,
"last_link_id": 15,
"nodes": [
{
"id": 2,
"id": 15,
"type": "TextNode",
"pos": [
77,
137
126,
426
],
"size": {
"0": 400,
@@ -21,99 +21,7 @@
"name": "TEXT",
"type": "TEXT",
"links": [
5
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "TextNode"
},
"widgets_values": [
"在面对挑战时,他展现了非凡的<strong>勇气</strong>与<strong>智慧</strong>。'"
]
},
{
"id": 9,
"type": "CosyVoiceNode",
"pos": [
553,
127
],
"size": {
"0": 315,
"1": 190
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "tts_text",
"type": "TEXT",
"link": 5
},
{
"name": "prompt_text",
"type": "TEXT",
"link": null,
"slot_index": 1
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": null,
"slot_index": 2
},
{
"name": "instruct_text",
"type": "TEXT",
"link": 12,
"slot_index": 3
}
],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
11
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
"自然语言控制",
"中文女",
208,
"randomize"
]
},
{
"id": 15,
"type": "TextNode",
"pos": [
126,
426
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
12
14
],
"shape": 3
}
@@ -143,7 +51,7 @@
{
"name": "audio",
"type": "AUDIO",
"link": 11
"link": 15
}
],
"properties": {
@@ -153,32 +61,123 @@
"audio/ComfyUI",
null
]
},
{
"id": 2,
"type": "TextNode",
"pos": [
77,
137
],
"size": {
"0": 400,
"1": 200
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "TEXT",
"type": "TEXT",
"links": [
13
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "TextNode"
},
"widgets_values": [
"在面对挑战时,他展现了非凡的<strong>勇气</strong>与<strong>智慧</strong>。'"
]
},
{
"id": 16,
"type": "CosyVoiceNode",
"pos": [
569,
117
],
"size": {
"0": 315,
"1": 214
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [
{
"name": "tts_text",
"type": "TEXT",
"link": 13
},
{
"name": "prompt_text",
"type": "TEXT",
"link": null
},
{
"name": "prompt_wav",
"type": "AUDIO",
"link": null
},
{
"name": "instruct_text",
"type": "TEXT",
"link": 14,
"slot_index": 3
}
],
"outputs": [
{
"name": "AUDIO",
"type": "AUDIO",
"links": [
15
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CosyVoiceNode"
},
"widgets_values": [
1,
"预训练音色",
"中文女",
368,
"randomize"
]
}
],
"links": [
[
5,
13,
2,
0,
9,
16,
0,
"TEXT"
],
[
11,
9,
14,
15,
0,
16,
3,
"TEXT"
],
[
15,
16,
0,
14,
0,
"AUDIO"
],
[
12,
15,
0,
9,
3,
"TEXT"
]
],
"groups": [],