From e484835e95801b3ea3199ffd59655e4985dccd26 Mon Sep 17 00:00:00 2001 From: AIFSH <1509359472@qq.com> Date: Thu, 18 Jul 2024 21:45:20 +0800 Subject: [PATCH] add speed control --- README.md | 11 + nodes.py | 116 +++++++-- requirements.txt | 2 +- ...ne_workflow.json => 3sclone_workflow.json} | 135 +++++----- workflows/base_workflow.json | 91 ++++--- ...gual_workflow.json => cross_workflow.json} | 204 +++++++-------- ...3s_workflow.json => 3sclone_workflow.json} | 242 +++++++++--------- ..._workflow.json => base_dubb_workflow.json} | 162 ++++++------ workflows/instruct_workflow.json | 221 ++++++++-------- 9 files changed, 629 insertions(+), 555 deletions(-) rename workflows/{3s_clone_workflow.json => 3sclone_workflow.json} (89%) rename workflows/{cross_lingual_workflow.json => cross_workflow.json} (89%) rename workflows/dubbing/{3s_workflow.json => 3sclone_workflow.json} (92%) rename workflows/dubbing/{base_workflow.json => base_dubb_workflow.json} (93%) diff --git a/README.md b/README.md index 03c587d..405cdc5 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,17 @@ # CosyVoice-ComfyUI a comfyui custom node for [CosyVoice](https://github.com/FunAudioLLM/CosyVoice),you can find workflow in [workflows](./workflows/) +## new Feature +suport `srt` file to single voice or mutiple voice clone + +input +- [tts_srt](./workflows/dubbing/zh_test.srt) +- [prompt_wav](./workflows/dubbing/test.mp3) +- [prompt_srt](./workflows/dubbing/en_test.srt)(optional) + +output + + ## Example test on 2080ti 11GB torch==2.3.0+cu121 python 3.10.8 diff --git a/nodes.py b/nodes.py index 0ddf0bb..9dc3d66 100644 --- a/nodes.py +++ b/nodes.py @@ -14,12 +14,9 @@ pretrained_models = os.path.join(now_dir,"pretrained_models") from modelscope import snapshot_download +import ffmpeg import audiosegment from srt import parse as SrtPare -from audiosegment import AudioSegment -import audiotsm -from audiotsm.io.wav import WavReader, WavWriter - from cosyvoice.cli.cosyvoice import CosyVoice sft_spk_list = ['中文女', '中文男', '日语男', '粤语女', '英文女', '英文男', '韩语女'] @@ -44,6 +41,32 @@ def postprocess(speech, top_db=60, hop_length=220, win_length=440): speech = torch.concat([speech, torch.zeros(1, int(target_sr * 0.2))], dim=1) return speech +def speed_change(input_audio, speed, sr): + # 检查输入数据类型和声道数 + if input_audio.dtype != np.int16: + raise ValueError("输入音频数据类型必须为 np.int16") + + + # 转换为字节流 + raw_audio = input_audio.astype(np.int16).tobytes() + + # 设置 ffmpeg 输入流 + input_stream = ffmpeg.input('pipe:', format='s16le', acodec='pcm_s16le', ar=str(sr), ac=1) + + # 变速处理 + output_stream = input_stream.filter('atempo', speed) + + # 输出流到管道 + out, _ = ( + output_stream.output('pipe:', format='s16le', acodec='pcm_s16le') + .run(input=raw_audio, capture_stdout=True, capture_stderr=True) + ) + + # 将管道输出解码为 NumPy 数组 + processed_audio = np.frombuffer(out, np.int16) + + return processed_audio + class TextNode: @classmethod def INPUT_TYPES(s): @@ -66,6 +89,9 @@ class CosyVoiceNode: return { "required":{ "tts_text":("TEXT",), + "speed":("FLOAT",{ + "default": 1.0 + }), "inference_mode":(inference_mode_list,{ "default": "预训练音色" }), @@ -91,7 +117,7 @@ class CosyVoiceNode: CATEGORY = "AIFSH_CosyVoice" - def generate(self,tts_text,inference_mode,sft_dropdown,seed, + def generate(self,tts_text,speed,inference_mode,sft_dropdown,seed, prompt_text=None,prompt_wav=None,instruct_text=None): if inference_mode == '自然语言控制': model_dir = os.path.join(pretrained_models,"CosyVoice-300M-Instruct") @@ -140,7 +166,11 @@ class CosyVoiceNode: set_all_random_seed(seed) print(self.model_dir) output = self.cosyvoice.inference_instruct(tts_text, sft_dropdown, instruct_text) - audio = {"waveform": [output['tts_speech']],"sample_rate":target_sr} + + output_numpy = output['tts_speech'].squeeze(0).numpy() * 32768 + output_numpy = output_numpy.astype(np.int16) + output_numpy = speed_change(output_numpy,speed,target_sr) + audio = {"waveform": [torch.Tensor(output_numpy/32768).unsqueeze(0)],"sample_rate":target_sr} return (audio,) class CosyVoiceDubbingNode: @@ -154,6 +184,9 @@ class CosyVoiceDubbingNode: "tts_srt":("SRT",), "prompt_wav": ("AUDIO",), "language":(["<|zh|>","<|en|>","<|jp|>","<|yue|>","<|ko|>"],), + "if_single":("BOOLEAN",{ + "default": True + }), "seed":("INT",{ "default": 42 }) @@ -171,7 +204,7 @@ class CosyVoiceDubbingNode: CATEGORY = "AIFSH_CosyVoice" - def generate(self,tts_srt,prompt_wav,language,seed,prompt_srt=None): + def generate(self,tts_srt,prompt_wav,language,if_single,seed,prompt_srt=None): model_dir = os.path.join(pretrained_models,"CosyVoice-300M") snapshot_download(model_id="iic/CosyVoice-300M",local_dir=model_dir) set_all_random_seed(seed) @@ -195,6 +228,7 @@ class CosyVoiceDubbingNode: speech_numpy = speech.squeeze(0).numpy() * 32768 speech_numpy = speech_numpy.astype(np.int16) audio_seg = audiosegment.from_numpy_array(speech_numpy,prompt_sr) + assert audio_seg.duration_seconds > 3, "prompt wav should be > 3s" # audio_seg.export(os.path.join(output_dir,"test.mp3"),format="mp3") new_audio_seg = audiosegment.silent(0,target_sr) for i,text_sub in enumerate(text_subtitles): @@ -202,16 +236,53 @@ class CosyVoiceDubbingNode: end_time = text_sub.end.total_seconds() * 1000 if i == 0: new_audio_seg += audio_seg[:start_time] - - curr_tts_text = language + text_sub.content + + if if_single: + curr_tts_text = language + text_sub.content + else: + curr_tts_text = language + text_sub.content[1:] + speaker_id = text_sub.content[0] prompt_wav_seg = audio_seg[start_time:end_time] - # prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav") + if prompt_srt: + prompt_text_list = [prompt_subtitles[i].content] + while prompt_wav_seg.duration_seconds < 30: + for j in range(i+1,len(text_subtitles)): + j_start = text_subtitles[j].start.total_seconds() * 1000 + j_end = text_subtitles[j].end.total_seconds() * 1000 + if if_single: + prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end]) + if prompt_srt: + prompt_text_list.append(prompt_subtitles[j].content) + else: + if text_subtitles[j].content[0] == speaker_id: + prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end]) + if prompt_srt: + prompt_text_list.append(prompt_subtitles[j].content) + for j in range(0,i): + j_start = text_subtitles[j].start.total_seconds() * 1000 + j_end = text_subtitles[j].end.total_seconds() * 1000 + if if_single: + prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end]) + if prompt_srt: + prompt_text_list.append(prompt_subtitles[j].content) + else: + if text_subtitles[j].content[0] == speaker_id: + prompt_wav_seg += (audiosegment.silent(500,frame_rate=prompt_sr) + audio_seg[j_start:j_end]) + if prompt_srt: + prompt_text_list.append(prompt_subtitles[j].content) + + if prompt_wav_seg.duration_seconds > 3: + break + print(f"prompt_wav {prompt_wav_seg.duration_seconds}s") + prompt_wav_seg.export(os.path.join(output_dir,f"{i}_prompt.wav"),format="wav") prompt_wav_seg_numpy = prompt_wav_seg.to_numpy_array() / 32768 # print(prompt_wav_seg_numpy.shape) prompt_speech_16k = postprocess(torch.Tensor(prompt_wav_seg_numpy).unsqueeze(0)) if prompt_srt: - prompt_text = prompt_subtitles[i].content + # prompt_text = prompt_subtitles[i].content + prompt_text = ','.join(prompt_text_list) + print(f"prompt_text:{prompt_text}") curr_output = self.cosyvoice.inference_zero_shot(curr_tts_text,prompt_text,prompt_speech_16k) else: curr_output = self.cosyvoice.inference_cross_lingual(curr_tts_text, prompt_speech_16k) @@ -233,31 +304,24 @@ class CosyVoiceDubbingNode: ratio = text_audio_dur_time / dur_time if text_audio_dur_time > dur_time: - tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav") + tmp_numpy = speed_change(curr_output_numpy,ratio,target_sr) + tmp_audio = audiosegment.from_numpy_array(tmp_numpy,target_sr) + # tmp_audio = self.map_vocal(text_audio,ratio,dur_time,f"{i}_res.wav") tmp_audio += audiosegment.silent(dur_time - tmp_audio.duration_seconds*1000,target_sr) else: tmp_audio = text_audio + audiosegment.silent(dur_time - text_audio_dur_time,target_sr) new_audio_seg += tmp_audio + + if i == len(text_subtitles) - 1: + new_audio_seg += audio_seg[end_time:] + output_numpy = new_audio_seg.to_numpy_array() / 32768 # print(output_numpy.shape) audio = {"waveform": [torch.Tensor(output_numpy).unsqueeze(0)],"sample_rate":target_sr} return (audio,) - - def map_vocal(self,audio,ratio,dur_time,wav_name): - os.makedirs(output_dir, exist_ok=True) - tmp_path = f"{output_dir}/{wav_name}" - audio.export(tmp_path, format="wav") - - clone_path = f"{output_dir}/speed_{wav_name}" - - with WavReader(tmp_path) as reader: - with WavWriter(clone_path, reader.channels, reader.samplerate) as writer: - tsm = audiotsm.phasevocoder(reader.channels, speed=ratio) - tsm.run(reader, writer) - audio_extended = audiosegment.from_file(clone_path) - return audio_extended[:dur_time] + class LoadSRT: @classmethod def INPUT_TYPES(s): diff --git a/requirements.txt b/requirements.txt index 016638c..01a5b20 100644 --- a/requirements.txt +++ b/requirements.txt @@ -28,4 +28,4 @@ pypinyin pydub audiosegment srt -audiotsm \ No newline at end of file +ffmpeg-python \ No newline at end of file diff --git a/workflows/3s_clone_workflow.json b/workflows/3sclone_workflow.json similarity index 89% rename from workflows/3s_clone_workflow.json rename to workflows/3sclone_workflow.json index ec31346..93ae456 100644 --- a/workflows/3s_clone_workflow.json +++ b/workflows/3sclone_workflow.json @@ -1,36 +1,38 @@ { - "last_node_id": 14, - "last_link_id": 11, + "last_node_id": 15, + "last_link_id": 15, "nodes": [ { - "id": 12, - "type": "TextNode", + "id": 13, + "type": "LoadAudio", "pos": [ - 47, - 398 + 502, + 432 ], "size": { - "0": 400, - "1": 200 + "0": 315, + "1": 124 }, "flags": {}, "order": 0, "mode": 0, "outputs": [ { - "name": "TEXT", - "type": "TEXT", + "name": "AUDIO", + "type": "AUDIO", "links": [ - 9 + 14 ], "shape": 3 } ], "properties": { - "Node name for S&R": "TextNode" + "Node name for S&R": "LoadAudio" }, "widgets_values": [ - "希望你以后能够做的比我还好呦。" + "zero_shot_prompt.wav", + null, + "" ] }, { @@ -52,7 +54,7 @@ "name": "TEXT", "type": "TEXT", "links": [ - 5 + 12 ], "shape": 3, "slot_index": 0 @@ -66,15 +68,47 @@ ] }, { - "id": 9, + "id": 12, + "type": "TextNode", + "pos": [ + 47, + 398 + ], + "size": { + "0": 400, + "1": 200 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "TEXT", + "type": "TEXT", + "links": [ + 13 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TextNode" + }, + "widgets_values": [ + "希望你以后能够做的比我还好呦。" + ] + }, + { + "id": 15, "type": "CosyVoiceNode", "pos": [ - 553, - 127 + 559, + 115 ], "size": { "0": 315, - "1": 190 + "1": 214 }, "flags": {}, "order": 3, @@ -83,18 +117,17 @@ { "name": "tts_text", "type": "TEXT", - "link": 5 + "link": 12 }, { "name": "prompt_text", "type": "TEXT", - "link": 9, - "slot_index": 1 + "link": 13 }, { "name": "prompt_wav", "type": "AUDIO", - "link": 10, + "link": 14, "slot_index": 2 }, { @@ -108,7 +141,7 @@ "name": "AUDIO", "type": "AUDIO", "links": [ - 11 + 15 ], "shape": 3, "slot_index": 0 @@ -118,45 +151,13 @@ "Node name for S&R": "CosyVoiceNode" }, "widgets_values": [ - "3s极速复刻", + 1.5, + "预训练音色", "中文女", - 1873, + 2002, "randomize" ] }, - { - "id": 13, - "type": "LoadAudio", - "pos": [ - 502, - 432 - ], - "size": { - "0": 315, - "1": 124 - }, - "flags": {}, - "order": 2, - "mode": 0, - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 10 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "LoadAudio" - }, - "widgets_values": [ - "zero_shot_prompt.wav", - null, - "" - ] - }, { "id": 14, "type": "SaveAudio", @@ -175,7 +176,7 @@ { "name": "audio", "type": "AUDIO", - "link": 11 + "link": 15 } ], "properties": { @@ -189,32 +190,32 @@ ], "links": [ [ - 5, + 12, 2, 0, - 9, + 15, 0, "TEXT" ], [ - 9, + 13, 12, 0, - 9, + 15, 1, "TEXT" ], [ - 10, + 14, 13, 0, - 9, + 15, 2, "AUDIO" ], [ - 11, - 9, + 15, + 15, 0, 14, 0, diff --git a/workflows/base_workflow.json b/workflows/base_workflow.json index c758af9..f91e968 100644 --- a/workflows/base_workflow.json +++ b/workflows/base_workflow.json @@ -1,35 +1,7 @@ { - "last_node_id": 11, - "last_link_id": 8, + "last_node_id": 12, + "last_link_id": 10, "nodes": [ - { - "id": 3, - "type": "PreviewAudio", - "pos": [ - 970, - 118 - ], - "size": { - "0": 315, - "1": 76 - }, - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "audio", - "type": "AUDIO", - "link": 6 - } - ], - "properties": { - "Node name for S&R": "PreviewAudio" - }, - "widgets_values": [ - null - ] - }, { "id": 2, "type": "TextNode", @@ -49,7 +21,7 @@ "name": "TEXT", "type": "TEXT", "links": [ - 5 + 9 ], "shape": 3, "slot_index": 0 @@ -63,15 +35,15 @@ ] }, { - "id": 9, + "id": 12, "type": "CosyVoiceNode", "pos": [ - 553, - 127 + 578, + 128 ], "size": { "0": 315, - "1": 190 + "1": 214 }, "flags": {}, "order": 1, @@ -80,19 +52,17 @@ { "name": "tts_text", "type": "TEXT", - "link": 5 + "link": 9 }, { "name": "prompt_text", "type": "TEXT", - "link": null, - "slot_index": 1 + "link": null }, { "name": "prompt_wav", "type": "AUDIO", - "link": null, - "slot_index": 2 + "link": null }, { "name": "instruct_text", @@ -105,7 +75,7 @@ "name": "AUDIO", "type": "AUDIO", "links": [ - 6 + 10 ], "shape": 3, "slot_index": 0 @@ -115,25 +85,54 @@ "Node name for S&R": "CosyVoiceNode" }, "widgets_values": [ + 1.2000000000000002, "预训练音色", "中文女", - 1712, + 1365, "randomize" ] + }, + { + "id": 3, + "type": "PreviewAudio", + "pos": [ + 970, + 118 + ], + "size": { + "0": 315, + "1": 76 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "audio", + "type": "AUDIO", + "link": 10 + } + ], + "properties": { + "Node name for S&R": "PreviewAudio" + }, + "widgets_values": [ + null + ] } ], "links": [ [ - 5, + 9, 2, 0, - 9, + 12, 0, "TEXT" ], [ - 6, - 9, + 10, + 12, 0, 3, 0, diff --git a/workflows/cross_lingual_workflow.json b/workflows/cross_workflow.json similarity index 89% rename from workflows/cross_lingual_workflow.json rename to workflows/cross_workflow.json index 82e860d..4ced739 100644 --- a/workflows/cross_lingual_workflow.json +++ b/workflows/cross_workflow.json @@ -1,98 +1,7 @@ { - "last_node_id": 14, - "last_link_id": 11, + "last_node_id": 15, + "last_link_id": 14, "nodes": [ - { - "id": 9, - "type": "CosyVoiceNode", - "pos": [ - 553, - 127 - ], - "size": { - "0": 315, - "1": 190 - }, - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "tts_text", - "type": "TEXT", - "link": 5 - }, - { - "name": "prompt_text", - "type": "TEXT", - "link": null, - "slot_index": 1 - }, - { - "name": "prompt_wav", - "type": "AUDIO", - "link": 10, - "slot_index": 2 - }, - { - "name": "instruct_text", - "type": "TEXT", - "link": null - } - ], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 11 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "CosyVoiceNode" - }, - "widgets_values": [ - "跨语种复刻", - "中文女", - 1883, - "randomize" - ] - }, - { - "id": 2, - "type": "TextNode", - "pos": [ - 77, - 137 - ], - "size": { - "0": 400, - "1": 200 - }, - "flags": {}, - "order": 0, - "mode": 0, - "outputs": [ - { - "name": "TEXT", - "type": "TEXT", - "links": [ - 5 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "TextNode" - }, - "widgets_values": [ - "<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing." - ] - }, { "id": 13, "type": "LoadAudio", @@ -105,14 +14,14 @@ "1": 124 }, "flags": {}, - "order": 1, + "order": 0, "mode": 0, "outputs": [ { "name": "AUDIO", "type": "AUDIO", "links": [ - 10 + 13 ], "shape": 3 } @@ -144,7 +53,7 @@ { "name": "audio", "type": "AUDIO", - "link": 11 + "link": 14 } ], "properties": { @@ -154,28 +63,119 @@ "audio/ComfyUI", null ] + }, + { + "id": 2, + "type": "TextNode", + "pos": [ + 77, + 137 + ], + "size": { + "0": 400, + "1": 200 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "TEXT", + "type": "TEXT", + "links": [ + 12 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TextNode" + }, + "widgets_values": [ + "<|en|>And then later on, fully acquiring that company. So keeping management in line, interest in line with the asset that\\'s coming into the family is a reason why sometimes we don\\'t buy the whole thing." + ] + }, + { + "id": 15, + "type": "CosyVoiceNode", + "pos": [ + 568, + 105 + ], + "size": { + "0": 315, + "1": 214 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "tts_text", + "type": "TEXT", + "link": 12 + }, + { + "name": "prompt_text", + "type": "TEXT", + "link": null + }, + { + "name": "prompt_wav", + "type": "AUDIO", + "link": 13, + "slot_index": 2 + }, + { + "name": "instruct_text", + "type": "TEXT", + "link": null + } + ], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 14 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CosyVoiceNode" + }, + "widgets_values": [ + 1, + "预训练音色", + "中文女", + 42, + "randomize" + ] } ], "links": [ [ - 5, + 12, 2, 0, - 9, + 15, 0, "TEXT" ], [ - 10, + 13, 13, 0, - 9, + 15, 2, "AUDIO" ], [ - 11, - 9, + 14, + 15, 0, 14, 0, diff --git a/workflows/dubbing/3s_workflow.json b/workflows/dubbing/3sclone_workflow.json similarity index 92% rename from workflows/dubbing/3s_workflow.json rename to workflows/dubbing/3sclone_workflow.json index 345c3bf..691a377 100644 --- a/workflows/dubbing/3s_workflow.json +++ b/workflows/dubbing/3sclone_workflow.json @@ -1,6 +1,6 @@ { - "last_node_id": 8, - "last_link_id": 8, + "last_node_id": 9, + "last_link_id": 11, "nodes": [ { "id": 2, @@ -36,89 +36,6 @@ "" ] }, - { - "id": 3, - "type": "VocalSeparationNode", - "pos": [ - 527, - 62 - ], - "size": { - "0": 315, - "1": 126 - }, - "flags": {}, - "order": 3, - "mode": 0, - "inputs": [ - { - "name": "music", - "type": "AUDIO", - "link": 2 - } - ], - "outputs": [ - { - "name": "vocals_AUDIO", - "type": "AUDIO", - "links": [ - 3, - 4 - ], - "shape": 3, - "slot_index": 0 - }, - { - "name": "instrumental_AUDIO", - "type": "AUDIO", - "links": [ - 5 - ], - "shape": 3, - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "VocalSeparationNode" - }, - "widgets_values": [ - "bs_roformer", - 4, - true - ] - }, - { - "id": 7, - "type": "LoadSRT", - "pos": [ - 70, - 133 - ], - "size": { - "0": 315, - "1": 82 - }, - "flags": {}, - "order": 1, - "mode": 0, - "outputs": [ - { - "name": "SRT", - "type": "SRT", - "links": [ - 7 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "LoadSRT" - }, - "widgets_values": [ - "zh_test.srt", - "Audio" - ] - }, { "id": 4, "type": "PreviewAudio", @@ -131,7 +48,7 @@ "1": 76 }, "flags": {}, - "order": 5, + "order": 4, "mode": 0, "inputs": [ { @@ -193,7 +110,7 @@ { "name": "audio", "type": "AUDIO", - "link": 6 + "link": 10 } ], "properties": { @@ -204,36 +121,118 @@ ] }, { - "id": 1, - "type": "CosyVoiceDubbingNode", + "id": 3, + "type": "VocalSeparationNode", "pos": [ - 550, - 374 + 527, + 62 ], "size": { "0": 315, - "1": 146 + "1": 126 }, "flags": {}, - "order": 4, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "music", + "type": "AUDIO", + "link": 2 + } + ], + "outputs": [ + { + "name": "vocals_AUDIO", + "type": "AUDIO", + "links": [ + 4, + 8 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "instrumental_AUDIO", + "type": "AUDIO", + "links": [ + 5 + ], + "shape": 3, + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "VocalSeparationNode" + }, + "widgets_values": [ + "bs_roformer", + 4, + true + ] + }, + { + "id": 7, + "type": "LoadSRT", + "pos": [ + 70, + 133 + ], + "size": { + "0": 315, + "1": 82 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "SRT", + "type": "SRT", + "links": [ + 9 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LoadSRT" + }, + "widgets_values": [ + "zh_test.srt", + "Audio" + ] + }, + { + "id": 8, + "type": "CosyVoiceDubbingNode", + "pos": [ + 532, + 365 + ], + "size": { + "0": 315, + "1": 170 + }, + "flags": {}, + "order": 5, "mode": 0, "inputs": [ { "name": "tts_srt", "type": "SRT", - "link": 7, - "slot_index": 0 + "link": 9 }, { "name": "prompt_wav", "type": "AUDIO", - "link": 3, - "slot_index": 1 + "link": 8 }, { "name": "prompt_srt", "type": "SRT", - "link": 8, + "link": 11, "slot_index": 2 } ], @@ -242,7 +241,7 @@ "name": "AUDIO", "type": "AUDIO", "links": [ - 6 + 10 ], "shape": 3, "slot_index": 0 @@ -253,16 +252,17 @@ }, "widgets_values": [ "<|zh|>", - 1085, + true, + 42, "randomize" ] }, { - "id": 8, + "id": 9, "type": "LoadSRT", "pos": [ - 118, - 535 + 115, + 549 ], "size": { "0": 315, @@ -276,7 +276,7 @@ "name": "SRT", "type": "SRT", "links": [ - 8 + 11 ], "shape": 3 } @@ -285,7 +285,7 @@ "Node name for S&R": "LoadSRT" }, "widgets_values": [ - "zh_test.srt", + "en_test.srt", "Audio" ] } @@ -299,14 +299,6 @@ 0, "AUDIO" ], - [ - 3, - 3, - 0, - 1, - 1, - "AUDIO" - ], [ 4, 3, @@ -324,26 +316,34 @@ "AUDIO" ], [ - 6, + 8, + 3, + 0, + 8, 1, - 0, - 6, - 0, "AUDIO" ], [ - 7, + 9, 7, 0, - 1, + 8, 0, "SRT" ], [ - 8, + 10, 8, 0, - 1, + 6, + 0, + "AUDIO" + ], + [ + 11, + 9, + 0, + 8, 2, "SRT" ] diff --git a/workflows/dubbing/base_workflow.json b/workflows/dubbing/base_dubb_workflow.json similarity index 93% rename from workflows/dubbing/base_workflow.json rename to workflows/dubbing/base_dubb_workflow.json index 51b2e13..7bd6745 100644 --- a/workflows/dubbing/base_workflow.json +++ b/workflows/dubbing/base_dubb_workflow.json @@ -1,6 +1,6 @@ { - "last_node_id": 7, - "last_link_id": 7, + "last_node_id": 8, + "last_link_id": 10, "nodes": [ { "id": 2, @@ -36,57 +36,6 @@ "" ] }, - { - "id": 3, - "type": "VocalSeparationNode", - "pos": [ - 527, - 62 - ], - "size": { - "0": 315, - "1": 126 - }, - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "music", - "type": "AUDIO", - "link": 2 - } - ], - "outputs": [ - { - "name": "vocals_AUDIO", - "type": "AUDIO", - "links": [ - 3, - 4 - ], - "shape": 3, - "slot_index": 0 - }, - { - "name": "instrumental_AUDIO", - "type": "AUDIO", - "links": [ - 5 - ], - "shape": 3, - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "VocalSeparationNode" - }, - "widgets_values": [ - "bs_roformer", - 4, - true - ] - }, { "id": 4, "type": "PreviewAudio", @@ -99,7 +48,7 @@ "1": 76 }, "flags": {}, - "order": 4, + "order": 3, "mode": 0, "inputs": [ { @@ -161,7 +110,7 @@ { "name": "audio", "type": "AUDIO", - "link": 6 + "link": 10 } ], "properties": { @@ -171,6 +120,57 @@ null ] }, + { + "id": 3, + "type": "VocalSeparationNode", + "pos": [ + 527, + 62 + ], + "size": { + "0": 315, + "1": 126 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "music", + "type": "AUDIO", + "link": 2 + } + ], + "outputs": [ + { + "name": "vocals_AUDIO", + "type": "AUDIO", + "links": [ + 4, + 8 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "instrumental_AUDIO", + "type": "AUDIO", + "links": [ + 5 + ], + "shape": 3, + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "VocalSeparationNode" + }, + "widgets_values": [ + "bs_roformer", + 4, + true + ] + }, { "id": 7, "type": "LoadSRT", @@ -190,9 +190,10 @@ "name": "SRT", "type": "SRT", "links": [ - 7 + 9 ], - "shape": 3 + "shape": 3, + "slot_index": 0 } ], "properties": { @@ -204,31 +205,29 @@ ] }, { - "id": 1, + "id": 8, "type": "CosyVoiceDubbingNode", "pos": [ - 550, - 374 + 532, + 365 ], "size": { "0": 315, - "1": 146 + "1": 170 }, "flags": {}, - "order": 3, + "order": 4, "mode": 0, "inputs": [ { "name": "tts_srt", "type": "SRT", - "link": 7, - "slot_index": 0 + "link": 9 }, { "name": "prompt_wav", "type": "AUDIO", - "link": 3, - "slot_index": 1 + "link": 8 }, { "name": "prompt_srt", @@ -241,7 +240,7 @@ "name": "AUDIO", "type": "AUDIO", "links": [ - 6 + 10 ], "shape": 3, "slot_index": 0 @@ -252,6 +251,7 @@ }, "widgets_values": [ "<|zh|>", + true, 42, "randomize" ] @@ -266,14 +266,6 @@ 0, "AUDIO" ], - [ - 3, - 3, - 0, - 1, - 1, - "AUDIO" - ], [ 4, 3, @@ -291,20 +283,28 @@ "AUDIO" ], [ - 6, + 8, + 3, + 0, + 8, 1, - 0, - 6, - 0, "AUDIO" ], [ - 7, + 9, 7, 0, - 1, + 8, 0, "SRT" + ], + [ + 10, + 8, + 0, + 6, + 0, + "AUDIO" ] ], "groups": [], diff --git a/workflows/instruct_workflow.json b/workflows/instruct_workflow.json index d1ad21f..070936a 100644 --- a/workflows/instruct_workflow.json +++ b/workflows/instruct_workflow.json @@ -1,13 +1,13 @@ { - "last_node_id": 15, - "last_link_id": 12, + "last_node_id": 16, + "last_link_id": 15, "nodes": [ { - "id": 2, + "id": 15, "type": "TextNode", "pos": [ - 77, - 137 + 126, + 426 ], "size": { "0": 400, @@ -21,99 +21,7 @@ "name": "TEXT", "type": "TEXT", "links": [ - 5 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "TextNode" - }, - "widgets_values": [ - "在面对挑战时,他展现了非凡的勇气与智慧。'" - ] - }, - { - "id": 9, - "type": "CosyVoiceNode", - "pos": [ - 553, - 127 - ], - "size": { - "0": 315, - "1": 190 - }, - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [ - { - "name": "tts_text", - "type": "TEXT", - "link": 5 - }, - { - "name": "prompt_text", - "type": "TEXT", - "link": null, - "slot_index": 1 - }, - { - "name": "prompt_wav", - "type": "AUDIO", - "link": null, - "slot_index": 2 - }, - { - "name": "instruct_text", - "type": "TEXT", - "link": 12, - "slot_index": 3 - } - ], - "outputs": [ - { - "name": "AUDIO", - "type": "AUDIO", - "links": [ - 11 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "CosyVoiceNode" - }, - "widgets_values": [ - "自然语言控制", - "中文女", - 208, - "randomize" - ] - }, - { - "id": 15, - "type": "TextNode", - "pos": [ - 126, - 426 - ], - "size": { - "0": 400, - "1": 200 - }, - "flags": {}, - "order": 1, - "mode": 0, - "outputs": [ - { - "name": "TEXT", - "type": "TEXT", - "links": [ - 12 + 14 ], "shape": 3 } @@ -143,7 +51,7 @@ { "name": "audio", "type": "AUDIO", - "link": 11 + "link": 15 } ], "properties": { @@ -153,32 +61,123 @@ "audio/ComfyUI", null ] + }, + { + "id": 2, + "type": "TextNode", + "pos": [ + 77, + 137 + ], + "size": { + "0": 400, + "1": 200 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "TEXT", + "type": "TEXT", + "links": [ + 13 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TextNode" + }, + "widgets_values": [ + "在面对挑战时,他展现了非凡的勇气与智慧。'" + ] + }, + { + "id": 16, + "type": "CosyVoiceNode", + "pos": [ + 569, + 117 + ], + "size": { + "0": 315, + "1": 214 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [ + { + "name": "tts_text", + "type": "TEXT", + "link": 13 + }, + { + "name": "prompt_text", + "type": "TEXT", + "link": null + }, + { + "name": "prompt_wav", + "type": "AUDIO", + "link": null + }, + { + "name": "instruct_text", + "type": "TEXT", + "link": 14, + "slot_index": 3 + } + ], + "outputs": [ + { + "name": "AUDIO", + "type": "AUDIO", + "links": [ + 15 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CosyVoiceNode" + }, + "widgets_values": [ + 1, + "预训练音色", + "中文女", + 368, + "randomize" + ] } ], "links": [ [ - 5, + 13, 2, 0, - 9, + 16, 0, "TEXT" ], [ - 11, - 9, + 14, + 15, + 0, + 16, + 3, + "TEXT" + ], + [ + 15, + 16, 0, 14, 0, "AUDIO" - ], - [ - 12, - 15, - 0, - 9, - 3, - "TEXT" ] ], "groups": [],