added youtube downloader node

This commit is contained in:
SayanoAI
2024-06-04 13:46:25 -04:00
parent c7bd031d2d
commit 0e8211c8ba
11 changed files with 2540 additions and 795 deletions
+4 -2
View File
@@ -1,7 +1,7 @@
from .custom_nodes.stt import AudioTranscriptionNode
from .custom_nodes.uvr import UVR5Node
from .custom_nodes.rvc import RVCNode
from .custom_nodes.loaders import LoadAudio, LoadWhisperModelNode, LoadRVCModelNode, LoadHubertModel, LoadPitchExtractionParams
from .custom_nodes.loaders import DownloadAudio, LoadAudio, LoadWhisperModelNode, LoadRVCModelNode, LoadHubertModel, LoadPitchExtractionParams
from .custom_nodes.output import PreviewAudio
from .custom_nodes.utils import AudioBatchValueNode, MergeImageBatches, MergeLatentBatches, ImageRepeatInterleavedNode, LatentRepeatInterleavedNode, MergeAudioNode
@@ -25,13 +25,15 @@ NODE_CLASS_MAPPINGS = {
"MergeImageBatches": MergeImageBatches,
"MergeLatentBatches": MergeLatentBatches,
"ImageRepeatInterleavedNode": ImageRepeatInterleavedNode,
"LatentRepeatInterleavedNode": LatentRepeatInterleavedNode
"LatentRepeatInterleavedNode": LatentRepeatInterleavedNode,
"DownloadAudio": DownloadAudio
}
# A dictionary that contains the friendly/humanly readable titles for the nodes
NODE_DISPLAY_NAME_MAPPINGS = {
"UVR5Node": "🌺Vocal Removal",
"LoadAudio": "🌺Load Audio",
"DownloadAudio": "🌺Youtube Downloader",
"PreviewAudio": "🌺Preview Audio",
"AudioTranscriptionNode": "🌺Transcribe Audio",
"LoadWhisperModelNode": "🌺Load Whisper Model",
+60 -9
View File
@@ -1,7 +1,8 @@
from io import BytesIO
import os
import subprocess
import sys
from pytube import YouTube
from .settings import PITCH_EXTRACTION_OPTIONS
from ..lib import BASE_MODELS_DIR
from ..lib.model_utils import load_hubert
@@ -11,12 +12,13 @@ import torch
import folder_paths
from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline
from ..vc_infer_pipeline import get_vc
from .settings.downloader import RVC_DOWNLOAD_LINK, RVC_MODELS, download_file
from ..lib.audio import SUPPORTED_AUDIO, audio_to_bytes, load_input_audio
from .settings.downloader import RVC_DOWNLOAD_LINK, RVC_MODELS, download_file, slugify_filepath
from ..lib.audio import SUPPORTED_AUDIO, audio_to_bytes, bytes_to_audio, load_input_audio, save_input_audio
from ..config import config
input_path = folder_paths.get_input_directory()
temp_path = folder_paths.get_temp_directory()
CATEGORY = "🌺RVC-Studio/loaders"
model_ids = [
'openai/whisper-large-v3',
@@ -75,7 +77,7 @@ class LoadPitchExtractionParams:
RETURN_TYPES = ('PITCH_EXTRACTION', )
RETURN_NAMES = ('pitch_extraction_params', )
CATEGORY = "🌺RVC-Studio/loaders"
CATEGORY = CATEGORY
FUNCTION = 'load_params'
@@ -96,7 +98,7 @@ class LoadHubertModel:
RETURN_TYPES = ('HUBERT_MODEL', )
RETURN_NAMES = ('hubert_model', )
CATEGORY = "🌺RVC-Studio/loaders"
CATEGORY = CATEGORY
FUNCTION = 'load_model'
@@ -124,7 +126,7 @@ class LoadRVCModelNode:
RETURN_TYPES = ('RVC_MODEL', 'STRING')
RETURN_NAMES = ('model', 'model_name')
CATEGORY = "🌺RVC-Studio/loaders"
CATEGORY = CATEGORY
FUNCTION = 'load_model'
@@ -167,7 +169,7 @@ class LoadWhisperModelNode:
RETURN_TYPES = ('TRANSCRIPTION_MODEL', )
RETURN_NAMES = ('model', )
CATEGORY = "🌺RVC-Studio/loaders"
CATEGORY = CATEGORY
FUNCTION = 'load_model'
@@ -233,7 +235,7 @@ class LoadAudio:
"sr": (["None",16000,44100,48000],{"default": "None"}),
}}
CATEGORY = "🌺RVC-Studio/loaders"
CATEGORY = CATEGORY
RETURN_TYPES = ("STRING","VHS_AUDIO")
RETURN_NAMES = ("audio_name","vhs_audio")
@@ -251,4 +253,53 @@ class LoadAudio:
def IS_CHANGED(cls, audio):
audio_path = folder_paths.get_annotated_filepath(audio)
print(f"{audio_path=}")
return get_file_hash(audio_path)
return get_file_hash(audio_path)
class DownloadAudio:
@classmethod
def INPUT_TYPES(cls):
input_dir = input_path
files = get_filenames(root=input_dir,exts=SUPPORTED_AUDIO,format_func=os.path.basename)
return {
"required": {
"url": ("STRING", {"default": ""})
},
"optional": {
"sr": (["None",16000,44100,48000],{"default": "None"}),
"song_name": ("STRING",{"default": ""},)
}
}
CATEGORY = CATEGORY
RETURN_TYPES = ("STRING","VHS_AUDIO")
RETURN_NAMES = ("audio_name","vhs_audio")
FUNCTION = "download_audio"
def download_audio(self, url, sr="None", song_name=""):
assert "youtube" in url, "Please provide a valid youtube URL!"
widgetId = get_hash(url, sr)
sr = None if sr=="None" else int(sr)
audio_name = widgetId if song_name=="" else song_name
audio_path = os.path.join(input_path,f"{audio_name}.mp3")
if os.path.isfile(audio_path): input_audio = load_input_audio(audio_path,sr=sr)
else:
youtube_video = YouTube(url)
audio = youtube_video.streams.get_audio_only(subtype="mp4")
buffer = BytesIO()
audio.stream_to_buffer(buffer)
buffer.seek(0)
with open(audio_path,"wb") as f:
f.write(buffer.read())
buffer.close()
input_audio = load_input_audio(audio_path,sr=sr)
del buffer, audio
return {"ui": {"preview": [{"filename": os.path.basename(audio_path), "type": "input", "widgetId": widgetId}]}, "result": (audio_name, lambda:audio_to_bytes(*input_audio))}
@classmethod
def IS_CHANGED(cls, url):
return get_hash(url)
+6 -6
View File
@@ -1,7 +1,7 @@
import os
import shutil
from ..lib.audio import audio_to_bytes, bytes_to_audio, load_input_audio, save_input_audio
from ..lib.audio import SUPPORTED_AUDIO, audio_to_bytes, bytes_to_audio, load_input_audio, save_input_audio
from ..vc_infer_pipeline import vc_single
import folder_paths
@@ -42,9 +42,7 @@ class RVCNode:
}),
},
"optional": {
"format":(["wav", "flac", "mp3"],{
"default": "flac"
}),
"format":(SUPPORTED_AUDIO,{"default": "flac"}),
"use_cache": ("BOOLEAN",{"default": True})
}
}
@@ -66,7 +64,10 @@ class RVCNode:
else:
input_audio = bytes_to_audio(audio())
output_audio = vc_single(hubert_model=hubert_model(),input_audio=input_audio,f0_up_key=f0_up_key,**model(),**pitch_extraction_params)
print(save_input_audio(cache_name, output_audio))
if use_cache:
print(save_input_audio(cache_name, output_audio))
if os.path.isfile(cache_name): output_audio = load_input_audio(cache_name)
tempdir = os.path.join(temp_path,"preview")
os.makedirs(tempdir, exist_ok=True)
@@ -78,7 +79,6 @@ class RVCNode:
@classmethod
def IS_CHANGED(cls, *args, **kwargs):
print(f"{args=} {kwargs=}")
return get_hash(*args, *kwargs.items())
+5 -5
View File
@@ -207,7 +207,6 @@ class AudioBatchValueNode:
@classmethod
def IS_CHANGED(cls, *args, **kwargs):
print(f"{args=} {kwargs=}")
return get_hash(*args, *kwargs.items())
class ImageRepeatInterleavedNode:
@@ -347,8 +346,8 @@ class MergeAudioNode:
}),
"merge_type": (MERGE_OPTIONS,{"default": "median"}),
"normalize": ("BOOLEAN",{"default": True}),
"audio3_opt": ("VHS_AUDIO",{"default": None, "forceInput": True}),
"audio4_opt": ("VHS_AUDIO",{"default": None, "forceInput": True}),
"audio3_opt": ("VHS_AUDIO",{"default": None}),
"audio4_opt": ("VHS_AUDIO",{"default": None}),
}
}
@@ -361,7 +360,7 @@ class MergeAudioNode:
def merge(self, audio1, audio2, sr="None", merge_type="median", normalize=False, audio3_opt=None, audio4_opt=None):
audios = [audio() for audio in filter(None,[audio1, audio2, audio3_opt, audio4_opt])]
audios = [audio() for audio in [audio1, audio2, audio3_opt, audio4_opt] if audio is not None]
widgetId = get_hash(*audios,sr,merge_type,normalize)
audio_path = os.path.join(temp_path,"preview",f"{widgetId}.flac")
@@ -374,11 +373,12 @@ class MergeAudioNode:
merged_audio = merge_func(pad_audio(*[audio for (audio,_) in input_audios],axis=0),axis=0), merged_sr
print(save_input_audio(audio_path,merged_audio))
del input_audios
if os.path.isfile(audio_path): merged_audio = load_input_audio(audio_path)
del audios
audio_name = os.path.basename(audio_path)
return {"ui": {"preview": [{"filename": audio_name, "type": "temp", "subfolder": "preview", "widgetId": widgetId}]}, "result": (lambda: audio_to_bytes(*merged_audio),)}
@classmethod
def IS_CHANGED(cls, *args, **kwargs):
print(f"{args=} {kwargs=}")
return get_hash(*args, *kwargs.items())
+39 -35
View File
@@ -1,10 +1,9 @@
import os
import audio_separator.separator as uvr
from ..lib.audio import audio_to_bytes, bytes_to_audio, save_input_audio, load_input_audio
import folder_paths
from ..lib.utils import get_filenames, get_hash, get_file_hash, get_optimal_torch_device
from ..lib.utils import get_filenames, get_hash, get_optimal_torch_device
from ..lib import BASE_CACHE_DIR, BASE_MODELS_DIR, karafan
from ..uvr5_cli import Separator
from .settings.downloader import KARAFAN_MODELS, MDX_MODELS, RVC_DOWNLOAD_LINK, VR_MODELS, download_file
temp_path = folder_paths.get_temp_directory()
@@ -45,8 +44,8 @@ class UVR5Node:
}
}
RETURN_TYPES = ("VHS_AUDIO","VHS_AUDIO","VHS_AUDIO")
RETURN_NAMES = ("primary_stem","secondary_stem","audio_passthrough")
RETURN_TYPES = ("VHS_AUDIO","VHS_AUDIO")
RETURN_NAMES = ("primary_stem","secondary_stem")
FUNCTION = "split"
@@ -64,49 +63,54 @@ class UVR5Node:
input_audio = bytes_to_audio(audio())
hash_name = get_hash(audio(), model, agg, format)
audio_path = os.path.join(temp_path,"uvr",f"{hash_name}.wav")
vocal_path = os.path.join(cache_dir,hash_name,f"primary.{format}")
instrumental_path = os.path.join(cache_dir,hash_name,f"secondary.{format}")
if os.path.isfile(vocal_path) and os.path.isfile(instrumental_path) and use_cache:
vocals = load_input_audio(vocal_path)
instrumental = load_input_audio(instrumental_path)
primary_path = os.path.join(cache_dir,hash_name,f"primary.{format}")
secondary_path = os.path.join(cache_dir,hash_name,f"secondary.{format}")
primary=secondary=None
if os.path.isfile(primary_path) and os.path.isfile(secondary_path) and use_cache:
primary = load_input_audio(primary_path)
secondary = load_input_audio(secondary_path)
else:
if not os.path.isfile(audio_path):
os.makedirs(os.path.dirname(audio_path),exist_ok=True)
print(save_input_audio(audio_path,input_audio))
vocals=instrumental=None
try: # try original RVC implementation
model = Separator(
model_path=model_path,
device=device,
is_half="cuda" in str(device),
cache_dir=cache_dir,
agg=agg
)
vocals, instrumental, input_audio = model.run_inference(audio_path,format=format)
except Exception as e:
print(f"Error: {e}")
try:
if "karafan" in model_path: # try karafan implementation
vocals, instrumental, input_audio = karafan.inference.Process(audio_path,cache_dir=temp_path,format=format)
primary, secondary, _ = karafan.inference.Process(audio_path,cache_dir=temp_path,format=format)
else: # try python-audio-separator implementation
import audio_separator.separator as uvr
model_dir = os.path.dirname(model_path)
model_name = os.path.basename(model_path)
vr_params={"batch_size": 4, "window_size": 512, "aggression": agg, "enable_tta": False, "enable_post_process": False, "post_process_threshold": 0.2, "high_end_process": "mirroring"}
model = uvr.Separator(model_file_dir=os.path.join(BASE_MODELS_DIR,model_dir),output_dir=temp_path,output_format=format,vr_params=vr_params)
mdx_params={"hop_length": 1024, "segment_size": 256, "overlap": 0.25, "batch_size": 4}
model = uvr.Separator(model_file_dir=os.path.join(BASE_MODELS_DIR,model_dir),output_dir=temp_path,output_format=format,vr_params=vr_params,mdx_params=mdx_params)
model.load_model(model_name)
output_files = model.separate(audio_path)
print(f"{output_files=}")
vocals = load_input_audio(os.path.join(temp_path,output_files[0]))
instrumental = load_input_audio(os.path.join(temp_path,output_files[1]))
primary = load_input_audio(os.path.join(temp_path,output_files[0]))
secondary = load_input_audio(os.path.join(temp_path,output_files[1]))
except Exception as e: # try RVC implementation
print(f"Error: {e}")
from ..uvr5_cli import Separator
model = Separator(
model_path=model_path,
device=device,
is_half="cuda" in str(device),
cache_dir=cache_dir,
agg=agg
)
primary, secondary, _ = model.run_inference(audio_path,format=format)
finally:
if vocals is not None and instrumental is not None and use_cache:
print(save_input_audio(vocal_path,vocals))
print(save_input_audio(instrumental_path,instrumental))
if primary is not None and secondary is not None and use_cache:
print(save_input_audio(primary_path,primary))
print(save_input_audio(secondary_path,secondary))
if os.path.isfile(primary_path) and os.path.isfile(secondary_path) and use_cache:
primary = load_input_audio(primary_path)
secondary = load_input_audio(secondary_path)
return (lambda:audio_to_bytes(*vocals), lambda:audio_to_bytes(*instrumental), audio)
return (lambda:audio_to_bytes(*primary), lambda:audio_to_bytes(*secondary))
@classmethod
def IS_CHANGED(cls, *args, **kwargs):
print(f"{args=} {kwargs=}")
return get_hash(args,kwargs)
return get_hash(*args,*kwargs.items())
+918
View File
@@ -0,0 +1,918 @@
{
"last_node_id": 43,
"last_link_id": 77,
"nodes": [
{
"id": 33,
"type": "MergeAudioNode",
"pos": [
1154,
-367
],
"size": [
315,
166
],
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "audio1",
"type": "VHS_AUDIO",
"link": 54
},
{
"name": "audio2",
"type": "VHS_AUDIO",
"link": 60
},
{
"name": "audio3_opt",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "audio4_opt",
"type": "VHS_AUDIO",
"link": null
}
],
"outputs": [
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
63
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MergeAudioNode"
},
"widgets_values": [
"None",
"mean",
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 39,
"type": "PreviewAudio",
"pos": [
1514,
-358
],
"size": [
315,
150
],
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 63
},
{
"name": "filename",
"type": "STRING",
"link": 64,
"widget": {
"name": "filename"
},
"slot_index": 1
}
],
"outputs": [
{
"name": "output_path",
"type": "STRING",
"links": null,
"shape": 3
},
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "PreviewAudio"
},
"widgets_values": [
"test",
"flac",
2,
true,
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 40,
"type": "UVR5Node",
"pos": [
-172,
-567
],
"size": {
"0": 315,
"1": 150
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 65
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": null,
"shape": 3
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": [
68,
69
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "UVR5Node"
},
"widgets_values": [
"UVR/UVR-DeEcho-DeReverb.pth",
true,
10,
"flac"
]
},
{
"id": 41,
"type": "UVR5Node",
"pos": [
269,
-736
],
"size": {
"0": 315,
"1": 150
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 68
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": [
70
],
"shape": 3,
"slot_index": 0
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": [
72
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "UVR5Node"
},
"widgets_values": [
"UVR/HP5-vocals+instrumentals.pth",
true,
10,
"flac"
]
},
{
"id": 42,
"type": "UVR5Node",
"pos": [
613,
-733
],
"size": {
"0": 315,
"1": 150
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 69
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": [
71
],
"shape": 3,
"slot_index": 0
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": [
73
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "UVR5Node"
},
"widgets_values": [
"karafan/MDX23C-8KFFT-InstVoc_HQ.ckpt",
true,
10,
"flac"
]
},
{
"id": 35,
"type": "MergeAudioNode",
"pos": [
271,
-545
],
"size": [
315,
166
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "audio1",
"type": "VHS_AUDIO",
"link": 70
},
{
"name": "audio2",
"type": "VHS_AUDIO",
"link": 71
},
{
"name": "audio3_opt",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "audio4_opt",
"type": "VHS_AUDIO",
"link": null
}
],
"outputs": [
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
74
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MergeAudioNode"
},
"widgets_values": [
"None",
"min",
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 36,
"type": "MergeAudioNode",
"pos": [
608,
-544
],
"size": [
315,
166
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "audio1",
"type": "VHS_AUDIO",
"link": 72
},
{
"name": "audio2",
"type": "VHS_AUDIO",
"link": 73
},
{
"name": "audio3_opt",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "audio4_opt",
"type": "VHS_AUDIO",
"link": null
}
],
"outputs": [
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
54
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MergeAudioNode"
},
"widgets_values": [
"None",
"mean",
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 6,
"type": "LoadPitchExtractionParams",
"pos": [
272,
101
],
"size": {
"0": 315,
"1": 178
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"links": [
59
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadPitchExtractionParams"
},
"widgets_values": [
"rmvpe+",
false,
0.75,
0,
0.25,
0.25
]
},
{
"id": 38,
"type": "LoadAudio",
"pos": [
-171,
-739
],
"size": {
"0": 309.8900146484375,
"1": 126
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "audio_name",
"type": "STRING",
"links": [
77
],
"shape": 3,
"slot_index": 0
},
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
65
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"grimlight-ost-1st-trailer---wish-upon-a-star.mp3",
"None",
"image"
]
},
{
"id": 31,
"type": "JoinStrings",
"pos": [
309,
-145
],
"size": {
"0": 315,
"1": 106
},
"flags": {
"collapsed": true
},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "string1",
"type": "STRING",
"link": 76,
"widget": {
"name": "string1"
}
},
{
"name": "string2",
"type": "STRING",
"link": 77,
"widget": {
"name": "string2"
}
}
],
"outputs": [
{
"name": "STRING",
"type": "STRING",
"links": [
64
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "JoinStrings"
},
"widgets_values": [
"",
"",
"\\"
]
},
{
"id": 4,
"type": "LoadRVCModelNode",
"pos": [
637,
-157
],
"size": {
"0": 315,
"1": 78
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "model",
"type": "RVC_MODEL",
"links": [
57
],
"shape": 3
},
{
"name": "model_name",
"type": "STRING",
"links": [
76
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "LoadRVCModelNode"
},
"widgets_values": [
"RVC/Sayano.pth"
]
},
{
"id": 43,
"type": "UVR5Node",
"pos": [
278,
-99
],
"size": {
"0": 315,
"1": 150
},
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 74
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": [
75
],
"shape": 3,
"slot_index": 0
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "UVR5Node"
},
"widgets_values": [
"UVR/HP5-vocals+instrumentals.pth",
true,
10,
"flac"
]
},
{
"id": 37,
"type": "RVCNode",
"pos": [
639,
61
],
"size": [
309.34271240234375,
166
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 75
},
{
"name": "model",
"type": "RVC_MODEL",
"link": 57
},
{
"name": "hubert_model",
"type": "HUBERT_MODEL",
"link": 58
},
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"link": 59
}
],
"outputs": [
{
"name": "VHS_AUDIO",
"type": "VHS_AUDIO",
"links": [
60
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "RVCNode"
},
"widgets_values": [
0,
"flac",
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 5,
"type": "LoadHubertModel",
"pos": [
642,
-34
],
"size": {
"0": 315,
"1": 58
},
"flags": {
"collapsed": false
},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "hubert_model",
"type": "HUBERT_MODEL",
"links": [
58
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadHubertModel"
},
"widgets_values": [
"hubert_base.pt"
]
}
],
"links": [
[
54,
36,
0,
33,
0,
"VHS_AUDIO"
],
[
57,
4,
0,
37,
1,
"RVC_MODEL"
],
[
58,
5,
0,
37,
2,
"HUBERT_MODEL"
],
[
59,
6,
0,
37,
3,
"PITCH_EXTRACTION"
],
[
60,
37,
0,
33,
1,
"VHS_AUDIO"
],
[
63,
33,
0,
39,
0,
"VHS_AUDIO"
],
[
64,
31,
0,
39,
1,
"STRING"
],
[
65,
38,
1,
40,
0,
"VHS_AUDIO"
],
[
68,
40,
1,
41,
0,
"VHS_AUDIO"
],
[
69,
40,
1,
42,
0,
"VHS_AUDIO"
],
[
70,
41,
0,
35,
0,
"VHS_AUDIO"
],
[
71,
42,
0,
35,
1,
"VHS_AUDIO"
],
[
72,
41,
1,
36,
0,
"VHS_AUDIO"
],
[
73,
42,
1,
36,
1,
"VHS_AUDIO"
],
[
74,
35,
0,
43,
0,
"VHS_AUDIO"
],
[
75,
43,
0,
37,
0,
"VHS_AUDIO"
],
[
76,
4,
1,
31,
0,
"STRING"
],
[
77,
38,
0,
31,
1,
"STRING"
]
],
"groups": [
{
"title": "Preprocess",
"bounding": [
-182,
-813,
335,
406
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Ensemble Vocal Separation",
"bounding": [
259,
-810,
677,
472
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Postprocess",
"bounding": [
262,
-236,
712,
529
],
"color": "#3f789e",
"font_size": 24
},
{
"title": "Output Audio",
"bounding": [
1144,
-441,
698,
255
],
"color": "#3f789e",
"font_size": 24
}
],
"config": {},
"extra": {
"ds": {
"scale": 0.6830134553650705,
"offset": [
417.8530750478458,
895.7208507202507
]
}
},
"version": 0.4
}
File diff suppressed because it is too large Load Diff
+239 -212
View File
@@ -1,44 +1,49 @@
{
"last_node_id": 8,
"last_link_id": 11,
"last_node_id": 9,
"last_link_id": 15,
"nodes": [
{
"id": 4,
"type": "LoadRVCModelNode",
"id": 6,
"type": "LoadPitchExtractionParams",
"pos": [
237,
-56
252,
-72
],
"size": {
"0": 315,
"1": 58
"1": 178
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "model",
"type": "RVC_MODEL",
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"links": [
3
5
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadRVCModelNode"
"Node name for S&R": "LoadPitchExtractionParams"
},
"widgets_values": [
"RVC/Sayano.pth"
"rmvpe",
false,
0.75,
0,
0.25,
0.25
]
},
{
"id": 5,
"type": "LoadHubertModel",
"pos": [
599,
-54
627,
12
],
"size": {
"0": 315,
@@ -65,149 +70,48 @@
]
},
{
"id": 6,
"type": "LoadPitchExtractionParams",
"id": 4,
"type": "LoadRVCModelNode",
"pos": [
953,
-50
631,
-114
],
"size": {
"0": 315,
"1": 178
"1": 78
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"name": "model",
"type": "RVC_MODEL",
"links": [
5
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadPitchExtractionParams"
},
"widgets_values": [
"rmvpe",
false,
0.75,
0,
0.25,
0.25
]
},
{
"id": 7,
"type": "UVR5Node",
"pos": [
581,
71
],
"size": {
"0": 315,
"1": 170
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 6
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": [
7
3
],
"shape": 3
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": [
8
],
"shape": 3,
"slot_index": 1
},
{
"name": "audio_passthrough",
"type": "VHS_AUDIO",
"name": "model_name",
"type": "STRING",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "UVR5Node"
"Node name for S&R": "LoadRVCModelNode"
},
"widgets_values": [
"UVR/HP5-vocals+instrumentals.pth",
10,
"flac",
true
]
},
{
"id": 8,
"type": "MergeAudioNode",
"pos": [
196,
311
],
"size": {
"0": 315,
"1": 126
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "audio1",
"type": "VHS_AUDIO",
"link": 8
},
{
"name": "audio2",
"type": "VHS_AUDIO",
"link": 9
}
],
"outputs": [
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
10
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MergeAudioNode"
},
"widgets_values": [
"None",
"mean",
true
"RVC/Sayano.pth"
]
},
{
"id": 3,
"type": "PreviewAudio",
"pos": [
584,
294
1347,
132
],
"size": [
315,
@@ -252,48 +156,8 @@
"test",
"flac",
1,
true
]
},
{
"id": 2,
"type": "LoadAudio",
"pos": [
223,
52
],
"size": [
315,
102
],
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "audio_name",
"type": "STRING",
"links": [
11
],
"shape": 3,
"slot_index": 0
},
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
6
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"already-enough.mp3",
"None",
true,
true,
{
"hidden": false,
"paused": false,
@@ -305,12 +169,12 @@
"id": 1,
"type": "RVCNode",
"pos": [
947,
193
621,
122
],
"size": [
313.7279268096661,
181.78793221310661
313.7279357910156,
166
],
"flags": {},
"order": 5,
@@ -319,7 +183,7 @@
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 7,
"link": 14,
"slot_index": 0
},
{
@@ -346,7 +210,7 @@
"name": "VHS_AUDIO",
"type": "VHS_AUDIO",
"links": [
9
15
],
"shape": 3,
"slot_index": 0
@@ -358,7 +222,170 @@
"widgets_values": [
0,
"flac",
true
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 9,
"type": "UVR5Node",
"pos": [
243,
157
],
"size": {
"0": 315,
"1": 150
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "audio",
"type": "VHS_AUDIO",
"link": 12
}
],
"outputs": [
{
"name": "primary_stem",
"type": "VHS_AUDIO",
"links": [
14
],
"shape": 3,
"slot_index": 0
},
{
"name": "secondary_stem",
"type": "VHS_AUDIO",
"links": [
13
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "UVR5Node"
},
"widgets_values": [
"UVR/HP5-vocals+instrumentals.pth",
true,
10,
"flac"
]
},
{
"id": 8,
"type": "MergeAudioNode",
"pos": [
981,
128
],
"size": [
315,
166
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "audio1",
"type": "VHS_AUDIO",
"link": 15
},
{
"name": "audio2",
"type": "VHS_AUDIO",
"link": 13
},
{
"name": "audio3_opt",
"type": "VHS_AUDIO",
"link": null
},
{
"name": "audio4_opt",
"type": "VHS_AUDIO",
"link": null
}
],
"outputs": [
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
10
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MergeAudioNode"
},
"widgets_values": [
"None",
"mean",
true,
{
"hidden": false,
"paused": false,
"params": {}
}
]
},
{
"id": 2,
"type": "LoadAudio",
"pos": [
-106,
162
],
"size": {
"0": 315,
"1": 126
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "audio_name",
"type": "STRING",
"links": [
11
],
"shape": 3,
"slot_index": 0
},
{
"name": "vhs_audio",
"type": "VHS_AUDIO",
"links": [
12
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadAudio"
},
"widgets_values": [
"grimlight-ost-1st-trailer---wish-upon-a-star.mp3",
"None",
{
"hidden": false,
"paused": false,
"params": {}
}
]
}
],
@@ -387,38 +414,6 @@
3,
"PITCH_EXTRACTION"
],
[
6,
2,
1,
7,
0,
"VHS_AUDIO"
],
[
7,
7,
0,
1,
0,
"VHS_AUDIO"
],
[
8,
7,
1,
8,
0,
"VHS_AUDIO"
],
[
9,
1,
0,
8,
1,
"VHS_AUDIO"
],
[
10,
8,
@@ -434,6 +429,38 @@
3,
1,
"STRING"
],
[
12,
2,
1,
9,
0,
"VHS_AUDIO"
],
[
13,
9,
1,
8,
1,
"VHS_AUDIO"
],
[
14,
9,
0,
1,
0,
"VHS_AUDIO"
],
[
15,
1,
0,
8,
0,
"VHS_AUDIO"
]
],
"groups": [],
@@ -442,8 +469,8 @@
"ds": {
"scale": 0.6830134553650706,
"offset": [
-172.5821681249739,
268.8813118680092
531.5168724939232,
463.2073890439136
]
}
},
+2 -2
View File
@@ -111,7 +111,7 @@ def audio_to_bytes(audio,sr,target_sr=None,to_int16=False,to_stereo=False,format
bytes_io.seek(0)
return bytes_io.read()
def bytes_to_audio(data: Union[io.BytesIO,bytes],**kwargs):
def bytes_to_audio(data: bytes,**kwargs):
with io.BytesIO(data) as bytes_io:
audio, sr = sf.read(bytes_io,**kwargs)
if audio.ndim>1:
@@ -167,7 +167,7 @@ def audio2bytes(audio: np.array, sr: int):
def pad_audio(*audios,axis=0):
maxlen = max(len(a) if a is not None else 0 for a in audios)
if maxlen>0:
stack = librosa.util.stack([librosa.util.pad_center(data=a,size=maxlen) for a in audios if a is not None],axis=axis)
stack = librosa.util.stack([librosa.util.fix_length(a,size=maxlen) for a in audios if a is not None],axis=axis)
return stack
else: return np.stack(audios,axis=axis)
+2 -1
View File
@@ -12,4 +12,5 @@ samplerate
pyaudio
spacy
monotonic_align
textacy
textacy
pytube
+3
View File
@@ -346,6 +346,9 @@ app.registerExtension({
addUploadWidget(nodeType, nodeData, "audio", "audio")
addPreviewWidget(nodeType, nodeData, "audio", "onNodeCreated" )
break
case "DownloadAudio":
addPreviewWidget(nodeType, nodeData, "audio", "onExecuted" )
break
case "PreviewAudio":
addPreviewWidget(nodeType, nodeData, "audio", "onExecuted" )
break