diff --git a/custom_nodes/audio_nodes.py b/custom_nodes/audio_nodes.py index 9bf347d..3b413f9 100644 --- a/custom_nodes/audio_nodes.py +++ b/custom_nodes/audio_nodes.py @@ -203,7 +203,7 @@ class PreviewAudio: output_path = increment_filename_no_overwrite(output_path) input_audio = get_audio(audio) - print(save_input_audio(output_path,input_audio,to_int16=True,to_stereo=save_channels==2)) + print(save_input_audio(output_path,input_audio,to_int16=save_format!="mp3",to_stereo=save_channels==2)) tempdir = os.path.join(temp_path,"preview") os.makedirs(tempdir, exist_ok=True) diff --git a/custom_nodes/rvc_nodes.py b/custom_nodes/rvc_nodes.py index 9264b8c..bbdf346 100644 --- a/custom_nodes/rvc_nodes.py +++ b/custom_nodes/rvc_nodes.py @@ -235,8 +235,7 @@ class RVCProcessDatasetNode: "n_threads": ("INT", {"default": get_optimal_threads(), "min": 1, "max": multiprocessing.cpu_count()}), "period": ("FLOAT", {"default": 3., "min": 1., "max": 10., "step": .1}), "overlap": ("FLOAT",{"default": .3, "min": .1, "max": 1., "step": .1}), - "max_volume": ("FLOAT",{"default": 1., "min": .1, "max": 1., "step": .05}), - "alpha": ("FLOAT",{"default": .75, "min": .05, "max": 1., "step": .05}), + "max_volume": ("FLOAT",{"default": .99, "min": .1, "max": 1., "step": .01}), "mute_ratio": ("FLOAT",{"default": .0, "min": .0, "max": .5, "step": .01}), } } @@ -248,13 +247,13 @@ class RVCProcessDatasetNode: CATEGORY = CATEGORY - def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., alpha=.75, mute_ratio=.0): + def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., mute_ratio=.0): assert model_name, "Please provide a model name!" assert dataset, "Please upload a dataset!" f0_method = pitch_extraction_params.get("f0_method", "") - cached_params = [model_name, dataset, period, overlap, max_volume, alpha, mute_ratio, sr, f0_method] + cached_params = [model_name, dataset, period, overlap, max_volume, mute_ratio, sr, f0_method] crepe_hop_length = pitch_extraction_params.get("crepe_hop_length",160) if "crepe" in f0_method: cached_params.append(crepe_hop_length) @@ -272,7 +271,7 @@ class RVCProcessDatasetNode: files = extract_zip_without_structure(os.path.join(dataset_path,dataset),dataset_dir) assert len(files), "Failed to extract zip file..." - print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume,alpha)) + print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume)) print(extract_features_trainset(hubert_model(), model_log_dir,n_p=n_threads,f0method=f0_method,device=device,if_f0=bool(f0_method),version="v2",crepe_hop_length=crepe_hop_length)) diff --git a/examples/rvc-model-trainer.json b/examples/rvc-model-trainer.json index bb54b71..51a0988 100644 --- a/examples/rvc-model-trainer.json +++ b/examples/rvc-model-trainer.json @@ -1,27 +1,27 @@ { - "last_node_id": 15, - "last_link_id": 25, + "last_node_id": 17, + "last_link_id": 30, "nodes": [ { "id": 3, "type": "LoadPitchExtractionParams", "pos": [ - -397, - 179 + -387, + 130 ], "size": { "0": 315, - "1": 178 + "1": 202 }, "flags": {}, - "order": 2, + "order": 0, "mode": 0, "outputs": [ { "name": "pitch_extraction_params", "type": "PITCH_EXTRACTION", "links": [ - 17 + 29 ], "slot_index": 0, "shape": 3 @@ -36,7 +36,8 @@ 0.75, 0, 0.25, - 0.25 + 0.25, + 160 ] }, { @@ -51,14 +52,14 @@ "1": 58 }, "flags": {}, - "order": 0, + "order": 1, "mode": 0, "outputs": [ { "name": "hubert_model", "type": "HUBERT_MODEL", "links": [ - 16 + 28 ], "slot_index": 0, "shape": 3 @@ -139,7 +140,7 @@ "1": 146 }, "flags": {}, - "order": 1, + "order": 2, "mode": 0, "outputs": [ { @@ -185,7 +186,7 @@ "1": 150 }, "flags": {}, - "order": 3, + "order": 4, "mode": 0, "inputs": [ { @@ -285,65 +286,12 @@ true ] }, - { - "id": 13, - "type": "RVCProcessDatasetNode", - "pos": [ - -8, - 176 - ], - "size": { - "0": 337.6000061035156, - "1": 294 - }, - "flags": {}, - "order": 4, - "mode": 0, - "inputs": [ - { - "name": "hubert_model", - "type": "HUBERT_MODEL", - "link": 16 - }, - { - "name": "pitch_extraction_params", - "type": "PITCH_EXTRACTION", - "link": 17 - } - ], - "outputs": [ - { - "name": "rvc_dataset_pipe", - "type": "RVC_DATASET_PIPE", - "links": [ - 22 - ], - "slot_index": 0, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "RVCProcessDatasetNode" - }, - "widgets_values": [ - "Comfy-Sayano", - "Sayano.zip", - "40k", - 1, - 3, - 0.3, - 0.9500000000000001, - 0.75, - 0.01, - "image" - ] - }, { "id": 15, "type": "RVCTrainModelNode", "pos": [ 374, - 119 + 118 ], "size": { "0": 506.4000244140625, @@ -356,7 +304,7 @@ { "name": "rvc_dataset_pipe", "type": "RVC_DATASET_PIPE", - "link": 22 + "link": 30 } ], "outputs": [ @@ -399,18 +347,70 @@ }, "widgets_values": [ "0", - 4, - 300, - 0, - "pretrained_v2/f0Ov2Super40kD.pth", - "pretrained_v2/f0Ov2Super40kG.pth", - true, + 5, + 100, + 10, + "pretrained_v2/G_DMR-V1-32k-f0.pth", + "pretrained_v2/D_DMR-V1-32k-f0.pth", true, false, + false, true, true, true ] + }, + { + "id": 17, + "type": "RVCProcessDatasetNode", + "pos": [ + -8, + 176 + ], + "size": { + "0": 337.6000061035156, + "1": 270 + }, + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "hubert_model", + "type": "HUBERT_MODEL", + "link": 28 + }, + { + "name": "pitch_extraction_params", + "type": "PITCH_EXTRACTION", + "link": 29 + } + ], + "outputs": [ + { + "name": "rvc_dataset_pipe", + "type": "RVC_DATASET_PIPE", + "links": [ + 30 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "RVCProcessDatasetNode" + }, + "widgets_values": [ + "Comfy-Mae", + "Mae.zip", + "32k", + 12, + 3, + 0.5, + 0.9500000000000001, + 0.01, + "image" + ] } ], "links": [ @@ -446,30 +446,6 @@ 1, "AUDIO,VHS_AUDIO" ], - [ - 16, - 5, - 0, - 13, - 0, - "HUBERT_MODEL" - ], - [ - 17, - 3, - 0, - 13, - 1, - "PITCH_EXTRACTION" - ], - [ - 22, - 13, - 0, - 15, - 0, - "RVC_DATASET_PIPE" - ], [ 23, 15, @@ -493,6 +469,30 @@ 6, 3, "PITCH_EXTRACTION" + ], + [ + 28, + 5, + 0, + 17, + 0, + "HUBERT_MODEL" + ], + [ + 29, + 3, + 0, + 17, + 1, + "PITCH_EXTRACTION" + ], + [ + 30, + 17, + 0, + 15, + 0, + "RVC_DATASET_PIPE" ] ], "groups": [ @@ -522,10 +522,10 @@ "config": {}, "extra": { "ds": { - "scale": 1.1, + "scale": 0.6830134553650705, "offset": [ - 652.4400126183286, - 108.78821868462816 + 544.5635228085466, + 563.3345057885277 ] } }, diff --git a/lib/audio.py b/lib/audio.py index ddbaf1e..153dc01 100644 --- a/lib/audio.py +++ b/lib/audio.py @@ -44,7 +44,7 @@ def load_audio(file, sr, **kwargs): return remix_audio((np.frombuffer(out, np.float32).flatten(), sr),**kwargs) -def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,**kwargs): +def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,max_volume=.99,**kwargs): audio = np.array(input_audio[0],dtype="float32") if target_sr is None: target_sr=input_audio[1] @@ -57,10 +57,10 @@ def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=Fa audio=merge_func(audio,axis=axis) if norm: audio = librosa.util.normalize(audio,axis=axis) - audio_max = np.abs(audio).max()/.99 + audio_max = np.abs(audio).max()/max_volume if audio_max > 1: audio = audio / audio_max - if to_int16: audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1).astype("int16") + if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1).astype("int16") print(f"after remix: shape={audio.shape}, max={audio.max()}, min={audio.min()}, mean={audio.mean()}, sr={target_sr}") return audio, target_sr @@ -71,16 +71,14 @@ def load_input_audio(fname,sr=None,**kwargs): print(f"loading sound {fname=} {audio.ndim=} {audio.max()=} {audio.min()=} {audio.dtype=} {sr=}") return audio, sr -def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False): +def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False,max_volume=.99): print(f"saving sound to {fname}") os.makedirs(os.path.dirname(fname),exist_ok=True) audio=np.array(input_audio[0],dtype="float32") - if to_int16: - audio_max = np.abs(audio).max()/.99 - if audio_max > 1: audio = audio / audio_max - audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1) - + audio_max = np.abs(audio).max()/max_volume + if audio_max > 1: audio = audio / audio_max + if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1) if to_stereo and audio.ndim<2: audio=np.stack([audio,audio],axis=-1) try: diff --git a/preprocessing_utils.py b/preprocessing_utils.py index f33710f..0a35048 100644 --- a/preprocessing_utils.py +++ b/preprocessing_utils.py @@ -1,19 +1,17 @@ import sys, os, multiprocessing from threading import Thread import numpy as np, os, traceback -from .lib.model_utils import load_hubert from .lib.slicer2 import Slicer -import librosa, traceback +import traceback from scipy.io import wavfile -from .lib.audio import load_audio from .pitch_extraction import FeatureExtractor -from .lib.audio import load_input_audio +from .lib.audio import load_input_audio, remix_audio from .lib.utils import gc_collect from .config import config import torch class Preprocess: - def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, alpha=.75, max_volume=.95): + def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, max_volume=.95): self.slicer = Slicer( sr=sr, threshold=-42, @@ -26,11 +24,10 @@ class Preprocess: self.per = period self.overlap = overlap self.tail = self.per + self.overlap - self.max = max_volume - self.alpha = alpha + self.max_volume = max_volume self.exp_dir = exp_dir - self.gt_wavs_dir = "%s/0_gt_wavs" % exp_dir - self.wavs16k_dir = "%s/1_16k_wavs" % exp_dir + self.gt_wavs_dir = os.path.join(exp_dir,"0_gt_wavs") + self.wavs16k_dir = os.path.join(exp_dir,"1_16k_wavs") self.noparallel = noparallel os.makedirs(self.exp_dir, exist_ok=True) os.makedirs(self.gt_wavs_dir, exist_ok=True) @@ -45,26 +42,9 @@ class Preprocess: # mutex.release() def norm_write(self, tmp_audio, idx0, idx1): - tmp_max = np.abs(tmp_audio).max() - if tmp_max > 2.5: - print("%s-%s-%s-filtered" % (idx0, idx1, tmp_max)) - return - tmp_audio = (tmp_audio / tmp_max * (self.max * self.alpha)) + ( - 1 - self.alpha - ) * tmp_audio - wavfile.write( - "%s/%s_%s.wav" % (self.gt_wavs_dir, idx0, idx1), - self.sr, - tmp_audio.astype(np.float32), - ) - tmp_audio = librosa.resample( - tmp_audio, orig_sr=self.sr, target_sr=16000 - ) # , res_type="soxr_vhq" - wavfile.write( - "%s/%s_%s.wav" % (self.wavs16k_dir, idx0, idx1), - 16000, - tmp_audio.astype(np.float32), - ) + wavfile.write(os.path.join(self.gt_wavs_dir, f"{idx0}_{idx1}.wav"),self.sr,tmp_audio.astype(np.float32)) + remixed_audio = remix_audio((tmp_audio, self.sr), target_sr=16000, norm=True,max_volume=self.max_volume) + wavfile.write(os.path.join(self.wavs16k_dir, f"{idx0}_{idx1}.wav"),16000,remixed_audio[0].astype(np.float32)) def pipeline(self, path, idx0): try: @@ -208,9 +188,9 @@ class FeatureInput(FeatureExtractor): except: self.printt("f0fail-%s-%s-%s" % (idx, inp_path, traceback.format_exc())) -def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1., alpha=.75): +def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1.): try: - pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume, alpha=alpha) + pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume) pp.println("start preprocess") pp.println(sys.argv) pp.pipeline_mp_inp_dir(inp_root, n_p)