better preprocessing norm
This commit is contained in:
@@ -203,7 +203,7 @@ class PreviewAudio:
|
||||
output_path = increment_filename_no_overwrite(output_path)
|
||||
|
||||
input_audio = get_audio(audio)
|
||||
print(save_input_audio(output_path,input_audio,to_int16=True,to_stereo=save_channels==2))
|
||||
print(save_input_audio(output_path,input_audio,to_int16=save_format!="mp3",to_stereo=save_channels==2))
|
||||
|
||||
tempdir = os.path.join(temp_path,"preview")
|
||||
os.makedirs(tempdir, exist_ok=True)
|
||||
|
||||
@@ -235,8 +235,7 @@ class RVCProcessDatasetNode:
|
||||
"n_threads": ("INT", {"default": get_optimal_threads(), "min": 1, "max": multiprocessing.cpu_count()}),
|
||||
"period": ("FLOAT", {"default": 3., "min": 1., "max": 10., "step": .1}),
|
||||
"overlap": ("FLOAT",{"default": .3, "min": .1, "max": 1., "step": .1}),
|
||||
"max_volume": ("FLOAT",{"default": 1., "min": .1, "max": 1., "step": .05}),
|
||||
"alpha": ("FLOAT",{"default": .75, "min": .05, "max": 1., "step": .05}),
|
||||
"max_volume": ("FLOAT",{"default": .99, "min": .1, "max": 1., "step": .01}),
|
||||
"mute_ratio": ("FLOAT",{"default": .0, "min": .0, "max": .5, "step": .01}),
|
||||
}
|
||||
}
|
||||
@@ -248,13 +247,13 @@ class RVCProcessDatasetNode:
|
||||
|
||||
CATEGORY = CATEGORY
|
||||
|
||||
def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., alpha=.75, mute_ratio=.0):
|
||||
def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., mute_ratio=.0):
|
||||
|
||||
assert model_name, "Please provide a model name!"
|
||||
assert dataset, "Please upload a dataset!"
|
||||
|
||||
f0_method = pitch_extraction_params.get("f0_method", "")
|
||||
cached_params = [model_name, dataset, period, overlap, max_volume, alpha, mute_ratio, sr, f0_method]
|
||||
cached_params = [model_name, dataset, period, overlap, max_volume, mute_ratio, sr, f0_method]
|
||||
crepe_hop_length = pitch_extraction_params.get("crepe_hop_length",160)
|
||||
|
||||
if "crepe" in f0_method: cached_params.append(crepe_hop_length)
|
||||
@@ -272,7 +271,7 @@ class RVCProcessDatasetNode:
|
||||
files = extract_zip_without_structure(os.path.join(dataset_path,dataset),dataset_dir)
|
||||
assert len(files), "Failed to extract zip file..."
|
||||
|
||||
print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume,alpha))
|
||||
print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume))
|
||||
|
||||
print(extract_features_trainset(hubert_model(), model_log_dir,n_p=n_threads,f0method=f0_method,device=device,if_f0=bool(f0_method),version="v2",crepe_hop_length=crepe_hop_length))
|
||||
|
||||
|
||||
+100
-100
@@ -1,27 +1,27 @@
|
||||
{
|
||||
"last_node_id": 15,
|
||||
"last_link_id": 25,
|
||||
"last_node_id": 17,
|
||||
"last_link_id": 30,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 3,
|
||||
"type": "LoadPitchExtractionParams",
|
||||
"pos": [
|
||||
-397,
|
||||
179
|
||||
-387,
|
||||
130
|
||||
],
|
||||
"size": {
|
||||
"0": 315,
|
||||
"1": 178
|
||||
"1": 202
|
||||
},
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"outputs": [
|
||||
{
|
||||
"name": "pitch_extraction_params",
|
||||
"type": "PITCH_EXTRACTION",
|
||||
"links": [
|
||||
17
|
||||
29
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
@@ -36,7 +36,8 @@
|
||||
0.75,
|
||||
0,
|
||||
0.25,
|
||||
0.25
|
||||
0.25,
|
||||
160
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -51,14 +52,14 @@
|
||||
"1": 58
|
||||
},
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"outputs": [
|
||||
{
|
||||
"name": "hubert_model",
|
||||
"type": "HUBERT_MODEL",
|
||||
"links": [
|
||||
16
|
||||
28
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
@@ -139,7 +140,7 @@
|
||||
"1": 146
|
||||
},
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"outputs": [
|
||||
{
|
||||
@@ -185,7 +186,7 @@
|
||||
"1": 150
|
||||
},
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
@@ -285,65 +286,12 @@
|
||||
true
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 13,
|
||||
"type": "RVCProcessDatasetNode",
|
||||
"pos": [
|
||||
-8,
|
||||
176
|
||||
],
|
||||
"size": {
|
||||
"0": 337.6000061035156,
|
||||
"1": 294
|
||||
},
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "hubert_model",
|
||||
"type": "HUBERT_MODEL",
|
||||
"link": 16
|
||||
},
|
||||
{
|
||||
"name": "pitch_extraction_params",
|
||||
"type": "PITCH_EXTRACTION",
|
||||
"link": 17
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "rvc_dataset_pipe",
|
||||
"type": "RVC_DATASET_PIPE",
|
||||
"links": [
|
||||
22
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "RVCProcessDatasetNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Comfy-Sayano",
|
||||
"Sayano.zip",
|
||||
"40k",
|
||||
1,
|
||||
3,
|
||||
0.3,
|
||||
0.9500000000000001,
|
||||
0.75,
|
||||
0.01,
|
||||
"image"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 15,
|
||||
"type": "RVCTrainModelNode",
|
||||
"pos": [
|
||||
374,
|
||||
119
|
||||
118
|
||||
],
|
||||
"size": {
|
||||
"0": 506.4000244140625,
|
||||
@@ -356,7 +304,7 @@
|
||||
{
|
||||
"name": "rvc_dataset_pipe",
|
||||
"type": "RVC_DATASET_PIPE",
|
||||
"link": 22
|
||||
"link": 30
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
@@ -399,18 +347,70 @@
|
||||
},
|
||||
"widgets_values": [
|
||||
"0",
|
||||
4,
|
||||
300,
|
||||
0,
|
||||
"pretrained_v2/f0Ov2Super40kD.pth",
|
||||
"pretrained_v2/f0Ov2Super40kG.pth",
|
||||
true,
|
||||
5,
|
||||
100,
|
||||
10,
|
||||
"pretrained_v2/G_DMR-V1-32k-f0.pth",
|
||||
"pretrained_v2/D_DMR-V1-32k-f0.pth",
|
||||
true,
|
||||
false,
|
||||
false,
|
||||
true,
|
||||
true,
|
||||
true
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 17,
|
||||
"type": "RVCProcessDatasetNode",
|
||||
"pos": [
|
||||
-8,
|
||||
176
|
||||
],
|
||||
"size": {
|
||||
"0": 337.6000061035156,
|
||||
"1": 270
|
||||
},
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "hubert_model",
|
||||
"type": "HUBERT_MODEL",
|
||||
"link": 28
|
||||
},
|
||||
{
|
||||
"name": "pitch_extraction_params",
|
||||
"type": "PITCH_EXTRACTION",
|
||||
"link": 29
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "rvc_dataset_pipe",
|
||||
"type": "RVC_DATASET_PIPE",
|
||||
"links": [
|
||||
30
|
||||
],
|
||||
"shape": 3,
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "RVCProcessDatasetNode"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Comfy-Mae",
|
||||
"Mae.zip",
|
||||
"32k",
|
||||
12,
|
||||
3,
|
||||
0.5,
|
||||
0.9500000000000001,
|
||||
0.01,
|
||||
"image"
|
||||
]
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -446,30 +446,6 @@
|
||||
1,
|
||||
"AUDIO,VHS_AUDIO"
|
||||
],
|
||||
[
|
||||
16,
|
||||
5,
|
||||
0,
|
||||
13,
|
||||
0,
|
||||
"HUBERT_MODEL"
|
||||
],
|
||||
[
|
||||
17,
|
||||
3,
|
||||
0,
|
||||
13,
|
||||
1,
|
||||
"PITCH_EXTRACTION"
|
||||
],
|
||||
[
|
||||
22,
|
||||
13,
|
||||
0,
|
||||
15,
|
||||
0,
|
||||
"RVC_DATASET_PIPE"
|
||||
],
|
||||
[
|
||||
23,
|
||||
15,
|
||||
@@ -493,6 +469,30 @@
|
||||
6,
|
||||
3,
|
||||
"PITCH_EXTRACTION"
|
||||
],
|
||||
[
|
||||
28,
|
||||
5,
|
||||
0,
|
||||
17,
|
||||
0,
|
||||
"HUBERT_MODEL"
|
||||
],
|
||||
[
|
||||
29,
|
||||
3,
|
||||
0,
|
||||
17,
|
||||
1,
|
||||
"PITCH_EXTRACTION"
|
||||
],
|
||||
[
|
||||
30,
|
||||
17,
|
||||
0,
|
||||
15,
|
||||
0,
|
||||
"RVC_DATASET_PIPE"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
@@ -522,10 +522,10 @@
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 1.1,
|
||||
"scale": 0.6830134553650705,
|
||||
"offset": [
|
||||
652.4400126183286,
|
||||
108.78821868462816
|
||||
544.5635228085466,
|
||||
563.3345057885277
|
||||
]
|
||||
}
|
||||
},
|
||||
|
||||
+7
-9
@@ -44,7 +44,7 @@ def load_audio(file, sr, **kwargs):
|
||||
|
||||
return remix_audio((np.frombuffer(out, np.float32).flatten(), sr),**kwargs)
|
||||
|
||||
def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,**kwargs):
|
||||
def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,max_volume=.99,**kwargs):
|
||||
audio = np.array(input_audio[0],dtype="float32")
|
||||
if target_sr is None: target_sr=input_audio[1]
|
||||
|
||||
@@ -57,10 +57,10 @@ def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=Fa
|
||||
audio=merge_func(audio,axis=axis)
|
||||
if norm: audio = librosa.util.normalize(audio,axis=axis)
|
||||
|
||||
audio_max = np.abs(audio).max()/.99
|
||||
audio_max = np.abs(audio).max()/max_volume
|
||||
if audio_max > 1: audio = audio / audio_max
|
||||
|
||||
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1).astype("int16")
|
||||
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1).astype("int16")
|
||||
print(f"after remix: shape={audio.shape}, max={audio.max()}, min={audio.min()}, mean={audio.mean()}, sr={target_sr}")
|
||||
|
||||
return audio, target_sr
|
||||
@@ -71,16 +71,14 @@ def load_input_audio(fname,sr=None,**kwargs):
|
||||
print(f"loading sound {fname=} {audio.ndim=} {audio.max()=} {audio.min()=} {audio.dtype=} {sr=}")
|
||||
return audio, sr
|
||||
|
||||
def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False):
|
||||
def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False,max_volume=.99):
|
||||
print(f"saving sound to {fname}")
|
||||
os.makedirs(os.path.dirname(fname),exist_ok=True)
|
||||
audio=np.array(input_audio[0],dtype="float32")
|
||||
|
||||
if to_int16:
|
||||
audio_max = np.abs(audio).max()/.99
|
||||
if audio_max > 1: audio = audio / audio_max
|
||||
audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1)
|
||||
|
||||
audio_max = np.abs(audio).max()/max_volume
|
||||
if audio_max > 1: audio = audio / audio_max
|
||||
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1)
|
||||
if to_stereo and audio.ndim<2: audio=np.stack([audio,audio],axis=-1)
|
||||
|
||||
try:
|
||||
|
||||
+11
-31
@@ -1,19 +1,17 @@
|
||||
import sys, os, multiprocessing
|
||||
from threading import Thread
|
||||
import numpy as np, os, traceback
|
||||
from .lib.model_utils import load_hubert
|
||||
from .lib.slicer2 import Slicer
|
||||
import librosa, traceback
|
||||
import traceback
|
||||
from scipy.io import wavfile
|
||||
from .lib.audio import load_audio
|
||||
from .pitch_extraction import FeatureExtractor
|
||||
from .lib.audio import load_input_audio
|
||||
from .lib.audio import load_input_audio, remix_audio
|
||||
from .lib.utils import gc_collect
|
||||
from .config import config
|
||||
import torch
|
||||
|
||||
class Preprocess:
|
||||
def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, alpha=.75, max_volume=.95):
|
||||
def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, max_volume=.95):
|
||||
self.slicer = Slicer(
|
||||
sr=sr,
|
||||
threshold=-42,
|
||||
@@ -26,11 +24,10 @@ class Preprocess:
|
||||
self.per = period
|
||||
self.overlap = overlap
|
||||
self.tail = self.per + self.overlap
|
||||
self.max = max_volume
|
||||
self.alpha = alpha
|
||||
self.max_volume = max_volume
|
||||
self.exp_dir = exp_dir
|
||||
self.gt_wavs_dir = "%s/0_gt_wavs" % exp_dir
|
||||
self.wavs16k_dir = "%s/1_16k_wavs" % exp_dir
|
||||
self.gt_wavs_dir = os.path.join(exp_dir,"0_gt_wavs")
|
||||
self.wavs16k_dir = os.path.join(exp_dir,"1_16k_wavs")
|
||||
self.noparallel = noparallel
|
||||
os.makedirs(self.exp_dir, exist_ok=True)
|
||||
os.makedirs(self.gt_wavs_dir, exist_ok=True)
|
||||
@@ -45,26 +42,9 @@ class Preprocess:
|
||||
# mutex.release()
|
||||
|
||||
def norm_write(self, tmp_audio, idx0, idx1):
|
||||
tmp_max = np.abs(tmp_audio).max()
|
||||
if tmp_max > 2.5:
|
||||
print("%s-%s-%s-filtered" % (idx0, idx1, tmp_max))
|
||||
return
|
||||
tmp_audio = (tmp_audio / tmp_max * (self.max * self.alpha)) + (
|
||||
1 - self.alpha
|
||||
) * tmp_audio
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.gt_wavs_dir, idx0, idx1),
|
||||
self.sr,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
tmp_audio = librosa.resample(
|
||||
tmp_audio, orig_sr=self.sr, target_sr=16000
|
||||
) # , res_type="soxr_vhq"
|
||||
wavfile.write(
|
||||
"%s/%s_%s.wav" % (self.wavs16k_dir, idx0, idx1),
|
||||
16000,
|
||||
tmp_audio.astype(np.float32),
|
||||
)
|
||||
wavfile.write(os.path.join(self.gt_wavs_dir, f"{idx0}_{idx1}.wav"),self.sr,tmp_audio.astype(np.float32))
|
||||
remixed_audio = remix_audio((tmp_audio, self.sr), target_sr=16000, norm=True,max_volume=self.max_volume)
|
||||
wavfile.write(os.path.join(self.wavs16k_dir, f"{idx0}_{idx1}.wav"),16000,remixed_audio[0].astype(np.float32))
|
||||
|
||||
def pipeline(self, path, idx0):
|
||||
try:
|
||||
@@ -208,9 +188,9 @@ class FeatureInput(FeatureExtractor):
|
||||
except:
|
||||
self.printt("f0fail-%s-%s-%s" % (idx, inp_path, traceback.format_exc()))
|
||||
|
||||
def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1., alpha=.75):
|
||||
def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1.):
|
||||
try:
|
||||
pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume, alpha=alpha)
|
||||
pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume)
|
||||
pp.println("start preprocess")
|
||||
pp.println(sys.argv)
|
||||
pp.pipeline_mp_inp_dir(inp_root, n_p)
|
||||
|
||||
Reference in New Issue
Block a user