better preprocessing norm

This commit is contained in:
SayanoAI
2024-08-25 21:20:23 -04:00
parent dadc0660ec
commit c899f86a1e
5 changed files with 123 additions and 146 deletions
+1 -1
View File
@@ -203,7 +203,7 @@ class PreviewAudio:
output_path = increment_filename_no_overwrite(output_path)
input_audio = get_audio(audio)
print(save_input_audio(output_path,input_audio,to_int16=True,to_stereo=save_channels==2))
print(save_input_audio(output_path,input_audio,to_int16=save_format!="mp3",to_stereo=save_channels==2))
tempdir = os.path.join(temp_path,"preview")
os.makedirs(tempdir, exist_ok=True)
+4 -5
View File
@@ -235,8 +235,7 @@ class RVCProcessDatasetNode:
"n_threads": ("INT", {"default": get_optimal_threads(), "min": 1, "max": multiprocessing.cpu_count()}),
"period": ("FLOAT", {"default": 3., "min": 1., "max": 10., "step": .1}),
"overlap": ("FLOAT",{"default": .3, "min": .1, "max": 1., "step": .1}),
"max_volume": ("FLOAT",{"default": 1., "min": .1, "max": 1., "step": .05}),
"alpha": ("FLOAT",{"default": .75, "min": .05, "max": 1., "step": .05}),
"max_volume": ("FLOAT",{"default": .99, "min": .1, "max": 1., "step": .01}),
"mute_ratio": ("FLOAT",{"default": .0, "min": .0, "max": .5, "step": .01}),
}
}
@@ -248,13 +247,13 @@ class RVCProcessDatasetNode:
CATEGORY = CATEGORY
def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., alpha=.75, mute_ratio=.0):
def process(self, model_name: str, dataset: str, hubert_model, pitch_extraction_params={}, sr="40k", n_threads=1, period=3., overlap=.3, max_volume=1., mute_ratio=.0):
assert model_name, "Please provide a model name!"
assert dataset, "Please upload a dataset!"
f0_method = pitch_extraction_params.get("f0_method", "")
cached_params = [model_name, dataset, period, overlap, max_volume, alpha, mute_ratio, sr, f0_method]
cached_params = [model_name, dataset, period, overlap, max_volume, mute_ratio, sr, f0_method]
crepe_hop_length = pitch_extraction_params.get("crepe_hop_length",160)
if "crepe" in f0_method: cached_params.append(crepe_hop_length)
@@ -272,7 +271,7 @@ class RVCProcessDatasetNode:
files = extract_zip_without_structure(os.path.join(dataset_path,dataset),dataset_dir)
assert len(files), "Failed to extract zip file..."
print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume,alpha))
print(preprocess_trainset(dataset_dir,SR_MAP[sr],n_threads,model_log_dir,period,overlap,max_volume))
print(extract_features_trainset(hubert_model(), model_log_dir,n_p=n_threads,f0method=f0_method,device=device,if_f0=bool(f0_method),version="v2",crepe_hop_length=crepe_hop_length))
+100 -100
View File
@@ -1,27 +1,27 @@
{
"last_node_id": 15,
"last_link_id": 25,
"last_node_id": 17,
"last_link_id": 30,
"nodes": [
{
"id": 3,
"type": "LoadPitchExtractionParams",
"pos": [
-397,
179
-387,
130
],
"size": {
"0": 315,
"1": 178
"1": 202
},
"flags": {},
"order": 2,
"order": 0,
"mode": 0,
"outputs": [
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"links": [
17
29
],
"slot_index": 0,
"shape": 3
@@ -36,7 +36,8 @@
0.75,
0,
0.25,
0.25
0.25,
160
]
},
{
@@ -51,14 +52,14 @@
"1": 58
},
"flags": {},
"order": 0,
"order": 1,
"mode": 0,
"outputs": [
{
"name": "hubert_model",
"type": "HUBERT_MODEL",
"links": [
16
28
],
"slot_index": 0,
"shape": 3
@@ -139,7 +140,7 @@
"1": 146
},
"flags": {},
"order": 1,
"order": 2,
"mode": 0,
"outputs": [
{
@@ -185,7 +186,7 @@
"1": 150
},
"flags": {},
"order": 3,
"order": 4,
"mode": 0,
"inputs": [
{
@@ -285,65 +286,12 @@
true
]
},
{
"id": 13,
"type": "RVCProcessDatasetNode",
"pos": [
-8,
176
],
"size": {
"0": 337.6000061035156,
"1": 294
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "hubert_model",
"type": "HUBERT_MODEL",
"link": 16
},
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"link": 17
}
],
"outputs": [
{
"name": "rvc_dataset_pipe",
"type": "RVC_DATASET_PIPE",
"links": [
22
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "RVCProcessDatasetNode"
},
"widgets_values": [
"Comfy-Sayano",
"Sayano.zip",
"40k",
1,
3,
0.3,
0.9500000000000001,
0.75,
0.01,
"image"
]
},
{
"id": 15,
"type": "RVCTrainModelNode",
"pos": [
374,
119
118
],
"size": {
"0": 506.4000244140625,
@@ -356,7 +304,7 @@
{
"name": "rvc_dataset_pipe",
"type": "RVC_DATASET_PIPE",
"link": 22
"link": 30
}
],
"outputs": [
@@ -399,18 +347,70 @@
},
"widgets_values": [
"0",
4,
300,
0,
"pretrained_v2/f0Ov2Super40kD.pth",
"pretrained_v2/f0Ov2Super40kG.pth",
true,
5,
100,
10,
"pretrained_v2/G_DMR-V1-32k-f0.pth",
"pretrained_v2/D_DMR-V1-32k-f0.pth",
true,
false,
false,
true,
true,
true
]
},
{
"id": 17,
"type": "RVCProcessDatasetNode",
"pos": [
-8,
176
],
"size": {
"0": 337.6000061035156,
"1": 270
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "hubert_model",
"type": "HUBERT_MODEL",
"link": 28
},
{
"name": "pitch_extraction_params",
"type": "PITCH_EXTRACTION",
"link": 29
}
],
"outputs": [
{
"name": "rvc_dataset_pipe",
"type": "RVC_DATASET_PIPE",
"links": [
30
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "RVCProcessDatasetNode"
},
"widgets_values": [
"Comfy-Mae",
"Mae.zip",
"32k",
12,
3,
0.5,
0.9500000000000001,
0.01,
"image"
]
}
],
"links": [
@@ -446,30 +446,6 @@
1,
"AUDIO,VHS_AUDIO"
],
[
16,
5,
0,
13,
0,
"HUBERT_MODEL"
],
[
17,
3,
0,
13,
1,
"PITCH_EXTRACTION"
],
[
22,
13,
0,
15,
0,
"RVC_DATASET_PIPE"
],
[
23,
15,
@@ -493,6 +469,30 @@
6,
3,
"PITCH_EXTRACTION"
],
[
28,
5,
0,
17,
0,
"HUBERT_MODEL"
],
[
29,
3,
0,
17,
1,
"PITCH_EXTRACTION"
],
[
30,
17,
0,
15,
0,
"RVC_DATASET_PIPE"
]
],
"groups": [
@@ -522,10 +522,10 @@
"config": {},
"extra": {
"ds": {
"scale": 1.1,
"scale": 0.6830134553650705,
"offset": [
652.4400126183286,
108.78821868462816
544.5635228085466,
563.3345057885277
]
}
},
+7 -9
View File
@@ -44,7 +44,7 @@ def load_audio(file, sr, **kwargs):
return remix_audio((np.frombuffer(out, np.float32).flatten(), sr),**kwargs)
def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,**kwargs):
def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=False,axis=0,merge_type=None,max_volume=.99,**kwargs):
audio = np.array(input_audio[0],dtype="float32")
if target_sr is None: target_sr=input_audio[1]
@@ -57,10 +57,10 @@ def remix_audio(input_audio,target_sr=None,norm=False,to_int16=False,resample=Fa
audio=merge_func(audio,axis=axis)
if norm: audio = librosa.util.normalize(audio,axis=axis)
audio_max = np.abs(audio).max()/.99
audio_max = np.abs(audio).max()/max_volume
if audio_max > 1: audio = audio / audio_max
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1).astype("int16")
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1).astype("int16")
print(f"after remix: shape={audio.shape}, max={audio.max()}, min={audio.min()}, mean={audio.mean()}, sr={target_sr}")
return audio, target_sr
@@ -71,16 +71,14 @@ def load_input_audio(fname,sr=None,**kwargs):
print(f"loading sound {fname=} {audio.ndim=} {audio.max()=} {audio.min()=} {audio.dtype=} {sr=}")
return audio, sr
def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False):
def save_input_audio(fname,input_audio,sr=None,to_int16=False,to_stereo=False,max_volume=.99):
print(f"saving sound to {fname}")
os.makedirs(os.path.dirname(fname),exist_ok=True)
audio=np.array(input_audio[0],dtype="float32")
if to_int16:
audio_max = np.abs(audio).max()/.99
if audio_max > 1: audio = audio / audio_max
audio = np.clip(audio * MAX_INT16, a_min=-MAX_INT16+1, a_max=MAX_INT16-1)
audio_max = np.abs(audio).max()/max_volume
if audio_max > 1: audio = audio / audio_max
if to_int16: audio = np.clip(audio * MAX_INT16, a_min=1-MAX_INT16, a_max=MAX_INT16-1)
if to_stereo and audio.ndim<2: audio=np.stack([audio,audio],axis=-1)
try:
+11 -31
View File
@@ -1,19 +1,17 @@
import sys, os, multiprocessing
from threading import Thread
import numpy as np, os, traceback
from .lib.model_utils import load_hubert
from .lib.slicer2 import Slicer
import librosa, traceback
import traceback
from scipy.io import wavfile
from .lib.audio import load_audio
from .pitch_extraction import FeatureExtractor
from .lib.audio import load_input_audio
from .lib.audio import load_input_audio, remix_audio
from .lib.utils import gc_collect
from .config import config
import torch
class Preprocess:
def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, alpha=.75, max_volume=.95):
def __init__(self, sr, exp_dir, noparallel=True, period=3.0, overlap=.3, max_volume=.95):
self.slicer = Slicer(
sr=sr,
threshold=-42,
@@ -26,11 +24,10 @@ class Preprocess:
self.per = period
self.overlap = overlap
self.tail = self.per + self.overlap
self.max = max_volume
self.alpha = alpha
self.max_volume = max_volume
self.exp_dir = exp_dir
self.gt_wavs_dir = "%s/0_gt_wavs" % exp_dir
self.wavs16k_dir = "%s/1_16k_wavs" % exp_dir
self.gt_wavs_dir = os.path.join(exp_dir,"0_gt_wavs")
self.wavs16k_dir = os.path.join(exp_dir,"1_16k_wavs")
self.noparallel = noparallel
os.makedirs(self.exp_dir, exist_ok=True)
os.makedirs(self.gt_wavs_dir, exist_ok=True)
@@ -45,26 +42,9 @@ class Preprocess:
# mutex.release()
def norm_write(self, tmp_audio, idx0, idx1):
tmp_max = np.abs(tmp_audio).max()
if tmp_max > 2.5:
print("%s-%s-%s-filtered" % (idx0, idx1, tmp_max))
return
tmp_audio = (tmp_audio / tmp_max * (self.max * self.alpha)) + (
1 - self.alpha
) * tmp_audio
wavfile.write(
"%s/%s_%s.wav" % (self.gt_wavs_dir, idx0, idx1),
self.sr,
tmp_audio.astype(np.float32),
)
tmp_audio = librosa.resample(
tmp_audio, orig_sr=self.sr, target_sr=16000
) # , res_type="soxr_vhq"
wavfile.write(
"%s/%s_%s.wav" % (self.wavs16k_dir, idx0, idx1),
16000,
tmp_audio.astype(np.float32),
)
wavfile.write(os.path.join(self.gt_wavs_dir, f"{idx0}_{idx1}.wav"),self.sr,tmp_audio.astype(np.float32))
remixed_audio = remix_audio((tmp_audio, self.sr), target_sr=16000, norm=True,max_volume=self.max_volume)
wavfile.write(os.path.join(self.wavs16k_dir, f"{idx0}_{idx1}.wav"),16000,remixed_audio[0].astype(np.float32))
def pipeline(self, path, idx0):
try:
@@ -208,9 +188,9 @@ class FeatureInput(FeatureExtractor):
except:
self.printt("f0fail-%s-%s-%s" % (idx, inp_path, traceback.format_exc()))
def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1., alpha=.75):
def preprocess_trainset(inp_root, sr, n_p, exp_dir, period=3.0, overlap=.3, max_volume=1.):
try:
pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume, alpha=alpha)
pp = Preprocess(sr, exp_dir, period=period, overlap=overlap, max_volume=max_volume)
pp.println("start preprocess")
pp.println(sys.argv)
pp.pipeline_mp_inp_dir(inp_root, n_p)