Squashed commit of the following:
commitfd32b14fdcAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Tue Dec 23 02:02:31 2025 +0200 Clean prints commit1776695e26Author: kijai <40791699+kijai@users.noreply.github.com> Date: Tue Dec 23 01:48:12 2025 +0200 Update nodes_model_loading.py commitef36204fa8Author: kijai <40791699+kijai@users.noreply.github.com> Date: Tue Dec 23 01:35:08 2025 +0200 Reduce peak VRAM use commitc6f32c1424Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Dec 22 23:53:41 2025 +0200 Norm dtype commit6d4a0f6e53Merge:e7e00063e45021Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Dec 22 22:11:38 2025 +0200 Merge branch 'main' into longcat_avatar commite7e00061e5Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Dec 22 00:43:01 2025 +0200 Update nodes_sampler.py commiteb5ec262a0Merge:7c0ba84fed3b22Author: kijai <40791699+kijai@users.noreply.github.com> Date: Mon Dec 22 00:42:53 2025 +0200 Merge branch 'main' into longcat_avatar commit7c0ba84a26Author: kijai <40791699+kijai@users.noreply.github.com> Date: Sun Dec 21 23:00:43 2025 +0200 remove prints commit06a86923e7Author: kijai <40791699+kijai@users.noreply.github.com> Date: Sun Dec 21 22:53:25 2025 +0200 Fix ref latent oops commitdca3106f10Author: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Dec 20 18:46:32 2025 +0200 Expose more options, make vid2vid easier commit175418b8d2Author: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Dec 20 03:15:24 2025 +0200 Create LongCatAvatar_testing_wip.json commit4a6e2d3c6cAuthor: kijai <40791699+kijai@users.noreply.github.com> Date: Sat Dec 20 03:14:49 2025 +0200 Init
This commit is contained in:
+19
-1
@@ -7,6 +7,8 @@ from ..utils import log, set_module_tensor_to_device
|
||||
import os
|
||||
import json
|
||||
import datetime
|
||||
import scipy.signal as ss
|
||||
import numpy as np
|
||||
|
||||
script_directory = os.path.dirname(os.path.abspath(__file__))
|
||||
folder_paths.add_model_folder_path("wav2vec2", os.path.join(folder_paths.models_dir, "wav2vec2"))
|
||||
@@ -134,6 +136,15 @@ def loudness_norm(audio_array, sr=16000, lufs=-23):
|
||||
return audio_array
|
||||
normalized_audio = pyloudnorm.normalize.loudness(audio_array, loudness, lufs)
|
||||
return normalized_audio
|
||||
|
||||
def _add_noise_floor(audio, noise_db=-45):
|
||||
noise_amp = 10 ** (noise_db / 20)
|
||||
noise = np.random.randn(len(audio)) * noise_amp
|
||||
return audio + noise
|
||||
|
||||
def _smooth_transients(audio, sr=16000):
|
||||
b, a = ss.butter(3, 3000 / (sr/2))
|
||||
return ss.lfilter(b, a, audio)
|
||||
|
||||
class MultiTalkWav2VecEmbeds:
|
||||
@classmethod
|
||||
@@ -153,6 +164,8 @@ class MultiTalkWav2VecEmbeds:
|
||||
"audio_3": ("AUDIO",),
|
||||
"audio_4": ("AUDIO",),
|
||||
"ref_target_masks": ("MASK", {"tooltip": "Per-speaker semantic mask(s) in pixel space. Supply one mask per speaker (plus optional background) to guide mouth assignment"}),
|
||||
"add_noise_floor": ("BOOLEAN", {"default": False, "tooltip": "Add a low-level noise floor to the audio to reduce silent gaps"}),
|
||||
"smooth_transients": ("BOOLEAN", {"default": False, "tooltip": "Apply a low-pass filter to the audio to smooth out transients"}),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -161,7 +174,8 @@ class MultiTalkWav2VecEmbeds:
|
||||
FUNCTION = "process"
|
||||
CATEGORY = "WanVideoWrapper"
|
||||
|
||||
def process(self, wav2vec_model, normalize_loudness, fps, num_frames, audio_1, audio_scale, audio_cfg_scale, multi_audio_type, audio_2=None, audio_3=None, audio_4=None, ref_target_masks=None):
|
||||
def process(self, wav2vec_model, normalize_loudness, fps, num_frames, audio_1, audio_scale, audio_cfg_scale, multi_audio_type, audio_2=None, audio_3=None, audio_4=None,
|
||||
ref_target_masks=None, add_noise_floor=False, smooth_transients=False):
|
||||
model_type = wav2vec_model["model_type"]
|
||||
if not "tencent" in model_type.lower():
|
||||
raise ValueError("Only tencent wav2vec2 models supported by MultiTalk")
|
||||
@@ -207,6 +221,10 @@ class MultiTalkWav2VecEmbeds:
|
||||
|
||||
if normalize_loudness:
|
||||
audio_segment = loudness_norm(audio_segment, sr=sr)
|
||||
if add_noise_floor:
|
||||
audio_segment = _add_noise_floor(audio_segment, noise_db=-45)
|
||||
if smooth_transients:
|
||||
audio_segment = _smooth_transients(audio_segment, sr=sr)
|
||||
|
||||
audio_feature = np.squeeze(
|
||||
wav2vec2_feature_extractor(audio_segment, sampling_rate=sr).input_values
|
||||
|
||||
Reference in New Issue
Block a user