add 'MW Audio Recorder' node
This commit is contained in:
@@ -0,0 +1,140 @@
|
||||
import numpy as np
|
||||
import torch
|
||||
import time
|
||||
import librosa
|
||||
import sounddevice as sd
|
||||
from scipy import ndimage
|
||||
from comfy.utils import ProgressBar
|
||||
|
||||
class AudioRecorder:
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
# 触发控制
|
||||
"trigger": ("BOOLEAN", {"default": False}),
|
||||
# 录音时长
|
||||
"record_sec": ("INT", {
|
||||
"default": 5,
|
||||
"min": 1,
|
||||
"max": 60,
|
||||
"step": 1 # 整数秒递增
|
||||
}),
|
||||
"sample_rate": (["16000", "44100", "48000"], { # 限定标准采样率
|
||||
"default": "48000"
|
||||
}),
|
||||
"n_fft": ("INT", { # 限定为2的幂次方
|
||||
"default": 2048,
|
||||
"min": 512,
|
||||
"max": 4096,
|
||||
"step": 512 # 只能选择512,1024,1536,2048,...4096
|
||||
}),
|
||||
"sensitivity": ("FLOAT", { # 灵敏度精确控制
|
||||
"default": 1.2,
|
||||
"min": 0.5,
|
||||
"max": 3.0,
|
||||
"step": 0.1 # 0.1步进
|
||||
}),
|
||||
"smooth": ("INT", { # 确保为奇数
|
||||
"default": 5,
|
||||
"min": 1,
|
||||
"max": 11,
|
||||
"step": 2 # 生成1,3,5,7,9,11
|
||||
}),
|
||||
"seed": ("INT", {"default": 0, "min": 0, "max": 0xFFFFFFFFFFFFFFFF}),
|
||||
}
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("AUDIO",)
|
||||
RETURN_NAMES = ("audio",)
|
||||
FUNCTION = "record_and_clean"
|
||||
CATEGORY = "MW-Step-Audio"
|
||||
|
||||
def _stft(self, y, n_fft):
|
||||
hop = n_fft // 4
|
||||
return librosa.stft(y, n_fft=n_fft, hop_length=hop, win_length=n_fft)
|
||||
|
||||
def _istft(self, spec, n_fft):
|
||||
hop = n_fft // 4
|
||||
return librosa.istft(spec, hop_length=hop, win_length=n_fft)
|
||||
|
||||
def _calc_noise_profile(self, noise_clip, n_fft):
|
||||
noise_spec = self._stft(noise_clip, n_fft)
|
||||
return {
|
||||
'mean': np.mean(np.abs(noise_spec), axis=1, keepdims=True),
|
||||
'std': np.std(np.abs(noise_spec), axis=1, keepdims=True)
|
||||
}
|
||||
|
||||
def _spectral_gate(self, spec, noise_profile, sensitivity):
|
||||
threshold = noise_profile['mean'] + sensitivity * noise_profile['std']
|
||||
return np.where(np.abs(spec) > threshold, spec, 0)
|
||||
|
||||
def _smooth_mask(self, mask, kernel_size):
|
||||
smoothed = ndimage.uniform_filter(mask, size=(kernel_size, kernel_size))
|
||||
return np.clip(smoothed * 1.2, 0, 1) # 增强边缘保留
|
||||
|
||||
def record_and_clean(self, trigger, record_sec, n_fft, sensitivity, smooth, sample_rate, seed):
|
||||
if not trigger:
|
||||
return (None,)
|
||||
|
||||
sr = int(sample_rate)
|
||||
final_audio = None
|
||||
|
||||
try:
|
||||
noise_clip = None
|
||||
# 主录音
|
||||
# print(f"开始主录音 {record_sec}秒...")
|
||||
main_rec = sd.rec(int(record_sec * sr), samplerate=sr, channels=1, dtype='float32')
|
||||
pb = ProgressBar(record_sec)
|
||||
for _ in range(record_sec * 2):
|
||||
time.sleep(0.5)
|
||||
pb.update(0.5)
|
||||
sd.wait()
|
||||
audio = main_rec.flatten()
|
||||
|
||||
# 自动噪声检测
|
||||
if noise_clip is None:
|
||||
# print("自动检测静默段作为噪声参考...")
|
||||
energy = librosa.feature.rms(y=audio, frame_length=n_fft, hop_length=n_fft//4)
|
||||
min_idx = np.argmin(energy)
|
||||
start = min_idx * (n_fft//4)
|
||||
noise_clip = audio[start:start + n_fft*2]
|
||||
|
||||
# 降噪处理
|
||||
# print("进行频谱降噪...")
|
||||
noise_profile = self._calc_noise_profile(noise_clip, n_fft)
|
||||
spec = self._stft(audio, n_fft)
|
||||
|
||||
# 多步骤处理
|
||||
mask = np.ones_like(spec) # 初始掩膜
|
||||
for _ in range(2): # 双重处理循环
|
||||
cleaned_spec = self._spectral_gate(spec, noise_profile, sensitivity)
|
||||
mask = np.where(np.abs(cleaned_spec) > 0, 1, 0)
|
||||
mask = self._smooth_mask(mask, smooth//2+1)
|
||||
spec = spec * mask
|
||||
|
||||
# 相位恢复重建
|
||||
processed = self._istft(spec * mask, n_fft)
|
||||
|
||||
# 动态增益归一化
|
||||
peak = np.max(np.abs(processed))
|
||||
processed = processed * (0.99 / peak) if peak > 0 else processed
|
||||
|
||||
# 格式转换
|
||||
waveform = torch.from_numpy(processed).float().unsqueeze(0).unsqueeze(0)
|
||||
final_audio = {"waveform": waveform, "sample_rate": sr}
|
||||
|
||||
except Exception as e:
|
||||
print(f"Recording/processing failed: {str(e)}")
|
||||
raise
|
||||
|
||||
return (final_audio,)
|
||||
|
||||
# 节点注册
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"AudioRecorder": AudioRecorder
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"AudioRecorder": "MW Audio Recorder"
|
||||
}
|
||||
@@ -6,6 +6,10 @@
|
||||
|
||||
## Update
|
||||
|
||||
[2025-03-06]⚒️: New recording node `MW Audio Recorder` can be used to record audio with a microphone, and the progress bar displays the recording progress:
|
||||
|
||||

|
||||
|
||||
[2025-03-02]⚒️: Add experimental `custom_mark`, surrounding with "()", for example `(温柔)(东北话)`, it may have an effect.
|
||||
|
||||
[2025-02-25]⚒️: Support custom speaker `custom_stpeaker`. But please note that the three places in the figure must maintain consistent speaker name. When customizing, the `speaker` will automatically become invalid.
|
||||
|
||||
@@ -6,6 +6,10 @@
|
||||
|
||||
## 更新
|
||||
|
||||
[2025-03-06]⚒️: 新增录音节点 `MW Audio Recorder` 可用麦克风录制音频, 进度条显示录制进度:
|
||||
|
||||

|
||||
|
||||
[2025-03-02]⚒️: 增加实验性的 `custom_mark`, 用 "()" 包围例如 `(温柔)(东北话)`, 它可能会有效.
|
||||
|
||||
[2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`. 但注意下图三个地方必须保持一致的说话者名称. 自定义时 `speaker` 将自动无效.
|
||||
|
||||
+8
-55
@@ -85,61 +85,6 @@ class StepAudioTTS:
|
||||
return self._music_cosy_model
|
||||
|
||||
def __call__(self, text: str, cosy_model, prompt_speaker_info, history):
|
||||
# instruction_name = self.detect_instruction_name(text)
|
||||
|
||||
# if "RAP" in instruction_name or "哼唱" in instruction_name:
|
||||
# cosy_model = self.music_cosy_model
|
||||
# else:
|
||||
# cosy_model = self.common_cosy_model
|
||||
|
||||
# prompt_speaker_info = {}
|
||||
# if clone_dict:
|
||||
# clone_prompt_code, clone_prompt_token, clone_prompt_token_len, clone_speech_feat, clone_speech_feat_len, clone_speech_embedding = (
|
||||
# self.preprocess_prompt_wav(clone_dict['audio'], cosy_model)
|
||||
# )
|
||||
# prompt_speaker_info = {
|
||||
# "prompt_text": clone_dict['prompt_text'],
|
||||
# "prompt_code": clone_prompt_code,
|
||||
# "cosy_speech_feat": clone_speech_feat.to(torch.bfloat16),
|
||||
# "cosy_speech_feat_len": clone_speech_feat_len,
|
||||
# "cosy_speech_embedding": clone_speech_embedding.to(torch.bfloat16),
|
||||
# "cosy_prompt_token": clone_prompt_token,
|
||||
# "cosy_prompt_token_len": clone_prompt_token_len,
|
||||
# }
|
||||
|
||||
# prompt_speaker = clone_dict['speaker']
|
||||
# # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n")
|
||||
|
||||
# else:
|
||||
# with open(f"{speaker_path}/speakers_info.json", "r") as f:
|
||||
# speakers_info = json.load(f)
|
||||
|
||||
# for speaker_id, prompt_text in speakers_info.items():
|
||||
# if speaker_id == prompt_speaker:
|
||||
# prompt_wav_path = f"{speaker_path}/{speaker_id}_prompt.wav"
|
||||
# waveform, sample_rate = torchaudio.load(prompt_wav_path)
|
||||
# audio = {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate}
|
||||
# prompt_code, prompt_token, prompt_token_len, speech_feat, speech_feat_len, speech_embedding = (
|
||||
# self.preprocess_prompt_wav(audio, cosy_model)
|
||||
# )
|
||||
# prompt_speaker_info = {
|
||||
# "prompt_text": prompt_text,
|
||||
# "prompt_code": prompt_code,
|
||||
# "cosy_speech_feat": speech_feat.to(torch.bfloat16),
|
||||
# "cosy_speech_feat_len": speech_feat_len,
|
||||
# "cosy_speech_embedding": speech_embedding.to(torch.bfloat16),
|
||||
# "cosy_prompt_token": prompt_token,
|
||||
# "cosy_prompt_token_len": prompt_token_len,
|
||||
# }
|
||||
# # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n")
|
||||
# break
|
||||
|
||||
# elif prompt_speaker not in speakers_info.keys():
|
||||
# raise ValueError("There is no such speaker")
|
||||
|
||||
# print("指定文本: ", text, "的说话者是: ", prompt_speaker, end="\n\n")
|
||||
|
||||
# cosy_model, prompt_speaker_info, history = self.data_preprocess()
|
||||
|
||||
_prefix_tokens = self.autotokenizer.encode("\n")
|
||||
target_token_encode = self.autotokenizer.encode("\n" + text)
|
||||
@@ -482,8 +427,16 @@ class StepAudioClone:
|
||||
audio_tensor = torch.cat(audio_data, dim=1).unsqueeze(0).float()
|
||||
return ({"waveform": audio_tensor, "sample_rate": sr},)
|
||||
|
||||
from MWAudioRecorder import AudioRecorder
|
||||
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"StepAudioRun": StepAudioRun,
|
||||
"StepAudioClone": StepAudioClone,
|
||||
"AudioRecorder": AudioRecorder
|
||||
}
|
||||
|
||||
NODE_DISPLAY_NAME_MAPPINGS = {
|
||||
"StepAudioRun": "Step Audio Run",
|
||||
"StepAudioClone": "Step Audio Clone",
|
||||
"AudioRecorder": "MW Audio Recorder"
|
||||
}
|
||||
+2
-2
@@ -4,6 +4,6 @@ import os
|
||||
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
sys.path.insert(0, current_dir)
|
||||
|
||||
from StepAudioTTS import NODE_CLASS_MAPPINGS
|
||||
from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
|
||||
|
||||
__all__ = ["NODE_CLASS_MAPPINGS"]
|
||||
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 27 KiB |
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "stepaudiotts_mw"
|
||||
description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice."
|
||||
version = "1.1.2"
|
||||
version = "1.1.3"
|
||||
license = {file = "LICENSE"}
|
||||
dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"]
|
||||
|
||||
|
||||
Reference in New Issue
Block a user