add 'MW Audio Recorder' node

This commit is contained in:
billwuhao
2025-03-06 21:35:11 +08:00
parent 33515baac4
commit e62f6c67a6
7 changed files with 159 additions and 58 deletions
+140
View File
@@ -0,0 +1,140 @@
import numpy as np
import torch
import time
import librosa
import sounddevice as sd
from scipy import ndimage
from comfy.utils import ProgressBar
class AudioRecorder:
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
# 触发控制
"trigger": ("BOOLEAN", {"default": False}),
# 录音时长
"record_sec": ("INT", {
"default": 5,
"min": 1,
"max": 60,
"step": 1 # 整数秒递增
}),
"sample_rate": (["16000", "44100", "48000"], { # 限定标准采样率
"default": "48000"
}),
"n_fft": ("INT", { # 限定为2的幂次方
"default": 2048,
"min": 512,
"max": 4096,
"step": 512 # 只能选择512,1024,1536,2048,...4096
}),
"sensitivity": ("FLOAT", { # 灵敏度精确控制
"default": 1.2,
"min": 0.5,
"max": 3.0,
"step": 0.1 # 0.1步进
}),
"smooth": ("INT", { # 确保为奇数
"default": 5,
"min": 1,
"max": 11,
"step": 2 # 生成1,3,5,7,9,11
}),
"seed": ("INT", {"default": 0, "min": 0, "max": 0xFFFFFFFFFFFFFFFF}),
}
}
RETURN_TYPES = ("AUDIO",)
RETURN_NAMES = ("audio",)
FUNCTION = "record_and_clean"
CATEGORY = "MW-Step-Audio"
def _stft(self, y, n_fft):
hop = n_fft // 4
return librosa.stft(y, n_fft=n_fft, hop_length=hop, win_length=n_fft)
def _istft(self, spec, n_fft):
hop = n_fft // 4
return librosa.istft(spec, hop_length=hop, win_length=n_fft)
def _calc_noise_profile(self, noise_clip, n_fft):
noise_spec = self._stft(noise_clip, n_fft)
return {
'mean': np.mean(np.abs(noise_spec), axis=1, keepdims=True),
'std': np.std(np.abs(noise_spec), axis=1, keepdims=True)
}
def _spectral_gate(self, spec, noise_profile, sensitivity):
threshold = noise_profile['mean'] + sensitivity * noise_profile['std']
return np.where(np.abs(spec) > threshold, spec, 0)
def _smooth_mask(self, mask, kernel_size):
smoothed = ndimage.uniform_filter(mask, size=(kernel_size, kernel_size))
return np.clip(smoothed * 1.2, 0, 1) # 增强边缘保留
def record_and_clean(self, trigger, record_sec, n_fft, sensitivity, smooth, sample_rate, seed):
if not trigger:
return (None,)
sr = int(sample_rate)
final_audio = None
try:
noise_clip = None
# 主录音
# print(f"开始主录音 {record_sec}秒...")
main_rec = sd.rec(int(record_sec * sr), samplerate=sr, channels=1, dtype='float32')
pb = ProgressBar(record_sec)
for _ in range(record_sec * 2):
time.sleep(0.5)
pb.update(0.5)
sd.wait()
audio = main_rec.flatten()
# 自动噪声检测
if noise_clip is None:
# print("自动检测静默段作为噪声参考...")
energy = librosa.feature.rms(y=audio, frame_length=n_fft, hop_length=n_fft//4)
min_idx = np.argmin(energy)
start = min_idx * (n_fft//4)
noise_clip = audio[start:start + n_fft*2]
# 降噪处理
# print("进行频谱降噪...")
noise_profile = self._calc_noise_profile(noise_clip, n_fft)
spec = self._stft(audio, n_fft)
# 多步骤处理
mask = np.ones_like(spec) # 初始掩膜
for _ in range(2): # 双重处理循环
cleaned_spec = self._spectral_gate(spec, noise_profile, sensitivity)
mask = np.where(np.abs(cleaned_spec) > 0, 1, 0)
mask = self._smooth_mask(mask, smooth//2+1)
spec = spec * mask
# 相位恢复重建
processed = self._istft(spec * mask, n_fft)
# 动态增益归一化
peak = np.max(np.abs(processed))
processed = processed * (0.99 / peak) if peak > 0 else processed
# 格式转换
waveform = torch.from_numpy(processed).float().unsqueeze(0).unsqueeze(0)
final_audio = {"waveform": waveform, "sample_rate": sr}
except Exception as e:
print(f"Recording/processing failed: {str(e)}")
raise
return (final_audio,)
# 节点注册
NODE_CLASS_MAPPINGS = {
"AudioRecorder": AudioRecorder
}
NODE_DISPLAY_NAME_MAPPINGS = {
"AudioRecorder": "MW Audio Recorder"
}
+4
View File
@@ -6,6 +6,10 @@
## Update
[2025-03-06]⚒️: New recording node `MW Audio Recorder` can be used to record audio with a microphone, and the progress bar displays the recording progress:
![](https://github.com/billwuhao/ComfyUI_StepAudioTTS/blob/master/assets/2025-03-06_21-29-09.png)
[2025-03-02]⚒️: Add experimental `custom_mark`, surrounding with "()", for example `(温柔)(东北话)`, it may have an effect.
[2025-02-25]⚒️: Support custom speaker `custom_stpeaker`. But please note that the three places in the figure must maintain consistent speaker name. When customizing, the `speaker` will automatically become invalid.
+4
View File
@@ -6,6 +6,10 @@
## 更新
[2025-03-06]⚒️: 新增录音节点 `MW Audio Recorder` 可用麦克风录制音频, 进度条显示录制进度:
![](https://github.com/billwuhao/ComfyUI_StepAudioTTS/blob/master/assets/2025-03-06_21-29-09.png)
[2025-03-02]⚒️: 增加实验性的 `custom_mark`, 用 "()" 包围例如 `(温柔)(东北话)`, 它可能会有效.
[2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`. 但注意下图三个地方必须保持一致的说话者名称. 自定义时 `speaker` 将自动无效.
+8 -55
View File
@@ -85,61 +85,6 @@ class StepAudioTTS:
return self._music_cosy_model
def __call__(self, text: str, cosy_model, prompt_speaker_info, history):
# instruction_name = self.detect_instruction_name(text)
# if "RAP" in instruction_name or "哼唱" in instruction_name:
# cosy_model = self.music_cosy_model
# else:
# cosy_model = self.common_cosy_model
# prompt_speaker_info = {}
# if clone_dict:
# clone_prompt_code, clone_prompt_token, clone_prompt_token_len, clone_speech_feat, clone_speech_feat_len, clone_speech_embedding = (
# self.preprocess_prompt_wav(clone_dict['audio'], cosy_model)
# )
# prompt_speaker_info = {
# "prompt_text": clone_dict['prompt_text'],
# "prompt_code": clone_prompt_code,
# "cosy_speech_feat": clone_speech_feat.to(torch.bfloat16),
# "cosy_speech_feat_len": clone_speech_feat_len,
# "cosy_speech_embedding": clone_speech_embedding.to(torch.bfloat16),
# "cosy_prompt_token": clone_prompt_token,
# "cosy_prompt_token_len": clone_prompt_token_len,
# }
# prompt_speaker = clone_dict['speaker']
# # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n")
# else:
# with open(f"{speaker_path}/speakers_info.json", "r") as f:
# speakers_info = json.load(f)
# for speaker_id, prompt_text in speakers_info.items():
# if speaker_id == prompt_speaker:
# prompt_wav_path = f"{speaker_path}/{speaker_id}_prompt.wav"
# waveform, sample_rate = torchaudio.load(prompt_wav_path)
# audio = {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate}
# prompt_code, prompt_token, prompt_token_len, speech_feat, speech_feat_len, speech_embedding = (
# self.preprocess_prompt_wav(audio, cosy_model)
# )
# prompt_speaker_info = {
# "prompt_text": prompt_text,
# "prompt_code": prompt_code,
# "cosy_speech_feat": speech_feat.to(torch.bfloat16),
# "cosy_speech_feat_len": speech_feat_len,
# "cosy_speech_embedding": speech_embedding.to(torch.bfloat16),
# "cosy_prompt_token": prompt_token,
# "cosy_prompt_token_len": prompt_token_len,
# }
# # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n")
# break
# elif prompt_speaker not in speakers_info.keys():
# raise ValueError("There is no such speaker")
# print("指定文本: ", text, "的说话者是: ", prompt_speaker, end="\n\n")
# cosy_model, prompt_speaker_info, history = self.data_preprocess()
_prefix_tokens = self.autotokenizer.encode("\n")
target_token_encode = self.autotokenizer.encode("\n" + text)
@@ -482,8 +427,16 @@ class StepAudioClone:
audio_tensor = torch.cat(audio_data, dim=1).unsqueeze(0).float()
return ({"waveform": audio_tensor, "sample_rate": sr},)
from MWAudioRecorder import AudioRecorder
NODE_CLASS_MAPPINGS = {
"StepAudioRun": StepAudioRun,
"StepAudioClone": StepAudioClone,
"AudioRecorder": AudioRecorder
}
NODE_DISPLAY_NAME_MAPPINGS = {
"StepAudioRun": "Step Audio Run",
"StepAudioClone": "Step Audio Clone",
"AudioRecorder": "MW Audio Recorder"
}
+2 -2
View File
@@ -4,6 +4,6 @@ import os
current_dir = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, current_dir)
from StepAudioTTS import NODE_CLASS_MAPPINGS
from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS
__all__ = ["NODE_CLASS_MAPPINGS"]
__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"]
Binary file not shown.

After

Width:  |  Height:  |  Size: 27 KiB

+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "stepaudiotts_mw"
description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice."
version = "1.1.2"
version = "1.1.3"
license = {file = "LICENSE"}
dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"]