diff --git a/MWAudioRecorder.py b/MWAudioRecorder.py new file mode 100644 index 0000000..ded4b9b --- /dev/null +++ b/MWAudioRecorder.py @@ -0,0 +1,140 @@ +import numpy as np +import torch +import time +import librosa +import sounddevice as sd +from scipy import ndimage +from comfy.utils import ProgressBar + +class AudioRecorder: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + # 触发控制 + "trigger": ("BOOLEAN", {"default": False}), + # 录音时长 + "record_sec": ("INT", { + "default": 5, + "min": 1, + "max": 60, + "step": 1 # 整数秒递增 + }), + "sample_rate": (["16000", "44100", "48000"], { # 限定标准采样率 + "default": "48000" + }), + "n_fft": ("INT", { # 限定为2的幂次方 + "default": 2048, + "min": 512, + "max": 4096, + "step": 512 # 只能选择512,1024,1536,2048,...4096 + }), + "sensitivity": ("FLOAT", { # 灵敏度精确控制 + "default": 1.2, + "min": 0.5, + "max": 3.0, + "step": 0.1 # 0.1步进 + }), + "smooth": ("INT", { # 确保为奇数 + "default": 5, + "min": 1, + "max": 11, + "step": 2 # 生成1,3,5,7,9,11 + }), + "seed": ("INT", {"default": 0, "min": 0, "max": 0xFFFFFFFFFFFFFFFF}), + } + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio",) + FUNCTION = "record_and_clean" + CATEGORY = "MW-Step-Audio" + + def _stft(self, y, n_fft): + hop = n_fft // 4 + return librosa.stft(y, n_fft=n_fft, hop_length=hop, win_length=n_fft) + + def _istft(self, spec, n_fft): + hop = n_fft // 4 + return librosa.istft(spec, hop_length=hop, win_length=n_fft) + + def _calc_noise_profile(self, noise_clip, n_fft): + noise_spec = self._stft(noise_clip, n_fft) + return { + 'mean': np.mean(np.abs(noise_spec), axis=1, keepdims=True), + 'std': np.std(np.abs(noise_spec), axis=1, keepdims=True) + } + + def _spectral_gate(self, spec, noise_profile, sensitivity): + threshold = noise_profile['mean'] + sensitivity * noise_profile['std'] + return np.where(np.abs(spec) > threshold, spec, 0) + + def _smooth_mask(self, mask, kernel_size): + smoothed = ndimage.uniform_filter(mask, size=(kernel_size, kernel_size)) + return np.clip(smoothed * 1.2, 0, 1) # 增强边缘保留 + + def record_and_clean(self, trigger, record_sec, n_fft, sensitivity, smooth, sample_rate, seed): + if not trigger: + return (None,) + + sr = int(sample_rate) + final_audio = None + + try: + noise_clip = None + # 主录音 + # print(f"开始主录音 {record_sec}秒...") + main_rec = sd.rec(int(record_sec * sr), samplerate=sr, channels=1, dtype='float32') + pb = ProgressBar(record_sec) + for _ in range(record_sec * 2): + time.sleep(0.5) + pb.update(0.5) + sd.wait() + audio = main_rec.flatten() + + # 自动噪声检测 + if noise_clip is None: + # print("自动检测静默段作为噪声参考...") + energy = librosa.feature.rms(y=audio, frame_length=n_fft, hop_length=n_fft//4) + min_idx = np.argmin(energy) + start = min_idx * (n_fft//4) + noise_clip = audio[start:start + n_fft*2] + + # 降噪处理 + # print("进行频谱降噪...") + noise_profile = self._calc_noise_profile(noise_clip, n_fft) + spec = self._stft(audio, n_fft) + + # 多步骤处理 + mask = np.ones_like(spec) # 初始掩膜 + for _ in range(2): # 双重处理循环 + cleaned_spec = self._spectral_gate(spec, noise_profile, sensitivity) + mask = np.where(np.abs(cleaned_spec) > 0, 1, 0) + mask = self._smooth_mask(mask, smooth//2+1) + spec = spec * mask + + # 相位恢复重建 + processed = self._istft(spec * mask, n_fft) + + # 动态增益归一化 + peak = np.max(np.abs(processed)) + processed = processed * (0.99 / peak) if peak > 0 else processed + + # 格式转换 + waveform = torch.from_numpy(processed).float().unsqueeze(0).unsqueeze(0) + final_audio = {"waveform": waveform, "sample_rate": sr} + + except Exception as e: + print(f"Recording/processing failed: {str(e)}") + raise + + return (final_audio,) + +# 节点注册 +NODE_CLASS_MAPPINGS = { + "AudioRecorder": AudioRecorder +} + +NODE_DISPLAY_NAME_MAPPINGS = { + "AudioRecorder": "MW Audio Recorder" +} \ No newline at end of file diff --git a/README-en.md b/README-en.md index 381209f..78247c7 100644 --- a/README-en.md +++ b/README-en.md @@ -6,6 +6,10 @@ ## Update +[2025-03-06]⚒️: New recording node `MW Audio Recorder` can be used to record audio with a microphone, and the progress bar displays the recording progress: + +![](https://github.com/billwuhao/ComfyUI_StepAudioTTS/blob/master/assets/2025-03-06_21-29-09.png) + [2025-03-02]⚒️: Add experimental `custom_mark`, surrounding with "()", for example `(温柔)(东北话)`, it may have an effect. [2025-02-25]⚒️: Support custom speaker `custom_stpeaker`. But please note that the three places in the figure must maintain consistent speaker name. When customizing, the `speaker` will automatically become invalid. diff --git a/README.md b/README.md index e424d09..6f23a0d 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,10 @@ ## 更新 +[2025-03-06]⚒️: 新增录音节点 `MW Audio Recorder` 可用麦克风录制音频, 进度条显示录制进度: + +![](https://github.com/billwuhao/ComfyUI_StepAudioTTS/blob/master/assets/2025-03-06_21-29-09.png) + [2025-03-02]⚒️: 增加实验性的 `custom_mark`, 用 "()" 包围例如 `(温柔)(东北话)`, 它可能会有效. [2025-02-25]⚒️: 支持自定义说话者 `custom_speaker`. 但注意下图三个地方必须保持一致的说话者名称. 自定义时 `speaker` 将自动无效. diff --git a/StepAudioTTS.py b/StepAudioTTS.py index a2a7e2c..f04aab3 100644 --- a/StepAudioTTS.py +++ b/StepAudioTTS.py @@ -85,61 +85,6 @@ class StepAudioTTS: return self._music_cosy_model def __call__(self, text: str, cosy_model, prompt_speaker_info, history): - # instruction_name = self.detect_instruction_name(text) - - # if "RAP" in instruction_name or "哼唱" in instruction_name: - # cosy_model = self.music_cosy_model - # else: - # cosy_model = self.common_cosy_model - - # prompt_speaker_info = {} - # if clone_dict: - # clone_prompt_code, clone_prompt_token, clone_prompt_token_len, clone_speech_feat, clone_speech_feat_len, clone_speech_embedding = ( - # self.preprocess_prompt_wav(clone_dict['audio'], cosy_model) - # ) - # prompt_speaker_info = { - # "prompt_text": clone_dict['prompt_text'], - # "prompt_code": clone_prompt_code, - # "cosy_speech_feat": clone_speech_feat.to(torch.bfloat16), - # "cosy_speech_feat_len": clone_speech_feat_len, - # "cosy_speech_embedding": clone_speech_embedding.to(torch.bfloat16), - # "cosy_prompt_token": clone_prompt_token, - # "cosy_prompt_token_len": clone_prompt_token_len, - # } - - # prompt_speaker = clone_dict['speaker'] - # # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n") - - # else: - # with open(f"{speaker_path}/speakers_info.json", "r") as f: - # speakers_info = json.load(f) - - # for speaker_id, prompt_text in speakers_info.items(): - # if speaker_id == prompt_speaker: - # prompt_wav_path = f"{speaker_path}/{speaker_id}_prompt.wav" - # waveform, sample_rate = torchaudio.load(prompt_wav_path) - # audio = {"waveform": waveform.unsqueeze(0), "sample_rate": sample_rate} - # prompt_code, prompt_token, prompt_token_len, speech_feat, speech_feat_len, speech_embedding = ( - # self.preprocess_prompt_wav(audio, cosy_model) - # ) - # prompt_speaker_info = { - # "prompt_text": prompt_text, - # "prompt_code": prompt_code, - # "cosy_speech_feat": speech_feat.to(torch.bfloat16), - # "cosy_speech_feat_len": speech_feat_len, - # "cosy_speech_embedding": speech_embedding.to(torch.bfloat16), - # "cosy_prompt_token": prompt_token, - # "cosy_prompt_token_len": prompt_token_len, - # } - # # print(prompt_speaker, " 内置文本: ", prompt_speaker_info["prompt_text"], end="\n\n") - # break - - # elif prompt_speaker not in speakers_info.keys(): - # raise ValueError("There is no such speaker") - - # print("指定文本: ", text, "的说话者是: ", prompt_speaker, end="\n\n") - - # cosy_model, prompt_speaker_info, history = self.data_preprocess() _prefix_tokens = self.autotokenizer.encode("\n") target_token_encode = self.autotokenizer.encode("\n" + text) @@ -482,8 +427,16 @@ class StepAudioClone: audio_tensor = torch.cat(audio_data, dim=1).unsqueeze(0).float() return ({"waveform": audio_tensor, "sample_rate": sr},) +from MWAudioRecorder import AudioRecorder NODE_CLASS_MAPPINGS = { "StepAudioRun": StepAudioRun, "StepAudioClone": StepAudioClone, + "AudioRecorder": AudioRecorder } + +NODE_DISPLAY_NAME_MAPPINGS = { + "StepAudioRun": "Step Audio Run", + "StepAudioClone": "Step Audio Clone", + "AudioRecorder": "MW Audio Recorder" +} \ No newline at end of file diff --git a/__init__.py b/__init__.py index 92e9f62..bfdf0d8 100644 --- a/__init__.py +++ b/__init__.py @@ -4,6 +4,6 @@ import os current_dir = os.path.dirname(os.path.abspath(__file__)) sys.path.insert(0, current_dir) -from StepAudioTTS import NODE_CLASS_MAPPINGS +from StepAudioTTS import NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS -__all__ = ["NODE_CLASS_MAPPINGS"] +__all__ = ["NODE_CLASS_MAPPINGS", "NODE_DISPLAY_NAME_MAPPINGS"] diff --git a/assets/2025-03-06_21-29-09.png b/assets/2025-03-06_21-29-09.png new file mode 100644 index 0000000..7ddf97f Binary files /dev/null and b/assets/2025-03-06_21-29-09.png differ diff --git a/pyproject.toml b/pyproject.toml index 9e97cdd..3c3bc2b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "stepaudiotts_mw" description = "A Text To Speech node using Step-Audio-TTS in ComfyUI. Can speak, rap, sing, or clone voice." -version = "1.1.2" +version = "1.1.3" license = {file = "LICENSE"} dependencies = ["transformers>=4.48.3", "openai-whisper>=20231117", "sox>=1.5.0", "hyperpyyaml", "conformer>=0.3.2", "funasr>=1.1.3"]