From ccfe6476e905daee5c17c942917323698e097117 Mon Sep 17 00:00:00 2001 From: "Salvador E. Tropea" Date: Sat, 12 Jul 2025 15:22:27 -0300 Subject: [PATCH] [Added] Audio Cut node --- .gitignore | 1 + README.md | 14 +++++++++ nodes_audio.py | 77 ++++++++++++++++++++++++++++++++++++++++++++------ utils/misc.py | 49 ++++++++++++++++++++++++++++++++ 4 files changed, 133 insertions(+), 8 deletions(-) diff --git a/.gitignore b/.gitignore index 89d1f24..d95843e 100644 --- a/.gitignore +++ b/.gitignore @@ -195,3 +195,4 @@ cython_debug/ *.dst *.epr +.*~ diff --git a/README.md b/README.md index 52d0557..83f47d0 100644 --- a/README.md +++ b/README.md @@ -14,6 +14,7 @@ workflows, especially when dealing with multiple audio inputs or outputs. - [5. Audio Resampler](#5-audio-resampler) - [6. Audio Channel Conv and Resampler](#6-audio-channel-conv-and-resampler) - [7. Audio Information](#7-audio-information) + - [8. Audio Cut](#8-audio-cut) - [🚀 Installation](#-installation) - [📦 Dependencies](#-dependencies) - [🖼️ Examples](#️-examples) @@ -131,6 +132,18 @@ workflows, especially when dealing with multiple audio inputs or outputs. - `num_samples` (INT): How many samples contains the audio. Duratio [s] = `num_samples` / `sample_rate` - `sample_rate` (INT): Sampling frequency, how many samples per second. +### 8. Audio Cut + - **Display Name:** `Audio Cut` + - **Internal Name:** `SET_AudioCut` + - **Category:** `audio/manipulation` + - **Description:** Cuts a portion of the input audio. + - **Inputs:** + - `audio` (AUDIO): The input audio. + - `start_time` (STRING): Starting time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 1:30 is 1 minute 30 seconds. + - `end_time` (STRING): Ending time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 90.5 is 1 minute 30 seconds and 500 ms. + - **Output:** + - `audio_out` (AUDIO): The selected portion of the audio. + ## 🚀 Installation You can install the nodes from the ComfyUI nodes manager, the name is *Audio Batch*, or just do it manually: @@ -196,3 +209,4 @@ Once installed the examples are available in the ComfyUI workflow templates, in ## 🙏 Attributions - Good part of the initial code and this README was generated using Gemini 2.5 Pro. +- Audio Cut is highly based on [audio-separation-nodes-comfyui](https://github.com/christian-byrne/audio-separation-nodes-comfyui/) diff --git a/nodes_audio.py b/nodes_audio.py index 76cbf6d..94d7d93 100644 --- a/nodes_audio.py +++ b/nodes_audio.py @@ -7,11 +7,13 @@ import torch import torchaudio.transforms as T from .utils.logger import main_logger +from .utils.misc import parse_time_to_seconds logger = main_logger BASE_CATEGORY = "audio" BATCH_CATEGORY = "batch" CONV_CATEGORY = "conversion" +MANIPULATION_CATEGORY = "manipulation" def convert_batch_to_stereo_tensor(audio_waveform_mono_batch: torch.Tensor) -> torch.Tensor: @@ -28,8 +30,8 @@ class AudioBatch: def INPUT_TYPES(cls): return { "required": { - "audio1": ("AUDIO",), - "audio2": ("AUDIO",), + "audio1": ("AUDIO", {"tooltip": "The first audio input. Can be a single audio item or a batch"}), + "audio2": ("AUDIO", {"tooltip": "The second audio input. Can be a single audio item or a batch"}), } } @@ -214,17 +216,24 @@ class SelectAudioFromBatch: "min": 0, "max": 0xffffffffffffffff, # Effectively unbounded, but UI might cap "step": 1, - "display": "number" + "display": "number", + "tootip": "The 0-based index of the audio stream to select from the batch" }), "behavior_out_of_range": (["silence_original_length", "silence_fixed_length", "error"], { - "default": "silence_original_length" + "default": "silence_original_length", + "tootip": ("silence_original_length: Output silent audio with the same channel count and duration as " + "items in the original batch.\n" + "silence_fixed_length: Output silent audio with a duration specified by " + " silence_duration_seconds.\n" + "error: Raise an error (which will halt the workflow and display an error in ComfyUI)") }), "silence_duration_seconds": ("FLOAT", { # Only used if behavior is "silence_fixed_length" "default": 1.0, "min": 0.01, "max": 3600.0, # 1 hour "step": 0.1, - "display": "number" + "display": "number", + "tootip": "The duration of the silent audio if `behavior_out_of_range` is set to `silence_fixed_length`" }), } } @@ -296,7 +305,10 @@ class AudioChannelConverter: "required": { "audio": ("AUDIO",), "channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"], - {"default": "keep"}), + {"default": "keep", + "tooltip": "keep: maintain same channels,\n" + "stereo_to_mono/force_mono: 1 channel,\n" + "mono_to_stereo/force_stereo: 2 channels"}), }, } @@ -424,8 +436,12 @@ class AudioProcessAdvanced: "required": { "audio": ("AUDIO",), "channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"], - {"default": "keep"}), - "target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100}), + {"default": "keep", + "tooltip": "keep: maintain same channels,\n" + "stereo_to_mono/force_mono: 1 channel,\n" + "mono_to_stereo/force_stereo: 2 channels"}), + "target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100., + "tooltip": "Output sample rate"}), }, } RETURN_TYPES = ("AUDIO",) @@ -494,3 +510,48 @@ class AudioForceChannels: def force_channels(self, audio: dict, channels: int): return AudioChannelConverter().convert_channels(audio, self.CHANNELS_TO_MODE[channels]) + + +class AudioCut: + """ From https://github.com/christian-byrne/audio-separation-nodes-comfyui/ """ + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio": ("AUDIO",), + "start_time": ("STRING", { + "default": "0:00", + "tooltip": "Start time in HH:MM:SS.ss format, or just a float." + }), + "end_time": ("STRING", { + "default": "1:00", + "tooltip": "End time in HH:MM:SS.ss format, or just a float." + }), + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_out",) + FUNCTION = "cut" + CATEGORY = BASE_CATEGORY + "/" + MANIPULATION_CATEGORY + DESCRIPTION = "Cuts a portion of the input audio. Can use seconds or HH:MM:SS.ss." + UNIQUE_NAME = "SET_AudioCut" + DISPLAY_NAME = "Audio Cut" + + def cut(self, audio: dict, start_time: str, end_time: str): + waveform = audio["waveform"] + sample_rate = audio["sample_rate"] + length = waveform.shape[-1] - 1 + + start_frame = int(parse_time_to_seconds(start_time) * sample_rate) + start_frame = max(0, min(start_frame, length)) + + end_frame = int(parse_time_to_seconds(end_time) * sample_rate) + end_frame = max(0, min(end_frame, length)) + + logger.debug(f"Cutting audio from {start_frame} to {end_frame} (SR: {sample_rate})") + if start_frame > end_frame: + raise ValueError("Audio Cut: Start time must be smaller than end time and both be within the audio length.") + + return ({"waveform": waveform[..., start_frame:end_frame], + "sample_rate": sample_rate},) diff --git a/utils/misc.py b/utils/misc.py index 4945c26..325a225 100644 --- a/utils/misc.py +++ b/utils/misc.py @@ -5,3 +5,52 @@ NODES_NAME = "AudioBatch" NODES_DEBUG_VAR = NODES_NAME.upper() + "_NODES_DEBUG" + + +def parse_time_to_seconds(time_str: str) -> float: + """ + Converts a flexible time string into total seconds. + + Handles multiple formats: + - A raw number of seconds: '123.45' + - Seconds: '45.5' + - Minutes and Seconds: '10:30.5' + - Hours, Minutes, and Seconds: '01:10:30.5' + + Args: + time_str: The string representing time. + + Returns: + The total number of seconds as a float. + + Raises: + ValueError: If the string format is invalid. + """ + if not isinstance(time_str, str): + raise TypeError("Input must be a string.") + + # First, try a direct float conversion for the simplest case. + try: + return float(time_str) + except ValueError: + # If it fails, it contains non-numeric characters, likely ':'. + pass + + # Scalable parsing for HH:MM:SS formats + try: + parts = time_str.split(':') + total_seconds = 0.0 + multiplier = 1 # Starts with seconds + + # Iterate through parts in reverse (from seconds to hours) + for part in reversed(parts): + if not part: # Handles empty parts like in "1::30" + raise ValueError("Empty part in time string") + total_seconds += float(part) * multiplier + multiplier *= 60 # Next multiplier is 60 times bigger + + return total_seconds + except (ValueError, TypeError) as e: + # Re-raise with a clear, user-friendly message. + raise ValueError(f"Invalid time format: '{time_str}'. " + "Expected 'SECONDS', 'MM:SS.ss', or 'HH:MM:SS.ss'.") from e