[Added] Audio Cut node

This commit is contained in:
Salvador E. Tropea
2025-07-12 15:22:27 -03:00
parent 0f78b6134b
commit ccfe6476e9
4 changed files with 133 additions and 8 deletions
+1
View File
@@ -195,3 +195,4 @@ cython_debug/
*.dst
*.epr
.*~
+14
View File
@@ -14,6 +14,7 @@ workflows, especially when dealing with multiple audio inputs or outputs.
- [5. Audio Resampler](#5-audio-resampler)
- [6. Audio Channel Conv and Resampler](#6-audio-channel-conv-and-resampler)
- [7. Audio Information](#7-audio-information)
- [8. Audio Cut](#8-audio-cut)
- [🚀 Installation](#-installation)
- [📦 Dependencies](#-dependencies)
- [🖼️ Examples](#️-examples)
@@ -131,6 +132,18 @@ workflows, especially when dealing with multiple audio inputs or outputs.
- `num_samples` (INT): How many samples contains the audio. Duratio [s] = `num_samples` / `sample_rate`
- `sample_rate` (INT): Sampling frequency, how many samples per second.
### 8. Audio Cut
- **Display Name:** `Audio Cut`
- **Internal Name:** `SET_AudioCut`
- **Category:** `audio/manipulation`
- **Description:** Cuts a portion of the input audio.
- **Inputs:**
- `audio` (AUDIO): The input audio.
- `start_time` (STRING): Starting time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 1:30 is 1 minute 30 seconds.
- `end_time` (STRING): Ending time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 90.5 is 1 minute 30 seconds and 500 ms.
- **Output:**
- `audio_out` (AUDIO): The selected portion of the audio.
## 🚀 Installation
You can install the nodes from the ComfyUI nodes manager, the name is *Audio Batch*, or just do it manually:
@@ -196,3 +209,4 @@ Once installed the examples are available in the ComfyUI workflow templates, in
## 🙏 Attributions
- Good part of the initial code and this README was generated using Gemini 2.5 Pro.
- Audio Cut is highly based on [audio-separation-nodes-comfyui](https://github.com/christian-byrne/audio-separation-nodes-comfyui/)
+69 -8
View File
@@ -7,11 +7,13 @@
import torch
import torchaudio.transforms as T
from .utils.logger import main_logger
from .utils.misc import parse_time_to_seconds
logger = main_logger
BASE_CATEGORY = "audio"
BATCH_CATEGORY = "batch"
CONV_CATEGORY = "conversion"
MANIPULATION_CATEGORY = "manipulation"
def convert_batch_to_stereo_tensor(audio_waveform_mono_batch: torch.Tensor) -> torch.Tensor:
@@ -28,8 +30,8 @@ class AudioBatch:
def INPUT_TYPES(cls):
return {
"required": {
"audio1": ("AUDIO",),
"audio2": ("AUDIO",),
"audio1": ("AUDIO", {"tooltip": "The first audio input. Can be a single audio item or a batch"}),
"audio2": ("AUDIO", {"tooltip": "The second audio input. Can be a single audio item or a batch"}),
}
}
@@ -214,17 +216,24 @@ class SelectAudioFromBatch:
"min": 0,
"max": 0xffffffffffffffff, # Effectively unbounded, but UI might cap
"step": 1,
"display": "number"
"display": "number",
"tootip": "The 0-based index of the audio stream to select from the batch"
}),
"behavior_out_of_range": (["silence_original_length", "silence_fixed_length", "error"], {
"default": "silence_original_length"
"default": "silence_original_length",
"tootip": ("silence_original_length: Output silent audio with the same channel count and duration as "
"items in the original batch.\n"
"silence_fixed_length: Output silent audio with a duration specified by "
" silence_duration_seconds.\n"
"error: Raise an error (which will halt the workflow and display an error in ComfyUI)")
}),
"silence_duration_seconds": ("FLOAT", { # Only used if behavior is "silence_fixed_length"
"default": 1.0,
"min": 0.01,
"max": 3600.0, # 1 hour
"step": 0.1,
"display": "number"
"display": "number",
"tootip": "The duration of the silent audio if `behavior_out_of_range` is set to `silence_fixed_length`"
}),
}
}
@@ -296,7 +305,10 @@ class AudioChannelConverter:
"required": {
"audio": ("AUDIO",),
"channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"],
{"default": "keep"}),
{"default": "keep",
"tooltip": "keep: maintain same channels,\n"
"stereo_to_mono/force_mono: 1 channel,\n"
"mono_to_stereo/force_stereo: 2 channels"}),
},
}
@@ -424,8 +436,12 @@ class AudioProcessAdvanced:
"required": {
"audio": ("AUDIO",),
"channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"],
{"default": "keep"}),
"target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100}),
{"default": "keep",
"tooltip": "keep: maintain same channels,\n"
"stereo_to_mono/force_mono: 1 channel,\n"
"mono_to_stereo/force_stereo: 2 channels"}),
"target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100.,
"tooltip": "Output sample rate"}),
},
}
RETURN_TYPES = ("AUDIO",)
@@ -494,3 +510,48 @@ class AudioForceChannels:
def force_channels(self, audio: dict, channels: int):
return AudioChannelConverter().convert_channels(audio, self.CHANNELS_TO_MODE[channels])
class AudioCut:
""" From https://github.com/christian-byrne/audio-separation-nodes-comfyui/ """
@classmethod
def INPUT_TYPES(cls):
return {
"required": {
"audio": ("AUDIO",),
"start_time": ("STRING", {
"default": "0:00",
"tooltip": "Start time in HH:MM:SS.ss format, or just a float."
}),
"end_time": ("STRING", {
"default": "1:00",
"tooltip": "End time in HH:MM:SS.ss format, or just a float."
}),
},
}
RETURN_TYPES = ("AUDIO",)
RETURN_NAMES = ("audio_out",)
FUNCTION = "cut"
CATEGORY = BASE_CATEGORY + "/" + MANIPULATION_CATEGORY
DESCRIPTION = "Cuts a portion of the input audio. Can use seconds or HH:MM:SS.ss."
UNIQUE_NAME = "SET_AudioCut"
DISPLAY_NAME = "Audio Cut"
def cut(self, audio: dict, start_time: str, end_time: str):
waveform = audio["waveform"]
sample_rate = audio["sample_rate"]
length = waveform.shape[-1] - 1
start_frame = int(parse_time_to_seconds(start_time) * sample_rate)
start_frame = max(0, min(start_frame, length))
end_frame = int(parse_time_to_seconds(end_time) * sample_rate)
end_frame = max(0, min(end_frame, length))
logger.debug(f"Cutting audio from {start_frame} to {end_frame} (SR: {sample_rate})")
if start_frame > end_frame:
raise ValueError("Audio Cut: Start time must be smaller than end time and both be within the audio length.")
return ({"waveform": waveform[..., start_frame:end_frame],
"sample_rate": sample_rate},)
+49
View File
@@ -5,3 +5,52 @@
NODES_NAME = "AudioBatch"
NODES_DEBUG_VAR = NODES_NAME.upper() + "_NODES_DEBUG"
def parse_time_to_seconds(time_str: str) -> float:
"""
Converts a flexible time string into total seconds.
Handles multiple formats:
- A raw number of seconds: '123.45'
- Seconds: '45.5'
- Minutes and Seconds: '10:30.5'
- Hours, Minutes, and Seconds: '01:10:30.5'
Args:
time_str: The string representing time.
Returns:
The total number of seconds as a float.
Raises:
ValueError: If the string format is invalid.
"""
if not isinstance(time_str, str):
raise TypeError("Input must be a string.")
# First, try a direct float conversion for the simplest case.
try:
return float(time_str)
except ValueError:
# If it fails, it contains non-numeric characters, likely ':'.
pass
# Scalable parsing for HH:MM:SS formats
try:
parts = time_str.split(':')
total_seconds = 0.0
multiplier = 1 # Starts with seconds
# Iterate through parts in reverse (from seconds to hours)
for part in reversed(parts):
if not part: # Handles empty parts like in "1::30"
raise ValueError("Empty part in time string")
total_seconds += float(part) * multiplier
multiplier *= 60 # Next multiplier is 60 times bigger
return total_seconds
except (ValueError, TypeError) as e:
# Re-raise with a clear, user-friendly message.
raise ValueError(f"Invalid time format: '{time_str}'. "
"Expected 'SECONDS', 'MM:SS.ss', or 'HH:MM:SS.ss'.") from e