[Added] Audio Cut node
This commit is contained in:
@@ -195,3 +195,4 @@ cython_debug/
|
||||
|
||||
*.dst
|
||||
*.epr
|
||||
.*~
|
||||
|
||||
@@ -14,6 +14,7 @@ workflows, especially when dealing with multiple audio inputs or outputs.
|
||||
- [5. Audio Resampler](#5-audio-resampler)
|
||||
- [6. Audio Channel Conv and Resampler](#6-audio-channel-conv-and-resampler)
|
||||
- [7. Audio Information](#7-audio-information)
|
||||
- [8. Audio Cut](#8-audio-cut)
|
||||
- [🚀 Installation](#-installation)
|
||||
- [📦 Dependencies](#-dependencies)
|
||||
- [🖼️ Examples](#️-examples)
|
||||
@@ -131,6 +132,18 @@ workflows, especially when dealing with multiple audio inputs or outputs.
|
||||
- `num_samples` (INT): How many samples contains the audio. Duratio [s] = `num_samples` / `sample_rate`
|
||||
- `sample_rate` (INT): Sampling frequency, how many samples per second.
|
||||
|
||||
### 8. Audio Cut
|
||||
- **Display Name:** `Audio Cut`
|
||||
- **Internal Name:** `SET_AudioCut`
|
||||
- **Category:** `audio/manipulation`
|
||||
- **Description:** Cuts a portion of the input audio.
|
||||
- **Inputs:**
|
||||
- `audio` (AUDIO): The input audio.
|
||||
- `start_time` (STRING): Starting time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 1:30 is 1 minute 30 seconds.
|
||||
- `end_time` (STRING): Ending time. Can be in the HH:MM:SS.ss format, or be just a float. I.e. 90.5 is 1 minute 30 seconds and 500 ms.
|
||||
- **Output:**
|
||||
- `audio_out` (AUDIO): The selected portion of the audio.
|
||||
|
||||
## 🚀 Installation
|
||||
|
||||
You can install the nodes from the ComfyUI nodes manager, the name is *Audio Batch*, or just do it manually:
|
||||
@@ -196,3 +209,4 @@ Once installed the examples are available in the ComfyUI workflow templates, in
|
||||
## 🙏 Attributions
|
||||
|
||||
- Good part of the initial code and this README was generated using Gemini 2.5 Pro.
|
||||
- Audio Cut is highly based on [audio-separation-nodes-comfyui](https://github.com/christian-byrne/audio-separation-nodes-comfyui/)
|
||||
|
||||
+69
-8
@@ -7,11 +7,13 @@
|
||||
import torch
|
||||
import torchaudio.transforms as T
|
||||
from .utils.logger import main_logger
|
||||
from .utils.misc import parse_time_to_seconds
|
||||
|
||||
logger = main_logger
|
||||
BASE_CATEGORY = "audio"
|
||||
BATCH_CATEGORY = "batch"
|
||||
CONV_CATEGORY = "conversion"
|
||||
MANIPULATION_CATEGORY = "manipulation"
|
||||
|
||||
|
||||
def convert_batch_to_stereo_tensor(audio_waveform_mono_batch: torch.Tensor) -> torch.Tensor:
|
||||
@@ -28,8 +30,8 @@ class AudioBatch:
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"audio1": ("AUDIO",),
|
||||
"audio2": ("AUDIO",),
|
||||
"audio1": ("AUDIO", {"tooltip": "The first audio input. Can be a single audio item or a batch"}),
|
||||
"audio2": ("AUDIO", {"tooltip": "The second audio input. Can be a single audio item or a batch"}),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -214,17 +216,24 @@ class SelectAudioFromBatch:
|
||||
"min": 0,
|
||||
"max": 0xffffffffffffffff, # Effectively unbounded, but UI might cap
|
||||
"step": 1,
|
||||
"display": "number"
|
||||
"display": "number",
|
||||
"tootip": "The 0-based index of the audio stream to select from the batch"
|
||||
}),
|
||||
"behavior_out_of_range": (["silence_original_length", "silence_fixed_length", "error"], {
|
||||
"default": "silence_original_length"
|
||||
"default": "silence_original_length",
|
||||
"tootip": ("silence_original_length: Output silent audio with the same channel count and duration as "
|
||||
"items in the original batch.\n"
|
||||
"silence_fixed_length: Output silent audio with a duration specified by "
|
||||
" silence_duration_seconds.\n"
|
||||
"error: Raise an error (which will halt the workflow and display an error in ComfyUI)")
|
||||
}),
|
||||
"silence_duration_seconds": ("FLOAT", { # Only used if behavior is "silence_fixed_length"
|
||||
"default": 1.0,
|
||||
"min": 0.01,
|
||||
"max": 3600.0, # 1 hour
|
||||
"step": 0.1,
|
||||
"display": "number"
|
||||
"display": "number",
|
||||
"tootip": "The duration of the silent audio if `behavior_out_of_range` is set to `silence_fixed_length`"
|
||||
}),
|
||||
}
|
||||
}
|
||||
@@ -296,7 +305,10 @@ class AudioChannelConverter:
|
||||
"required": {
|
||||
"audio": ("AUDIO",),
|
||||
"channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"],
|
||||
{"default": "keep"}),
|
||||
{"default": "keep",
|
||||
"tooltip": "keep: maintain same channels,\n"
|
||||
"stereo_to_mono/force_mono: 1 channel,\n"
|
||||
"mono_to_stereo/force_stereo: 2 channels"}),
|
||||
},
|
||||
}
|
||||
|
||||
@@ -424,8 +436,12 @@ class AudioProcessAdvanced:
|
||||
"required": {
|
||||
"audio": ("AUDIO",),
|
||||
"channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"],
|
||||
{"default": "keep"}),
|
||||
"target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100}),
|
||||
{"default": "keep",
|
||||
"tooltip": "keep: maintain same channels,\n"
|
||||
"stereo_to_mono/force_mono: 1 channel,\n"
|
||||
"mono_to_stereo/force_stereo: 2 channels"}),
|
||||
"target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100.,
|
||||
"tooltip": "Output sample rate"}),
|
||||
},
|
||||
}
|
||||
RETURN_TYPES = ("AUDIO",)
|
||||
@@ -494,3 +510,48 @@ class AudioForceChannels:
|
||||
|
||||
def force_channels(self, audio: dict, channels: int):
|
||||
return AudioChannelConverter().convert_channels(audio, self.CHANNELS_TO_MODE[channels])
|
||||
|
||||
|
||||
class AudioCut:
|
||||
""" From https://github.com/christian-byrne/audio-separation-nodes-comfyui/ """
|
||||
@classmethod
|
||||
def INPUT_TYPES(cls):
|
||||
return {
|
||||
"required": {
|
||||
"audio": ("AUDIO",),
|
||||
"start_time": ("STRING", {
|
||||
"default": "0:00",
|
||||
"tooltip": "Start time in HH:MM:SS.ss format, or just a float."
|
||||
}),
|
||||
"end_time": ("STRING", {
|
||||
"default": "1:00",
|
||||
"tooltip": "End time in HH:MM:SS.ss format, or just a float."
|
||||
}),
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("AUDIO",)
|
||||
RETURN_NAMES = ("audio_out",)
|
||||
FUNCTION = "cut"
|
||||
CATEGORY = BASE_CATEGORY + "/" + MANIPULATION_CATEGORY
|
||||
DESCRIPTION = "Cuts a portion of the input audio. Can use seconds or HH:MM:SS.ss."
|
||||
UNIQUE_NAME = "SET_AudioCut"
|
||||
DISPLAY_NAME = "Audio Cut"
|
||||
|
||||
def cut(self, audio: dict, start_time: str, end_time: str):
|
||||
waveform = audio["waveform"]
|
||||
sample_rate = audio["sample_rate"]
|
||||
length = waveform.shape[-1] - 1
|
||||
|
||||
start_frame = int(parse_time_to_seconds(start_time) * sample_rate)
|
||||
start_frame = max(0, min(start_frame, length))
|
||||
|
||||
end_frame = int(parse_time_to_seconds(end_time) * sample_rate)
|
||||
end_frame = max(0, min(end_frame, length))
|
||||
|
||||
logger.debug(f"Cutting audio from {start_frame} to {end_frame} (SR: {sample_rate})")
|
||||
if start_frame > end_frame:
|
||||
raise ValueError("Audio Cut: Start time must be smaller than end time and both be within the audio length.")
|
||||
|
||||
return ({"waveform": waveform[..., start_frame:end_frame],
|
||||
"sample_rate": sample_rate},)
|
||||
|
||||
@@ -5,3 +5,52 @@
|
||||
|
||||
NODES_NAME = "AudioBatch"
|
||||
NODES_DEBUG_VAR = NODES_NAME.upper() + "_NODES_DEBUG"
|
||||
|
||||
|
||||
def parse_time_to_seconds(time_str: str) -> float:
|
||||
"""
|
||||
Converts a flexible time string into total seconds.
|
||||
|
||||
Handles multiple formats:
|
||||
- A raw number of seconds: '123.45'
|
||||
- Seconds: '45.5'
|
||||
- Minutes and Seconds: '10:30.5'
|
||||
- Hours, Minutes, and Seconds: '01:10:30.5'
|
||||
|
||||
Args:
|
||||
time_str: The string representing time.
|
||||
|
||||
Returns:
|
||||
The total number of seconds as a float.
|
||||
|
||||
Raises:
|
||||
ValueError: If the string format is invalid.
|
||||
"""
|
||||
if not isinstance(time_str, str):
|
||||
raise TypeError("Input must be a string.")
|
||||
|
||||
# First, try a direct float conversion for the simplest case.
|
||||
try:
|
||||
return float(time_str)
|
||||
except ValueError:
|
||||
# If it fails, it contains non-numeric characters, likely ':'.
|
||||
pass
|
||||
|
||||
# Scalable parsing for HH:MM:SS formats
|
||||
try:
|
||||
parts = time_str.split(':')
|
||||
total_seconds = 0.0
|
||||
multiplier = 1 # Starts with seconds
|
||||
|
||||
# Iterate through parts in reverse (from seconds to hours)
|
||||
for part in reversed(parts):
|
||||
if not part: # Handles empty parts like in "1::30"
|
||||
raise ValueError("Empty part in time string")
|
||||
total_seconds += float(part) * multiplier
|
||||
multiplier *= 60 # Next multiplier is 60 times bigger
|
||||
|
||||
return total_seconds
|
||||
except (ValueError, TypeError) as e:
|
||||
# Re-raise with a clear, user-friendly message.
|
||||
raise ValueError(f"Invalid time format: '{time_str}'. "
|
||||
"Expected 'SECONDS', 'MM:SS.ss', or 'HH:MM:SS.ss'.") from e
|
||||
|
||||
Reference in New Issue
Block a user