diff --git a/README.md b/README.md index 0099c69..38400d4 100644 --- a/README.md +++ b/README.md @@ -18,6 +18,8 @@ workflows, especially when dealing with multiple audio inputs or outputs. - [9. Audio Blend](#9-audio-blend) - [10. Audio Test Signal Generator](#10-audio-test-signal-generator) - [11. Audio Musical Note](#11-audio-musical-note) + - [12. Audio Join 2 Channels](#12-audio-join-2-channels) + - [13. Audio Split 2 Channels](#13-audio-split-2-channels) - [🚀 Installation](#-installation) - [📦 Dependencies](#-dependencies) - [🖼️ Examples](#️-examples) @@ -195,6 +197,32 @@ workflows, especially when dealing with multiple audio inputs or outputs. - **Output:** - `frequency` (FLOAT): The calculated frequency of the note in Hz. +### 12. Audio Join 2 Channels + - **Display Name:** `Audio Join 2 Channels` + - **Internal Name:** `SET_AudioJoin2Channels` + - **Category:** `audio/manipulation` + - **Description:** Combines two separate audio inputs into a single stereo audio output, treating the first input as the left channel and the second as the right. It intelligently handles misaligned inputs. + - **Inputs:** + - `audio_left` (AUDIO): The audio signal for the left channel. If it's stereo or multi-channel, it will be automatically converted to mono before being used. + - `audio_right` (AUDIO): The audio signal for the right channel. Will also be converted to mono. + - **Output:** + - `audio_out` (AUDIO): A stereo audio signal. + - **Behavior Details:** + - **Channel Conversion:** Both `audio_left` and `audio_right` are first forced into mono to ensure they each represent a single channel stream. + - **Alignment:** The two mono signals are then aligned to have the same sample rate and length, using the same logic as the "Batch Audios" node (resamples to match `audio_left`'s SR, pads to match the longest duration). + - **Batch Handling:** If the inputs have different batch sizes, the last item of the shorter batch is repeated to match the length of the longer batch. + +### 13. Audio Split 2 Channels + - **Display Name:** `Audio Split 2 Channels` + - **Internal Name:** `SET_AudioSplit2Channels` + - **Category:** `audio/manipulation` + - **Description:** Takes a stereo audio input and separates it into two mono audio outputs, one for the left channel and one for the right. + - **Inputs:** + - `audio` (AUDIO): The stereo audio signal to be split. **The node will raise an error if the input is not 2-channel stereo.** + - **Outputs:** + - `audio_left` (AUDIO): A mono audio signal containing only the left channel data. + - `audio_right` (AUDIO): A mono audio signal containing only the right channel data. + ## 🚀 Installation You can install the nodes from the ComfyUI nodes manager, the name is *Audio Batch*, or just do it manually: diff --git a/example_workflows/join_and_split.jpg b/example_workflows/join_and_split.jpg new file mode 100644 index 0000000..709c176 Binary files /dev/null and b/example_workflows/join_and_split.jpg differ diff --git a/example_workflows/join_and_split.json b/example_workflows/join_and_split.json new file mode 100644 index 0000000..b398141 --- /dev/null +++ b/example_workflows/join_and_split.json @@ -0,0 +1 @@ +{"id":"7a6f4421-11da-41f5-aa24-226ec984ac27","revision":0,"last_node_id":11,"last_link_id":7,"nodes":[{"id":2,"type":"SET_AudioTestSignalGenerator","pos":[1173.3482666015625,1591.795654296875],"size":[275.0025329589844,322],"flags":{},"order":0,"mode":0,"inputs":[{"localized_name":"waveform_type","name":"waveform_type","type":"COMBO","widget":{"name":"waveform_type"},"link":null},{"localized_name":"frequency","name":"frequency","type":"FLOAT","widget":{"name":"frequency"},"link":null},{"localized_name":"frequency_end","name":"frequency_end","type":"FLOAT","widget":{"name":"frequency_end"},"link":null},{"localized_name":"amplitude","name":"amplitude","type":"FLOAT","widget":{"name":"amplitude"},"link":null},{"localized_name":"dc_offset","name":"dc_offset","type":"FLOAT","widget":{"name":"dc_offset"},"link":null},{"localized_name":"phase","name":"phase","type":"FLOAT","widget":{"name":"phase"},"link":null},{"localized_name":"duration","name":"duration","type":"STRING","widget":{"name":"duration"},"link":null},{"localized_name":"sample_rate","name":"sample_rate","type":"INT","widget":{"name":"sample_rate"},"link":null},{"localized_name":"batch_size","name":"batch_size","type":"INT","widget":{"name":"batch_size"},"link":null},{"localized_name":"channels","name":"channels","type":"INT","widget":{"name":"channels"},"link":null},{"localized_name":"seed","name":"seed","shape":7,"type":"INT","widget":{"name":"seed"},"link":null}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[3]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"c5e110e58b6bb0f5c777084a1edcadb7b86eb5b7","Node name for S&R":"SET_AudioTestSignalGenerator"},"widgets_values":["sweep",880,440,0.5,0,0,"3.0",44100,1,1,735791842302287,"randomize"],"color":"#232","bgcolor":"#353"},{"id":5,"type":"MarkdownNote","pos":[908.039794921875,1221.835693359375],"size":[233.68882751464844,106.25144958496094],"flags":{},"order":1,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["# A tone going up in frequency for the left channel"],"color":"#432","bgcolor":"#653"},{"id":6,"type":"MarkdownNote","pos":[918.7133178710938,1593.033203125],"size":[233.68882751464844,106.25144958496094],"flags":{},"order":2,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["# A tone going down in frequency for the right channel"],"color":"#432","bgcolor":"#653"},{"id":1,"type":"SET_AudioTestSignalGenerator","pos":[1171.8756103515625,1218.4552001953125],"size":[275.0025329589844,322],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"waveform_type","name":"waveform_type","type":"COMBO","widget":{"name":"waveform_type"},"link":null},{"localized_name":"frequency","name":"frequency","type":"FLOAT","widget":{"name":"frequency"},"link":null},{"localized_name":"frequency_end","name":"frequency_end","type":"FLOAT","widget":{"name":"frequency_end"},"link":null},{"localized_name":"amplitude","name":"amplitude","type":"FLOAT","widget":{"name":"amplitude"},"link":null},{"localized_name":"dc_offset","name":"dc_offset","type":"FLOAT","widget":{"name":"dc_offset"},"link":null},{"localized_name":"phase","name":"phase","type":"FLOAT","widget":{"name":"phase"},"link":null},{"localized_name":"duration","name":"duration","type":"STRING","widget":{"name":"duration"},"link":null},{"localized_name":"sample_rate","name":"sample_rate","type":"INT","widget":{"name":"sample_rate"},"link":null},{"localized_name":"batch_size","name":"batch_size","type":"INT","widget":{"name":"batch_size"},"link":null},{"localized_name":"channels","name":"channels","type":"INT","widget":{"name":"channels"},"link":null},{"localized_name":"seed","name":"seed","shape":7,"type":"INT","widget":{"name":"seed"},"link":null}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[2]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"c5e110e58b6bb0f5c777084a1edcadb7b86eb5b7","Node name for S&R":"SET_AudioTestSignalGenerator"},"widgets_values":["sweep",440,880,0.5,0,0,"3.0",44100,1,1,811796107259925,"randomize"],"color":"#232","bgcolor":"#353"},{"id":4,"type":"SET_AudioJoin2Channels","pos":[1534.503662109375,1465.3570556640625],"size":[179.99569702148438,46],"flags":{},"order":6,"mode":0,"inputs":[{"localized_name":"audio_left","name":"audio_left","type":"AUDIO","link":2},{"localized_name":"audio_right","name":"audio_right","type":"AUDIO","link":3}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[4,5]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"c5e110e58b6bb0f5c777084a1edcadb7b86eb5b7","Node name for S&R":"SET_AudioJoin2Channels"},"color":"#323","bgcolor":"#535"},{"id":3,"type":"PreviewAudio","pos":[1804.0137939453125,1533.0361328125],"size":[270,88],"flags":{},"order":7,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":4},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.43","Node name for S&R":"PreviewAudio"},"widgets_values":[],"color":"#222","bgcolor":"#000"},{"id":8,"type":"SET_AudioSplit2Channels","pos":[1885.25048828125,1428.18798828125],"size":[181.54745483398438,46],"flags":{},"order":8,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":5}],"outputs":[{"localized_name":"audio_left","name":"audio_left","type":"AUDIO","links":[6]},{"localized_name":"audio_right","name":"audio_right","type":"AUDIO","links":[7]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"c5e110e58b6bb0f5c777084a1edcadb7b86eb5b7","Node name for S&R":"SET_AudioSplit2Channels"},"color":"#323","bgcolor":"#535"},{"id":9,"type":"PreviewAudio","pos":[2175.8046875,1371.262451171875],"size":[270,88],"flags":{},"order":9,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":6},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.43","Node name for S&R":"PreviewAudio"},"widgets_values":[],"color":"#222","bgcolor":"#000"},{"id":10,"type":"PreviewAudio","pos":[2176.989501953125,1520.69091796875],"size":[270,88],"flags":{},"order":10,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":7},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.43","Node name for S&R":"PreviewAudio"},"widgets_values":[],"color":"#222","bgcolor":"#000"},{"id":7,"type":"MarkdownNote","pos":[1523.5401611328125,1565.7567138671875],"size":[214.71388244628906,88],"flags":{},"order":4,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["# Join both signals"],"color":"#432","bgcolor":"#653"},{"id":11,"type":"MarkdownNote","pos":[1876.94873046875,1281.13232421875],"size":[214.71388244628906,88],"flags":{},"order":5,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["# Split them"],"color":"#432","bgcolor":"#653"}],"links":[[2,1,0,4,0,"AUDIO"],[3,2,0,4,1,"AUDIO"],[4,4,0,3,0,"AUDIO"],[5,4,0,8,0,"AUDIO"],[6,8,0,9,0,"AUDIO"],[7,8,1,10,0,"AUDIO"]],"groups":[],"config":{},"extra":{"ds":{"scale":0.8432169278238191,"offset":[-795.3761557698062,-955.0005895799211]}},"version":0.4} \ No newline at end of file diff --git a/source/nodes/nodes_audio.py b/source/nodes/nodes_audio.py index f273ee6..3dc1119 100644 --- a/source/nodes/nodes_audio.py +++ b/source/nodes/nodes_audio.py @@ -659,3 +659,118 @@ class AudioMusicalNote: # A UI warning could also be sent if this were a generator. logger.error(f"Error parsing note: {e}. Defaulting to 440.0 Hz.") return (440.0,) + + +class AudioJoin2Channels: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio_left": ("AUDIO", {"tooltip": "The audio signal for the left channel. Will be converted to mono."}), + "audio_right": ("AUDIO", {"tooltip": "The audio signal for the right channel. Will be converted to mono."}), + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_out",) + FUNCTION = "join_channels" + CATEGORY = BASE_CATEGORY + "/" + MANIPULATION_CATEGORY + DESCRIPTION = "Joins two audio signals (L/R) into a single stereo audio signal." + UNIQUE_NAME = "SET_AudioJoin2Channels" + DISPLAY_NAME = "Audio Join 2 Channels" + + def join_channels(self, audio_left: dict, audio_right: dict): + # 1. Force both inputs to be mono to ensure they represent single channels. + # We reuse the logic from AudioChannelConverter. + channel_converter = AudioChannelConverter() + (mono_left_audio,) = channel_converter.convert_channels(audio_left, "force_mono") + (mono_right_audio,) = channel_converter.convert_channels(audio_right, "force_mono") + + # 2. Align the two mono signals in terms of sample rate and length. + # This reuses the robust logic from AudioBatchAligner. + # The output of the aligner will have channels=1 since both inputs are mono. + aligner = AudioBatchAligner(mono_left_audio, mono_right_audio) + aligned_left_wf, aligned_right_wf, target_sr = aligner.get_aligned_waveforms() + + # aligned_left_wf is (B_left, 1, N), aligned_right_wf is (B_right, 1, N) + + # 3. Handle mismatched batch sizes. + b_left, b_right = aligned_left_wf.shape[0], aligned_right_wf.shape[0] + target_batch_size = max(b_left, b_right) + + # Create final waveforms with the target batch size, repeating the last item if needed. + final_left_wf = aligned_left_wf + if b_left < target_batch_size: + last_left_item = aligned_left_wf[-1:, :, :] # (1, 1, N) + repeats_needed = target_batch_size - b_left + final_left_wf = torch.cat([aligned_left_wf, last_left_item.repeat(repeats_needed, 1, 1)], dim=0) + + final_right_wf = aligned_right_wf + if b_right < target_batch_size: + last_right_item = aligned_right_wf[-1:, :, :] # (1, 1, N) + repeats_needed = target_batch_size - b_right + final_right_wf = torch.cat([aligned_right_wf, last_right_item.repeat(repeats_needed, 1, 1)], dim=0) + + # At this point, final_left_wf and final_right_wf are both (target_batch_size, 1, N) + + # 4. Concatenate the mono channels into a stereo signal. + # `torch.cat` along the channel dimension (dim=1). + stereo_waveform = torch.cat((final_left_wf, final_right_wf), dim=1) # (B, 2, N) + + logger.info(f"Joined L/R channels into stereo audio. Final shape: {stereo_waveform.shape}, SR: {target_sr}") + + output_audio = { + "waveform": stereo_waveform, + "sample_rate": target_sr + } + return (output_audio,) + + +class AudioSplit2Channels: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio": ("AUDIO", {"tooltip": "A stereo audio signal to split into separate channels."}), + }, + } + + RETURN_TYPES = ("AUDIO", "AUDIO") + RETURN_NAMES = ("audio_left", "audio_right") + FUNCTION = "split_channels" + CATEGORY = BASE_CATEGORY + "/" + MANIPULATION_CATEGORY + DESCRIPTION = "Splits a stereo audio signal into two separate mono audio signals (L/R)." + UNIQUE_NAME = "SET_AudioSplit2Channels" + DISPLAY_NAME = "Audio Split 2 Channels" + + def split_channels(self, audio: dict): + waveform = audio['waveform'] # (B, C, N) + sample_rate = audio['sample_rate'] + + num_channels = waveform.shape[1] + + # 1. Validate that the input is stereo. + if num_channels != 2: + msg = f"Input audio must be stereo (2 channels) to be split. Got {num_channels} channels." + logger.error(msg) + # This is a hard requirement, so raising an error is appropriate. + raise ValueError(msg) + + # 2. Slice the tensor along the channel dimension. + # Slicing with [:, 0:1, :] keeps the channel dimension as 1, so the output is (B, 1, N) + left_channel_wf = waveform[:, 0:1, :] + right_channel_wf = waveform[:, 1:2, :] + + logger.info(f"Split stereo audio into L/R channels. Output shape for each: {left_channel_wf.shape}") + + # 3. Package each channel into its own ComfyUI AUDIO dict. + audio_left = { + "waveform": left_channel_wf, + "sample_rate": sample_rate + } + audio_right = { + "waveform": right_channel_wf, + "sample_rate": sample_rate + } + + return (audio_left, audio_right) diff --git a/source/tests/test_audio_join_split.py b/source/tests/test_audio_join_split.py new file mode 100644 index 0000000..8d3f71f --- /dev/null +++ b/source/tests/test_audio_join_split.py @@ -0,0 +1,126 @@ +""" +Regression tests for the AudioJoin2Channels and AudioSplit2Channels nodes in ComfyUI-AudioBatch. +""" + +import bootstrap # noqa: F401 +import torch +import pytest +from nodes.nodes_audio import AudioJoin2Channels, AudioSplit2Channels + + +# Helper function +def create_dummy_audio(batch_size, channels, samples, sr, device='cpu'): + waveform = torch.randn(batch_size, channels, samples, device=device, dtype=torch.float32) + return {"waveform": waveform, "sample_rate": sr} + + +@pytest.fixture +def join_node(): + return AudioJoin2Channels() + + +@pytest.fixture +def split_node(): + return AudioSplit2Channels() + + +# --- Tests for AudioJoin2Channels --- + +def test_join_simple(join_node): + """Tests joining two perfectly matched mono signals.""" + sr = 44100 + samples = 1000 + left_audio = create_dummy_audio(1, 1, samples, sr) + right_audio = create_dummy_audio(1, 1, samples, sr) + + (stereo_audio,) = join_node.join_channels(left_audio, right_audio) + + assert stereo_audio['waveform'].shape == (1, 2, samples) + # Check if L/R channels match the original mono inputs + assert torch.allclose(stereo_audio['waveform'][:, 0:1, :], left_audio['waveform']) + assert torch.allclose(stereo_audio['waveform'][:, 1:2, :], right_audio['waveform']) + + +def test_join_converts_inputs_to_mono(join_node): + """Tests that stereo inputs are correctly converted to mono before joining.""" + sr = 44100 + samples = 1000 + # Left input is stereo, Right is mono + left_stereo_in = create_dummy_audio(1, 2, samples, sr) + right_mono_in = create_dummy_audio(1, 1, samples, sr) + + (result_audio,) = join_node.join_channels(left_stereo_in, right_mono_in) + + # Left channel of the output should be the average of the stereo input + expected_left_mono = torch.mean(left_stereo_in['waveform'], dim=1, keepdim=True) + + assert result_audio['waveform'].shape == (1, 2, samples) + assert torch.allclose(result_audio['waveform'][:, 0:1, :], expected_left_mono) + assert torch.allclose(result_audio['waveform'][:, 1:2, :], right_mono_in['waveform']) + + +def test_join_aligns_and_batches(join_node): + """Tests that mismatched SR, length, and batch sizes are handled.""" + # Left: B=2, 1ch, 44.1k SR, 1s long + audio_left = create_dummy_audio(2, 1, 44100, 44100) + # Right: B=3, 2ch, 22.05k SR, 0.5s long + audio_right = create_dummy_audio(3, 2, 11025, 22050) + + (result_audio,) = join_node.join_channels(audio_left, audio_right) + + # Expected output shape: + # Batch size = max(2, 3) = 3 + # Channels = 2 (stereo) + # SR = 44100 (from left) + # Samples = 44100 (from left, since right is 11025*2=22050 after resampling) + assert result_audio['waveform'].shape == (3, 2, 44100) + assert result_audio['sample_rate'] == 44100 + + # Check the repeated last item for the left channel + assert torch.allclose(result_audio['waveform'][1, 0, :], result_audio['waveform'][2, 0, :]) + + +# --- Tests for AudioSplit2Channels --- + +def test_split_simple(split_node): + """Tests splitting a standard stereo signal.""" + sr = 44100 + samples = 1000 + stereo_audio = create_dummy_audio(1, 2, samples, sr) + + (left_out, right_out) = split_node.split_channels(stereo_audio) + + # Check left channel output + assert left_out['sample_rate'] == sr + assert left_out['waveform'].shape == (1, 1, samples) + assert torch.allclose(left_out['waveform'], stereo_audio['waveform'][:, 0:1, :]) + + # Check right channel output + assert right_out['sample_rate'] == sr + assert right_out['waveform'].shape == (1, 1, samples) + assert torch.allclose(right_out['waveform'], stereo_audio['waveform'][:, 1:2, :]) + + +def test_split_with_batch(split_node): + """Tests that splitting preserves the batch dimension.""" + sr = 44100 + samples = 1000 + stereo_batch = create_dummy_audio(5, 2, samples, sr) + + (left_out, right_out) = split_node.split_channels(stereo_batch) + + assert left_out['waveform'].shape == (5, 1, samples) + assert right_out['waveform'].shape == (5, 1, samples) + + +def test_split_raises_error_on_non_stereo(split_node): + """Tests that an error is raised if the input is not stereo.""" + # Test with mono + mono_audio = create_dummy_audio(1, 1, 1000, 44100) + with pytest.raises(ValueError, match="Input audio must be stereo"): + split_node.split_channels(mono_audio) + + # Test with 3 channels + three_ch_audio = create_dummy_audio(1, 3, 1000, 44100) + with pytest.raises(ValueError, match="Input audio must be stereo"): + split_node.split_channels(three_ch_audio)