diff --git a/.flake8 b/.flake8 new file mode 100644 index 0000000..10dbcd5 --- /dev/null +++ b/.flake8 @@ -0,0 +1,8 @@ +# .flake8 +[flake8] +max-line-length = 127 +# extend-ignore = E203, W503 +# Add plugins if you installed them via additional_dependencies: +# select = C,E,F,W,B,B950 # Example: using flake8-bugbear (B, B950) +# docstring-convention = google +# ... other flake8 or plugin settings diff --git a/.gitignore b/.gitignore index 7b004e5..89d1f24 100644 --- a/.gitignore +++ b/.gitignore @@ -174,9 +174,9 @@ cython_debug/ .abstra/ # Visual Studio Code -# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore +# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore # that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore -# and can be added to the global gitignore or merged into this file. However, if you prefer, +# and can be added to the global gitignore or merged into this file. However, if you prefer, # you could uncomment the following to ignore the enitre vscode folder # .vscode/ @@ -191,4 +191,7 @@ cython_debug/ # exclude from AI features like autocomplete and code analysis. Recommended for sensitive data # refer to https://docs.cursor.com/context/ignore-files .cursorignore -.cursorindexingignore \ No newline at end of file +.cursorindexingignore + +*.dst +*.epr diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 0000000..dd05e9e --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,43 @@ +# .pre-commit-config.yaml + +# Optional: Set a minimum pre-commit version +# min_pre_commit_version: '2.9.0' + +repos: +- repo: https://github.com/pre-commit/pre-commit-hooks + rev: v4.6.0 # Use the latest stable version + hooks: + - id: check-yaml + - id: end-of-file-fixer + - id: trailing-whitespace + # - id: check-added-large-files # Optional: good for catching accidental large file commits + +- repo: https://github.com/pycqa/flake8 + rev: 7.0.0 # Use the latest stable version of flake8 + hooks: + - id: flake8 + # Optional: specify arguments for flake8 + # args: ['--max-line-length=88', '--extend-ignore=E203'] + # It's often better to configure flake8 via .flake8, setup.cfg, or pyproject.toml + # so it's consistent whether run via pre-commit or manually. + additional_dependencies: [ + # Add any flake8 plugins you use here, e.g.: + # 'flake8-bugbear', + # 'flake8-comprehensions', + # 'flake8-docstrings', + # 'pep8-naming' + ] + +- repo: https://github.com/codespell-project/codespell + rev: v2.2.6 # Use the latest stable version + hooks: + - id: codespell + args: [ + # "--ignore-words-list=your,custom,words,here", + # Or point to a file: + # "--ignore-words=.codespellignore", + "--skip=*.json,*.lock,*.svg,*.css,*.html", # Skip file types less likely to need it + # "--check-filenames", + # "--check-hidden" + ] + # You can create a .codespellignore file with one word per line for words to ignore. diff --git a/README.md b/README.md index d3866fb..e98e87f 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,126 @@ -# ComfyUI-AudioBatch -Audio batch, resampler and channel adjust nodes for ComfyUI +# ComfyUI Audio Batch & Utility Nodes + +This repository provides a set of custom nodes for ComfyUI focused on audio batching and common audio processing tasks like +channel conversion and resampling. These nodes are designed to help manage and prepare audio data within your ComfyUI +workflows, especially when dealing with multiple audio inputs or outputs. + +## Nodes + +### 1. Batch Audios + - **Display Name:** `Batch Audios` + - **Internal Name:** `SET_AudioBatch` + - **Category:** `audio/batch` + - **Description:** Takes two audio inputs (which can themselves be batches) and combines them into a single, larger audio batch. The node handles differences in sample rate, channel count, and length between the two inputs to produce a unified batch. + - **Inputs:** + - `audio1` (AUDIO): The first audio input. Can be a single audio item or a batch. + - `audio2` (AUDIO): The second audio input. Can be a single audio item or a batch. + - **Output:** + - `audio_batch` (AUDIO): A single audio object where the waveforms from `audio1` and `audio2` are concatenated along the batch dimension. + - **Behavior Details:** + - **Sample Rate:** All audio in the output batch will be resampled to match the sample rate of `audio1`. + - **Channels:** + - If both inputs are mono, the output batch will be mono. + - If one input is mono and the other is stereo, the mono input will be converted to "fake stereo" (by duplicating its channel), and the output batch will be stereo. + - If both inputs are stereo, the output batch will be stereo. + - For more complex multi-channel inputs, it defaults to the maximum channel count of the two inputs with a warning, as advanced downmixing is not performed. + - **Length (Samples):** All audio clips in the output batch will be padded with silence at the end to match the length of the longest clip (after any resampling). + - **Input Batch Handling:** If `audio1` has B1 items and `audio2` has B2 items, the output `audio_batch` will contain B1 + B2 items. + +### 2. Select Audio from Batch + - **Display Name:** `Select Audio from Batch` + - **Internal Name:** `SET_SelectAudioFromBatch` + - **Category:** `audio/batch` + - **Description:** Selects a single audio stream from an input audio batch based on a specified index. Provides options for handling out-of-range indices. + - **Inputs:** + - `audio_batch` (AUDIO): An audio batch (e.g., from the "Batch Audios" node). + - `index` (INT): The 0-based index of the audio stream to select from the batch. + - `behavior_out_of_range` (COMBO): What to do if the `index` is out of range: + - `silence_original_length` (default): Output silent audio with the same channel count and duration as items in the original batch. + - `silence_fixed_length`: Output silent audio with a duration specified by `silence_duration_seconds`. + - `error`: Raise an error (which will halt the workflow and display an error in ComfyUI). + - `silence_duration_seconds` (FLOAT): The duration of the silent audio if `behavior_out_of_range` is set to `silence_fixed_length`. + - **Output:** + - `selected_audio` (AUDIO): The selected audio stream (as a batch of 1) or silent audio if the index was out of range (and behavior was not "error"). + +### 3. Audio Channel Converter + - **Display Name:** `Audio Channel Converter` + - **Internal Name:** `SET_AudioChannelConverter` + - **Category:** `audio/conversion` + - **Description:** Converts the channel layout of an input audio (e.g., mono to stereo, stereo to mono). Handles batches. + - **Inputs:** + - `audio` (AUDIO): The input audio. + - `channel_conversion` (COMBO): The desired channel conversion strategy: + - `keep` (default): No changes are made to the channel count. Logs a warning if input has more than 2 channels. + - `stereo_to_mono`: Converts stereo (or multi-channel) audio to mono by averaging all input channels. If already mono, no change. + - `mono_to_stereo`: Converts mono audio to "fake stereo" by duplicating the mono channel. If already stereo, no change. For multi-channel (>2) inputs, it takes the first channel and duplicates it to create stereo. + - `force_mono`: Always converts the input to mono by averaging all channels, regardless of the original channel count. + - `force_stereo`: + - If input is mono, converts to "fake stereo". + - If input is stereo, no change. + - If input has more than 2 channels, it takes the first channel and duplicates it to create stereo. + - **Output:** + - `audio_out` (AUDIO): The audio with the converted channel layout. The batch size and sample rate are preserved. + +### 4. Audio Resampler + - **Display Name:** `Audio Resampler` + - **Internal Name:** `SET_AudioResampler` + - **Category:** `audio/conversion` + - **Description:** Resamples the input audio to a specified target sample rate using `torchaudio.transforms.Resample`. Handles batches. + - **Inputs:** + - `audio` (AUDIO): The input audio. + - `target_sample_rate` (INT): The desired sample rate in Hz (e.g., 44100, 48000, 16000). If set to 0 or if it matches the original sample rate, resampling is skipped. + - **Output:** + - `audio_out` (AUDIO): The resampled audio. The batch size and channel count are preserved. + +### 5. Audio Channel Conv and Resampler + - **Display Name:** `Audio Channel Conv and Resampler` + - **Internal Name:** `SET_AudioChannelConvResampler` + - **Category:** `audio/conversion` + - **Description:** A convenience node that combines channel conversion and resampling into a single step. + - **Inputs:** + - `audio` (AUDIO): The input audio. + - `channel_conversion` (COMBO): Same options as the "Audio Channel Converter" node. + - `target_sample_rate` (INT): Same options as the "Audio Resampler" node. + - **Output:** + - `audio_out` (AUDIO): The audio after both channel conversion and resampling have been applied. + +## Installation + +1. Clone this repository into your `ComfyUI/custom_nodes/` directory: + ```bash + cd ComfyUI/custom_nodes/ + git clone https://github.com/set-soft/ComfyUI-AudioBatch ComfyUI-AudioBatch + ``` +2. Restart ComfyUI. + +The nodes should then appear under the "audio/batch" and "audio/conversion" categories in the "Add Node" menu. + +## Dependencies + +- PyTorch +- Torchaudio (for resampling and potentially other audio operations) +- NumPy (often used with audio data) + +These are typically already present in a standard ComfyUI environment. + +## Usage Notes + +- **AUDIO Type:** These nodes work with ComfyUI's standard "AUDIO" data type, which is a Python dictionary containing: + - `'waveform'`: A `torch.Tensor` of shape `(batch_size, num_channels, num_samples)`. + - `'sample_rate'`: An `int` representing the sample rate in Hz. +- **Logging:** The nodes use Python's `logging` module. Debug messages can be helpful for understanding the transformations being applied. + You can control log verbosity through ComfyUI's startup arguments (e.g., `--preview-method auto --verbose` for more detailed ComfyUI logs + which might also affect custom node loggers if they are configured to inherit levels). The logger name used is "AudioBatch". + You can force debugging level for these nodes defining the `AUDIOBATCH_NODES_DEBUG` environment variable to `1`. + +## Future Improvements / TODO + +- Add more sophisticated downmixing options for multi-channel audio (e.g., 5.1 to stereo). +- Allow user to choose padding value (e.g., silence, edge, reflect) for length matching in "Batch Audios". +- Option in "Batch Audios" to truncate to shortest instead of padding to longest. +- More options for stereo-to-mono conversion (e.g., take left channel, take right channel). +- If you are interested on them, please open an issue. + +## License + +[GPL-3.0](LICENSE) diff --git a/__init__.py b/__init__.py new file mode 100644 index 0000000..19f2aa7 --- /dev/null +++ b/__init__.py @@ -0,0 +1,26 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 Salvador E. Tropea +# Copyright (c) 2025 Instituto Nacional de Tecnologïa Industrial +# License: GPL-3.0 +# Project: ComfyUI-AudioBatch +from . import nodes_audio +import inspect +import logging +from .utils.misc import NODES_NAME + +init_logger = logging.getLogger(NODES_NAME + ".__init__") + +NODE_CLASS_MAPPINGS = {} +NODE_DISPLAY_NAME_MAPPINGS = {} + +for name, obj in inspect.getmembers(nodes_audio): + if not inspect.isclass(obj) or not hasattr(obj, "INPUT_TYPES"): + continue + assert hasattr(obj, "UNIQUE_NAME"), f"No name for {obj.__name__}" + NODE_CLASS_MAPPINGS[obj.UNIQUE_NAME] = obj + NODE_DISPLAY_NAME_MAPPINGS[obj.UNIQUE_NAME] = obj.DISPLAY_NAME + +init_logger.info(f"Registering {len(NODE_CLASS_MAPPINGS)} node(s).") +init_logger.debug(f"{list(NODE_DISPLAY_NAME_MAPPINGS.values())}") + +__all__ = ['NODE_CLASS_MAPPINGS', 'NODE_DISPLAY_NAME_MAPPINGS'] diff --git a/example_workflows/audio_batch_select_example.jpg b/example_workflows/audio_batch_select_example.jpg new file mode 100644 index 0000000..c10ceb4 Binary files /dev/null and b/example_workflows/audio_batch_select_example.jpg differ diff --git a/example_workflows/audio_batch_select_example.json b/example_workflows/audio_batch_select_example.json new file mode 100644 index 0000000..17d4101 --- /dev/null +++ b/example_workflows/audio_batch_select_example.json @@ -0,0 +1 @@ +{"id":"c006d818-3f65-40be-857c-e20ee6fe6016","revision":0,"last_node_id":12,"last_link_id":6,"nodes":[{"id":1,"type":"LoadAudio","pos":[2029.5302734375,607.4900512695312],"size":[350,150],"flags":{},"order":0,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"COMBO","widget":{"name":"audio"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null},{"localized_name":"choose file to upload","name":"upload","type":"AUDIOUPLOAD","widget":{"name":"upload"},"link":null}],"outputs":[{"localized_name":"AUDIO","name":"AUDIO","type":"AUDIO","links":[1]}],"properties":{"cnr_id":"comfy-core","ver":"0.3.34","widget_ue_connectable":{},"Node name for S&R":"LoadAudio"},"widgets_values":["Palpatine_1.wav",null,null],"color":"#233","bgcolor":"#355"},{"id":2,"type":"LoadAudio","pos":[2030,810],"size":[350,150],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"COMBO","widget":{"name":"audio"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null},{"localized_name":"choose file to upload","name":"upload","type":"AUDIOUPLOAD","widget":{"name":"upload"},"link":null}],"outputs":[{"localized_name":"AUDIO","name":"AUDIO","type":"AUDIO","links":[2]}],"properties":{"cnr_id":"comfy-core","ver":"0.3.34","widget_ue_connectable":{},"Node name for S&R":"LoadAudio"},"widgets_values":["Palpatine_2.wav",null,null],"color":"#233","bgcolor":"#355"},{"id":4,"type":"LoadAudio","pos":[2030,1020],"size":[350,150],"flags":{},"order":2,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"COMBO","widget":{"name":"audio"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null},{"localized_name":"choose file to upload","name":"upload","type":"AUDIOUPLOAD","widget":{"name":"upload"},"link":null}],"outputs":[{"localized_name":"AUDIO","name":"AUDIO","type":"AUDIO","links":[4]}],"properties":{"cnr_id":"comfy-core","ver":"0.3.34","widget_ue_connectable":{},"Node name for S&R":"LoadAudio"},"widgets_values":["aud-sample-vs-1.wav",null,null],"color":"#233","bgcolor":"#355"},{"id":3,"type":"SET_AudioBatch","pos":[2459.5302734375,757.4900512695312],"size":[146.42578125,46],"flags":{},"order":7,"mode":0,"inputs":[{"localized_name":"audio1","name":"audio1","type":"AUDIO","link":1},{"localized_name":"audio2","name":"audio2","type":"AUDIO","link":2}],"outputs":[{"localized_name":"audio_batch","name":"audio_batch","type":"AUDIO","links":[3]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","widget_ue_connectable":{},"Node name for S&R":"SET_AudioBatch"},"color":"#2a363b","bgcolor":"#3f5159"},{"id":5,"type":"SET_AudioBatch","pos":[2650,860],"size":[146.42578125,46],"flags":{},"order":8,"mode":0,"inputs":[{"localized_name":"audio1","name":"audio1","type":"AUDIO","link":3},{"localized_name":"audio2","name":"audio2","type":"AUDIO","link":4}],"outputs":[{"localized_name":"audio_batch","name":"audio_batch","type":"AUDIO","links":[5]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","widget_ue_connectable":{},"Node name for S&R":"SET_AudioBatch"},"color":"#2a363b","bgcolor":"#3f5159"},{"id":10,"type":"MarkdownNote","pos":[2450,570],"size":[210,110],"flags":{},"order":4,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["## This node will create a batch from the first 2"],"color":"#432","bgcolor":"#653"},{"id":11,"type":"MarkdownNote","pos":[2600,960],"size":[210,110],"flags":{},"order":5,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["## You can easily extend the batch"],"color":"#432","bgcolor":"#653"},{"id":12,"type":"MarkdownNote","pos":[2880,680],"size":[210,110],"flags":{},"order":6,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["## Select which audio you need here"],"color":"#432","bgcolor":"#653"},{"id":9,"type":"MarkdownNote","pos":[1770,610],"size":[230,150],"flags":{},"order":3,"mode":0,"inputs":[],"outputs":[],"properties":{},"widgets_values":["# Load 3 audio files\n\nAudios in the batch will have the same number of channels, the same sample rate and the same duration.\n\nThis is all adapted by the `Batch Audio` nodes."],"color":"#432","bgcolor":"#653"},{"id":6,"type":"SET_SelectAudioFromBatch","pos":[2840,860],"size":[350,106],"flags":{},"order":9,"mode":0,"inputs":[{"localized_name":"audio_batch","name":"audio_batch","type":"AUDIO","link":5},{"localized_name":"index","name":"index","type":"INT","widget":{"name":"index"},"link":null},{"localized_name":"behavior_out_of_range","name":"behavior_out_of_range","type":"COMBO","widget":{"name":"behavior_out_of_range"},"link":null},{"localized_name":"silence_duration_seconds","name":"silence_duration_seconds","type":"FLOAT","widget":{"name":"silence_duration_seconds"},"link":null}],"outputs":[{"localized_name":"selected_audio","name":"selected_audio","type":"AUDIO","links":[6]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","widget_ue_connectable":{},"Node name for S&R":"SET_SelectAudioFromBatch"},"widgets_values":[2,"silence_original_length",1],"color":"#2a363b","bgcolor":"#3f5159"},{"id":7,"type":"PreviewAudio","pos":[3220,860],"size":[270,88],"flags":{},"order":10,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":6},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.39","widget_ue_connectable":{},"Node name for S&R":"PreviewAudio"},"widgets_values":[],"color":"#233","bgcolor":"#355"}],"links":[[1,1,0,3,0,"AUDIO"],[2,2,0,3,1,"AUDIO"],[3,3,0,5,0,"AUDIO"],[4,4,0,5,1,"AUDIO"],[5,5,0,6,0,"AUDIO"],[6,6,0,7,0,"AUDIO"]],"groups":[],"config":{},"extra":{"ue_links":[],"links_added_by_ue":[],"ds":{"scale":1.0313326743041535,"offset":[-1630.6237988782925,-363.19559199608597]}},"version":0.4} diff --git a/example_workflows/resample_force_stereo.jpg b/example_workflows/resample_force_stereo.jpg new file mode 100644 index 0000000..b089cce Binary files /dev/null and b/example_workflows/resample_force_stereo.jpg differ diff --git a/example_workflows/resample_force_stereo.json b/example_workflows/resample_force_stereo.json new file mode 100644 index 0000000..2769275 --- /dev/null +++ b/example_workflows/resample_force_stereo.json @@ -0,0 +1 @@ +{"id":"f8820f6e-7280-4a6e-858d-4aa7fd65e8aa","revision":0,"last_node_id":10,"last_link_id":9,"nodes":[{"id":3,"type":"SET_AudioResampler","pos":[2450,450],"size":[270,58],"flags":{},"order":3,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":1},{"localized_name":"target_sample_rate","name":"target_sample_rate","type":"INT","widget":{"name":"target_sample_rate"},"link":null}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[4]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","Node name for S&R":"SET_AudioResampler","widget_ue_connectable":{}},"widgets_values":[44100],"color":"#2a363b","bgcolor":"#3f5159"},{"id":5,"type":"SET_AudioChannelConverter","pos":[2750,450],"size":[270.740234375,58],"flags":{},"order":5,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":4},{"localized_name":"channel_conversion","name":"channel_conversion","type":"COMBO","widget":{"name":"channel_conversion"},"link":null}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[6]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","Node name for S&R":"SET_AudioChannelConverter","widget_ue_connectable":{}},"widgets_values":["force_stereo"],"color":"#2a363b","bgcolor":"#3f5159"},{"id":1,"type":"LoadAudio","pos":[2050,450],"size":[350,150],"flags":{},"order":0,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"COMBO","widget":{"name":"audio"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null},{"localized_name":"choose file to upload","name":"upload","type":"AUDIOUPLOAD","widget":{"name":"upload"},"link":null}],"outputs":[{"localized_name":"AUDIO","name":"AUDIO","type":"AUDIO","links":[1]}],"properties":{"cnr_id":"comfy-core","ver":"0.3.34","Node name for S&R":"LoadAudio","widget_ue_connectable":{}},"widgets_values":["Palpatine_1.wav",null,null],"color":"#233","bgcolor":"#355"},{"id":7,"type":"LoadAudio","pos":[2050,850],"size":[350,150],"flags":{},"order":1,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"COMBO","widget":{"name":"audio"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null},{"localized_name":"choose file to upload","name":"upload","type":"AUDIOUPLOAD","widget":{"name":"upload"},"link":null}],"outputs":[{"localized_name":"AUDIO","name":"AUDIO","type":"AUDIO","links":[7]}],"properties":{"cnr_id":"comfy-core","ver":"0.3.34","Node name for S&R":"LoadAudio","widget_ue_connectable":{}},"widgets_values":["Palpatine_1.wav",null,null],"color":"#233","bgcolor":"#355"},{"id":6,"type":"SaveAudio","pos":[3080,450],"size":[270,112],"flags":{},"order":7,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":6},{"localized_name":"filename_prefix","name":"filename_prefix","type":"STRING","widget":{"name":"filename_prefix"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.39","Node name for S&R":"SaveAudio","widget_ue_connectable":{}},"widgets_values":["audio/ComfyUI"],"color":"#233","bgcolor":"#355"},{"id":9,"type":"SaveAudio","pos":[3080,850],"size":[270,112],"flags":{},"order":6,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":9},{"localized_name":"filename_prefix","name":"filename_prefix","type":"STRING","widget":{"name":"filename_prefix"},"link":null},{"localized_name":"audioUI","name":"audioUI","type":"AUDIO_UI","widget":{"name":"audioUI"},"link":null}],"outputs":[],"properties":{"cnr_id":"comfy-core","ver":"0.3.39","Node name for S&R":"SaveAudio","widget_ue_connectable":{}},"widgets_values":["audio/ComfyUI"],"color":"#233","bgcolor":"#355"},{"id":4,"type":"SET_AudioChannelConvResampler","pos":[2580,850],"size":[327.1470642089844,82],"flags":{},"order":4,"mode":0,"inputs":[{"localized_name":"audio","name":"audio","type":"AUDIO","link":7},{"localized_name":"channel_conversion","name":"channel_conversion","type":"COMBO","widget":{"name":"channel_conversion"},"link":null},{"localized_name":"target_sample_rate","name":"target_sample_rate","type":"INT","widget":{"name":"target_sample_rate"},"link":null}],"outputs":[{"localized_name":"audio_out","name":"audio_out","type":"AUDIO","links":[9]}],"properties":{"aux_id":"set-soft/ComfyUI-AudioBatch","ver":"65e3da3da285b8593e74f138d114ec0b576a22bd","Node name for S&R":"SET_AudioChannelConvResampler","widget_ue_connectable":{}},"widgets_values":["force_stereo",44100],"color":"#2a363b","bgcolor":"#3f5159"},{"id":10,"type":"MarkdownNote","pos":[2060,240],"size":[530,130],"flags":{},"order":2,"mode":0,"inputs":[],"outputs":[],"properties":{"widget_ue_connectable":{}},"widgets_values":["# Forcing channels and sample rate\n\n## These two examples will force stereo output sampled at 44100 Hz"],"color":"#432","bgcolor":"#653"}],"links":[[1,1,0,3,0,"AUDIO"],[4,3,0,5,0,"AUDIO"],[6,5,0,6,0,"AUDIO"],[7,7,0,4,0,"AUDIO"],[9,4,0,9,0,"AUDIO"]],"groups":[],"config":{},"extra":{"ue_links":[],"ds":{"scale":1.0313326743041535,"offset":[-1112.3182273496227,-193.07144659156816]},"links_added_by_ue":[]},"version":0.4} diff --git a/nodes_audio.py b/nodes_audio.py new file mode 100644 index 0000000..05b3932 --- /dev/null +++ b/nodes_audio.py @@ -0,0 +1,450 @@ +# -*- coding: utf-8 -*- +# Copyright (c) 2025 Salvador E. Tropea +# Copyright (c) 2025 Instituto Nacional de Tecnologïa Industrial +# License: GPL-3.0 +# Project: ComfyUI-AudioBatch +# From code generated by Gemini 2.5 Pro +import torch +import torchaudio.transforms as T +from .utils.logger import main_logger + +logger = main_logger +BASE_CATEGORY = "audio" +BATCH_CATEGORY = "batch" +CONV_CATEGORY = "conversion" + + +def convert_batch_to_stereo_tensor(audio_waveform_mono_batch: torch.Tensor) -> torch.Tensor: + """ Converts a batch of mono audio tensors (B, 1, N) to stereo (B, 2, N). """ + if audio_waveform_mono_batch.ndim != 3 or audio_waveform_mono_batch.shape[1] != 1: + # This could also happen if an input was (N) or (1,N) and wasn't unsqueezed to (B,1,N) yet + raise ValueError("Input for stereo conversion must be a batch of mono audio (B, 1, N), " + f"got {audio_waveform_mono_batch.shape}") + return audio_waveform_mono_batch.repeat(1, 2, 1) # (B, 1, N) -> (B, 2, N) + + +class AudioBatch: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio1": ("AUDIO",), + "audio2": ("AUDIO",), + } + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_batch",) + FUNCTION = "batch_audio" + CATEGORY = BASE_CATEGORY + "/" + BATCH_CATEGORY + DESCRIPTION = "Audio batch creator" + UNIQUE_NAME = "SET_AudioBatch" + DISPLAY_NAME = "Batch Audios" + + def _preprocess_waveform_batch( + self, + waveform: torch.Tensor, # (B, C_in, N_in) + original_sr: int, + target_sr: int, + target_channels: int, + target_samples: int, + reference_dtype: torch.dtype, + reference_device: torch.device + ) -> torch.Tensor: + """ + Helper function to resample, change channels, and pad a batch of waveforms. + Returns a tensor of shape (B, target_channels, target_samples). + """ + # Ensure correct device and dtype + processed_wf = waveform.to(device=reference_device, dtype=reference_dtype) + current_batch_size, current_channels, current_samples = processed_wf.shape + + # 1. Resample if necessary + if original_sr != target_sr: + logger.debug(f"Resampling batch from {original_sr} to {target_sr}. Input shape: {processed_wf.shape}") + # Resample expects (..., time) + # For (B, C, N), we can reshape to (B*C, N), resample, then reshape back. + # Or, if T.Resample handles batch dims appropriately (some versions might if C=1 or applied per channel). + # Let's reshape for robustness with T.Resample. + resampler = T.Resample(orig_freq=original_sr, new_freq=target_sr, dtype=reference_dtype).to(reference_device) + + if current_channels == 1: + # Reshape (B, 1, N) to (B, N) for resampler, then unsqueeze back + processed_wf_reshaped = processed_wf.squeeze(1) # (B, N) + resampled_wf_reshaped = resampler(processed_wf_reshaped) # (B, N_new) + processed_wf = resampled_wf_reshaped.unsqueeze(1) # (B, 1, N_new) + else: # Multi-channel (e.g., stereo) + # Resample each channel in the batch separately + # This is more complex if T.Resample doesn't broadcast correctly. + # A common way: permute to (C, B, N), reshape to (C*B, N), resample, reshape back. + # Or loop (less efficient for GPU tensors). + # Let's try reshaping to (B*C, N) + original_shape = processed_wf.shape + processed_wf_flat_batch_channel = processed_wf.reshape(-1, current_samples) # (B*C, N) + resampled_wf_flat = resampler(processed_wf_flat_batch_channel) # (B*C, N_new) + # Reshape back to (B, C, N_new) + processed_wf = resampled_wf_flat.reshape(original_shape[0], original_shape[1], -1) + + current_samples = processed_wf.shape[2] # Update sample count + logger.debug(f"Batch after resampling: shape={processed_wf.shape}") + + # 2. Adjust number of channels + if current_channels == 1 and target_channels == 2: + processed_wf = convert_batch_to_stereo_tensor(processed_wf) # (B, 1, N) -> (B, 2, N) + logger.debug(f"Batch converted to stereo: shape={processed_wf.shape}") + elif current_channels == 2 and target_channels == 1: + # Example: Convert stereo to mono by averaging (can make this an option) + processed_wf = processed_wf.mean(dim=1, keepdim=True) # (B, 2, N) -> (B, 1, N) + logger.debug(f"Batch converted to mono (avg): shape={processed_wf.shape}") + elif current_channels != target_channels: + # Fallback or error for other unsupported channel conversions (e.g. 5.1 to stereo) + # For now, if shapes don't match after mono/stereo adjustment, it might error later. + # A more robust node would handle various conversions or error clearly. + logger.warning(f"Unhandled channel conversion from {current_channels} to {target_channels}." + f" Resulting channels: {processed_wf.shape[1]}") + # If, for example, target is 2, and current is 5, how to downmix? + # For this node's purpose (batching existing audio), major downmixing is out of scope. + # We primarily handle mono <-> stereo alignment. + if processed_wf.shape[1] != target_channels: + raise ValueError(f"Cannot align channels: input has {processed_wf.shape[1]} after initial processing, " + f"target is {target_channels}") + + # 3. Pad length if necessary + if current_samples < target_samples: + padding_needed = target_samples - current_samples + # Pad only the last dimension (samples) + processed_wf = torch.nn.functional.pad(processed_wf, (0, padding_needed)) + logger.debug(f"Batch padded: shape={processed_wf.shape}") + elif current_samples > target_samples: # Should not happen if target_samples is max length + logger.warning(f"Waveform has {current_samples} samples, but target is {target_samples}. " + "This indicates an issue in target_samples calculation.") + # Truncate as a fallback, though logic should prevent this. + processed_wf = processed_wf[..., :target_samples] + + # Final check for shape + if processed_wf.shape[1] != target_channels or processed_wf.shape[2] != target_samples: + raise RuntimeError(f"Internal Preprocessing Error: Waveform shape {processed_wf.shape} " + f"does not match target ({current_batch_size}, {target_channels}, {target_samples}).") + + return processed_wf + + def batch_audio(self, audio1: dict, audio2: dict): + waveform1_orig = audio1['waveform'] # (B1, C1, N1) + sr1 = audio1['sample_rate'] + waveform2_orig = audio2['waveform'] # (B2, C2, N2) + sr2 = audio2['sample_rate'] + + logger.debug(f"Audio1 input: shape={waveform1_orig.shape}, sr={sr1}, dtype={waveform1_orig.dtype}, " + f"device={waveform1_orig.device}") + logger.debug(f"Audio2 input: shape={waveform2_orig.shape}, sr={sr2}, dtype={waveform2_orig.dtype}, " + f"device={waveform2_orig.device}") + + # Use properties of audio1 as the reference for the output batch + # (e.g., device, dtype, and target sample rate) + reference_device = waveform1_orig.device + reference_dtype = waveform1_orig.dtype + target_sr = sr1 # All audio will be resampled to sr1 + + # Determine target number of channels for the batch (max of inputs, ensure mono/stereo alignment) + c1 = waveform1_orig.shape[1] + c2 = waveform2_orig.shape[1] + # If one is mono and other is stereo, output will be stereo. Otherwise, max (handles both mono or both stereo). + if (c1 == 1 and c2 == 2) or (c1 == 2 and c2 == 1) or (c1 == 2 and c2 == 2): + target_channels = 2 + elif c1 == 1 and c2 == 1: + target_channels = 1 + else: + # For other multi-channel counts (e.g. 5.1), this simple logic might not be ideal. + # For now, default to max, but this could be an error or require downmixing node. + logger.warning(f"Complex channel counts detected (C1={c1}, C2={c2}). Defaulting to max channels ({max(c1,c2)}) " + "and hoping for downstream compatibility or further processing. " + "Explicit mono/stereo alignment is preferred for this node.") + target_channels = max(c1, c2) + + # Determine target number of samples (length) for the batch + # First, calculate lengths *after* potential resampling + n1_orig = waveform1_orig.shape[2] + n2_orig = waveform2_orig.shape[2] + + n1_after_resample = n1_orig # No resampling for audio1 as it's the target_sr reference + n2_after_resample = n2_orig + if sr2 != target_sr: + n2_after_resample = int(n2_orig * (target_sr / sr2)) + + target_samples = max(n1_after_resample, n2_after_resample) + logger.debug(f"Unified Batch Params: SR={target_sr}, Channels={target_channels}, Samples={target_samples}") + + # Preprocess both waveform batches + wf1_processed = self._preprocess_waveform_batch(waveform1_orig, sr1, target_sr, target_channels, target_samples, + reference_dtype, reference_device) + wf2_processed = self._preprocess_waveform_batch(waveform2_orig, sr2, target_sr, target_channels, target_samples, + reference_dtype, reference_device) + + # wf1_processed is (B1, target_channels, target_samples) + # wf2_processed is (B2, target_channels, target_samples) + + # Concatenate along the batch dimension (dim=0) + try: + batched_waveform_final = torch.cat((wf1_processed, wf2_processed), dim=0) + except RuntimeError as e: + logger.error(f"Error during final torch.cat: {e}") + logger.error(f"Processed Waveform1 shape: {wf1_processed.shape}, dtype: {wf1_processed.dtype}, " + f"device: {wf1_processed.device}") + logger.error(f"Processed Waveform2 shape: {wf2_processed.shape}, dtype: {wf2_processed.dtype}, " + f"device: {wf2_processed.device}") + raise # Re-raise to make error visible + + logger.info(f"Final batched audio: shape={batched_waveform_final.shape}, sr={target_sr}") + + output_audio = { + "waveform": batched_waveform_final, + "sample_rate": target_sr + } + return (output_audio,) + + +class SelectAudioFromBatch: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio_batch": ("AUDIO",), # Expects {'waveform': (B, C, N), 'sample_rate': int} + "index": ("INT", { + "default": 0, + "min": 0, + "max": 0xffffffffffffffff, # Effectively unbounded, but UI might cap + "step": 1, + "display": "number" + }), + "behavior_out_of_range": (["silence_original_length", "silence_fixed_length", "error"], { + "default": "silence_original_length" + }), + "silence_duration_seconds": ("FLOAT", { # Only used if behavior is "silence_fixed_length" + "default": 1.0, + "min": 0.01, + "max": 3600.0, # 1 hour + "step": 0.1, + "display": "number" + }), + } + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("selected_audio",) + FUNCTION = "select_audio" + CATEGORY = BASE_CATEGORY + "/" + BATCH_CATEGORY + DESCRIPTION = "Selects an audio from a batch" + UNIQUE_NAME = "SET_SelectAudioFromBatch" + DISPLAY_NAME = "Select Audio from Batch" + + def select_audio(self, audio_batch: dict, index: int, behavior_out_of_range: str, silence_duration_seconds: float): + + waveform_batch = audio_batch['waveform'] # (B, C, N) + sample_rate = audio_batch['sample_rate'] + + batch_size, num_channels, num_samples_per_item = waveform_batch.shape + + logger.debug(f"Input audio batch: shape={waveform_batch.shape}, sr={sample_rate}, index={index}") + logger.debug(f"Out of range behavior: {behavior_out_of_range}, silence duration: {silence_duration_seconds}s") + + selected_waveform = None + + if 0 <= index < batch_size: + # Valid index, select the audio + # Slicing with index:index+1 keeps the batch dimension, resulting in (1, C, N) + selected_waveform = waveform_batch[index:index+1, :, :] + logger.info(f"Selected audio at index {index}. Shape: {selected_waveform.shape}") + else: + # Index is out of range + msg = f"Index {index} is out of range for batch of size {batch_size}." + logger.warning(msg) + + if behavior_out_of_range == "error": + raise ValueError(msg) + elif behavior_out_of_range == "silence_original_length": + logger.info(f"Outputting silence with original length: {num_samples_per_item} samples, " + f"{num_channels} channels.") + selected_waveform = torch.zeros((1, num_channels if num_channels > 0 else 1, num_samples_per_item), + dtype=waveform_batch.dtype, device=waveform_batch.device) + elif behavior_out_of_range == "silence_fixed_length": + silence_samples = int(silence_duration_seconds * sample_rate) + if silence_samples <= 0: # Ensure positive sample count + silence_samples = 1 + logger.warning(f"Calculated silence samples is <=0 ({silence_samples} from " + f"{silence_duration_seconds}s). Using 1 sample.") + logger.info(f"Outputting silence with fixed length: {silence_samples} samples, {num_channels} channels.") + selected_waveform = torch.zeros((1, num_channels if num_channels > 0 else 1, silence_samples), + dtype=waveform_batch.dtype, device=waveform_batch.device) + + if selected_waveform is None: # Should not happen if logic above is complete + logger.error("Selected waveform is None unexpectedly. Defaulting to silence.") + selected_waveform = torch.zeros((1, num_channels if num_channels > 0 else 1, 1), + dtype=waveform_batch.dtype, device=waveform_batch.device) + + output_audio = { + "waveform": selected_waveform, # Shape (1, C, N_selected) + "sample_rate": sample_rate + } + + return (output_audio,) + + +class AudioChannelConverter: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio": ("AUDIO",), + "channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"], + {"default": "keep"}), + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_out",) + FUNCTION = "convert_channels" + CATEGORY = BASE_CATEGORY + "/" + CONV_CATEGORY + DESCRIPTION = "Converts audio channels (mono/stereo/multi-channel handling)." + UNIQUE_NAME = "SET_AudioChannelConverter" + DISPLAY_NAME = "Audio Channel Converter" + + def convert_channels(self, audio: dict, channel_conversion: str): + waveform = audio['waveform'] # (B, C, T) + sample_rate = audio['sample_rate'] + + original_batch_size, original_channels, original_samples = waveform.shape + logger.debug(f"Input audio: {original_batch_size}B, {original_channels}C, {original_samples}T @ {sample_rate}Hz") + logger.debug(f"Channel conversion mode: {channel_conversion}") + + output_waveform = waveform.clone() # Start with a copy + + if channel_conversion == "keep": + # No change needed, but log if > 2 channels + if original_channels > 2: + logger.warning(f"Channel mode 'keep': Input has {original_channels} channels. Outputting all " + f"{original_channels} channels.") + # output_waveform remains as is + + elif channel_conversion == "stereo_to_mono" or channel_conversion == "force_mono": + if original_channels == 1: + logger.debug("Input is already mono. No change for stereo_to_mono/force_mono.") + # output_waveform remains as is + else: # Stereo or Multi-channel to Mono + if original_channels > 2 and channel_conversion == "stereo_to_mono": + logger.warning(f"Channel mode 'stereo_to_mono': Input has {original_channels} channels. " + "Averaging all to mono.") + elif channel_conversion == "force_mono": + logger.info(f"Channel mode 'force_mono': Input has {original_channels} channels. Averaging all to mono.") + # Average across the channel dimension (dim=1) + output_waveform = torch.mean(waveform, dim=1, keepdim=True) + logger.debug(f"Converted to mono. New shape: {output_waveform.shape}") + + elif channel_conversion == "mono_to_stereo" or channel_conversion == "force_stereo": + if original_channels == 2: + logger.debug("Input is already stereo. No change for mono_to_stereo/force_stereo.") + # output_waveform remains as is + elif original_channels == 1: # Mono to Stereo + # Duplicate the mono channel + output_waveform = waveform.repeat(1, 2, 1) + logger.debug(f"Converted mono to stereo by duplication. New shape: {output_waveform.shape}") + else: # Multi-channel (>2) to Stereo + if channel_conversion == "mono_to_stereo": + logger.warning(f"Channel mode 'mono_to_stereo': Input has {original_channels} channels. " + "Taking the first channel and duplicating it to create stereo.") + elif channel_conversion == "force_stereo": + logger.info(f"Channel mode 'force_stereo': Input has {original_channels} channels. " + "Taking the first channel and duplicating it to create stereo.") + output_waveform = waveform[:, 0:1, :].repeat(1, 2, 1) # Take first channel, make it (B,1,T), then repeat + logger.debug(f"Converted {original_channels}ch to stereo. New shape: {output_waveform.shape}") + else: + logger.error(f"Unknown channel_conversion mode: {channel_conversion}. Returning original.") + # output_waveform remains as is (original waveform) + + processed_audio_dict = {"waveform": output_waveform, "sample_rate": sample_rate} + return (processed_audio_dict,) + + +class AudioResampler: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio": ("AUDIO",), + "target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100}), + }, + } + + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_out",) + FUNCTION = "resample_audio" + CATEGORY = BASE_CATEGORY + "/" + CONV_CATEGORY + DESCRIPTION = "Resamples audio to a target sample rate using torchaudio." + UNIQUE_NAME = "SET_AudioResampler" + DISPLAY_NAME = "Audio Resampler" + + def resample_audio(self, audio: dict, target_sample_rate: int): + waveform = audio['waveform'] # (B, C, T) + original_sample_rate = audio['sample_rate'] + + logger.debug(f"Input audio SR: {original_sample_rate}Hz, Target SR: {target_sample_rate}Hz") + + if target_sample_rate == 0 or target_sample_rate == original_sample_rate: + logger.info(f"Target sample rate ({target_sample_rate}Hz) is 0 or matches original ({original_sample_rate}Hz). " + "Skipping resampling.") + return (audio,) # Return original audio dict + + try: + # Ensure waveform is on the CPU for torchaudio transforms if it might be on GPU + # Though many torchaudio ops support GPU tensors. Resample does. + # For consistency or if issues arise: + # device = waveform.device + # waveform_cpu = waveform.cpu() + # resampler = T.Resample(orig_freq=original_sample_rate, new_freq=target_sample_rate).to(device) + # resampled_waveform = resampler(waveform) + resampler = T.Resample(orig_freq=original_sample_rate, new_freq=target_sample_rate, + dtype=waveform.dtype # Preserve dtype + ).to(waveform.device) # Perform resampling on the tensor's current device + + resampled_waveform = resampler(waveform) + logger.info(f"Resampled audio from {original_sample_rate}Hz to {target_sample_rate}Hz. " + f"Original shape: {waveform.shape}, New shape: {resampled_waveform.shape}") + processed_audio_dict = {"waveform": resampled_waveform, "sample_rate": target_sample_rate} + return (processed_audio_dict,) + + except Exception as e: + logger.error(f"Error during resampling: {e}", exc_info=True) + # Fallback: return original audio if resampling fails + return (audio,) + + +class AudioProcessAdvanced: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "audio": ("AUDIO",), + "channel_conversion": (["keep", "stereo_to_mono", "mono_to_stereo", "force_mono", "force_stereo"], + {"default": "keep"}), + "target_sample_rate": ("INT", {"default": 0, "min": 0, "max": 192000, "step": 100}), + }, + } + RETURN_TYPES = ("AUDIO",) + RETURN_NAMES = ("audio_out",) + FUNCTION = "process_audio" + CATEGORY = BASE_CATEGORY + "/" + CONV_CATEGORY + DESCRIPTION = "Applies channel conversion and resampling to audio." + UNIQUE_NAME = "SET_AudioChannelConvResampler" + DISPLAY_NAME = "Audio Channel Conv and Resampler" + + def process_audio(self, audio: dict, channel_conversion: str, target_sample_rate: int): + # Instantiate helper classes (or move their logic directly here) + channel_converter_node = AudioChannelConverter() + resampler_node = AudioResampler() + + # 1. Channel Conversion + (audio_after_channels,) = channel_converter_node.convert_channels(audio, channel_conversion) + + # 2. Resampling + (audio_after_resample,) = resampler_node.resample_audio(audio_after_channels, target_sample_rate) + + return (audio_after_resample,) diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..353d544 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,13 @@ +[project] +name = "audio-batch" +description = "Audio batch creation, extraction, resample, mono and stereo conversion." +version = "1.0.0" +license = { file = "LICENSE" } +dependencies = [] + +[project.urls] +Repository = "https://github.com/set-soft/ComfyUI-AudioBatch" + +[tool.comfy] +PublisherId = "set-soft" +DisplayName = "Audio Batch" diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..76cf8ed --- /dev/null +++ b/requirements.txt @@ -0,0 +1,6 @@ +torch +torchaudio +numpy + +# non essential dependencies: +colorama diff --git a/utils/ansi.py b/utils/ansi.py new file mode 100644 index 0000000..5e7d0cc --- /dev/null +++ b/utils/ansi.py @@ -0,0 +1,113 @@ +# Copyright Jonathan Hartley 2013. BSD 3-Clause license, see LICENSE file. +''' +This module generates ANSI character codes to printing colors to terminals. +See: http://en.wikipedia.org/wiki/ANSI_escape_code +''' +import sys +import os + +CSI = '\033[' +OSC = '\033]' +BEL = '\a' +is_a_tty = sys.stderr.isatty() and os.name == 'posix' + + +def code_to_chars(code): + return CSI + str(code) + 'm' if is_a_tty else '' + + +def set_title(title): + return OSC + '2;' + title + BEL + + +def clear_screen(mode=2): + return CSI + str(mode) + 'J' + + +def clear_line(mode=2): + return CSI + str(mode) + 'K' + + +class AnsiCodes(object): + def __init__(self): + # the subclasses declare class attributes which are numbers. + # Upon instantiation we define instance attributes, which are the same + # as the class attributes but wrapped with the ANSI escape sequence + for name in dir(self): + if not name.startswith('_'): + value = getattr(self, name) + setattr(self, name, code_to_chars(value)) + + +class AnsiCursor(object): + def UP(self, n=1): + return CSI + str(n) + 'A' + + def DOWN(self, n=1): + return CSI + str(n) + 'B' + + def FORWARD(self, n=1): + return CSI + str(n) + 'C' + + def BACK(self, n=1): + return CSI + str(n) + 'D' + + def POS(self, x=1, y=1): + return CSI + str(y) + ';' + str(x) + 'H' + + +class AnsiFore(AnsiCodes): + BLACK = 30 + RED = 31 + GREEN = 32 + YELLOW = 33 + BLUE = 34 + MAGENTA = 35 + CYAN = 36 + WHITE = 37 + RESET = 39 + + # These are fairly well supported, but not part of the standard. + LIGHTBLACK_EX = 90 + LIGHTRED_EX = 91 + LIGHTGREEN_EX = 92 + LIGHTYELLOW_EX = 93 + LIGHTBLUE_EX = 94 + LIGHTMAGENTA_EX = 95 + LIGHTCYAN_EX = 96 + LIGHTWHITE_EX = 97 + + +class AnsiBack(AnsiCodes): + BLACK = 40 + RED = 41 + GREEN = 42 + YELLOW = 43 + BLUE = 44 + MAGENTA = 45 + CYAN = 46 + WHITE = 47 + RESET = 49 + + # These are fairly well supported, but not part of the standard. + LIGHTBLACK_EX = 100 + LIGHTRED_EX = 101 + LIGHTGREEN_EX = 102 + LIGHTYELLOW_EX = 103 + LIGHTBLUE_EX = 104 + LIGHTMAGENTA_EX = 105 + LIGHTCYAN_EX = 106 + LIGHTWHITE_EX = 107 + + +class AnsiStyle(AnsiCodes): + BRIGHT = 1 + DIM = 2 + NORMAL = 22 + RESET_ALL = 0 + + +Fore = AnsiFore() +Back = AnsiBack() +Style = AnsiStyle() +Cursor = AnsiCursor() diff --git a/utils/logger.py b/utils/logger.py new file mode 100644 index 0000000..ce927d5 --- /dev/null +++ b/utils/logger.py @@ -0,0 +1,84 @@ +# Copyright (c) 2025 Salvador E. Tropea +# Copyright (c) 2025 Instituto Nacional de Tecnologïa Industrial +# License: GPL-3.0 +# Project: ComfyUI-AudioBatch +import os +import sys +import logging +from .misc import NODES_NAME, NODES_DEBUG_VAR + +no_colorama = False +try: + from colorama import init as colorama_init, Fore, Back, Style +except ImportError: + no_colorama = True +# If colorama isn't installed use an ANSI basic replacement +if no_colorama: + from .ansi import Fore, Back, Style # noqa: F811 +else: + colorama_init() + + +class CustomFormatter(logging.Formatter): + """Logging Formatter to add colors""" + + def __init__(self): + super(logging.Formatter, self).__init__() + white = Fore.WHITE + Style.BRIGHT + yellow = Fore.YELLOW + Style.BRIGHT + red = Fore.RED + Style.BRIGHT + red_alarm = Fore.RED + Back.WHITE + Style.BRIGHT + cyan = Fore.CYAN + Style.BRIGHT + reset = Style.RESET_ALL + # format = "%(asctime)s - %(name)s - %(levelname)s - %(message)s " + # "(%(filename)s:%(lineno)d)" + format = f"[{NODES_NAME} %(levelname)s] %(message)s (%(name)s - %(filename)s:%(lineno)d)" + format_simple = f"[{NODES_NAME}] %(message)s" + + self.FORMATS = { + logging.DEBUG: cyan + format + reset, + logging.INFO: white + format_simple + reset, + logging.WARNING: yellow + format + reset, + logging.ERROR: red + format + reset, + logging.CRITICAL: red_alarm + format + reset + } + + def format(self, record): + log_fmt = self.FORMATS.get(record.levelno) + formatter = logging.Formatter(log_fmt) + return formatter.format(record) + + +print("Crea el logger") +# Create a new logger +logger = logging.getLogger(NODES_NAME) +logger.propagate = False + +# Add handler if we don't have one. +if not logger.handlers: + handler = logging.StreamHandler(sys.stdout) + handler.setFormatter(CustomFormatter()) + logger.addHandler(handler) + +# ###################### +# Logger setup +# ###################### +# 1. Determine the ComfyUI global log level (influenced by --verbose) +main_logger = logger +comfy_root_logger = logging.getLogger('comfy') +effective_comfy_level = logging.getLogger().getEffectiveLevel() +# 2. Check our custom environment variable for more verbosity +try: + nodes_debug_env = int(os.environ.get(NODES_DEBUG_VAR, "0")) + print(nodes_debug_env) +except ValueError: + nodes_debug_env = 0 +# 3. Set node's logger level +if nodes_debug_env: + main_logger.setLevel(logging.DEBUG - (nodes_debug_env - 1)) + final_level_str = f"DEBUG (due to {NODES_DEBUG_VAR}={nodes_debug_env})" +else: + main_logger.setLevel(effective_comfy_level) + final_level_str = logging.getLevelName(effective_comfy_level) + " (matching ComfyUI global)" +_initial_setup_logger = logging.getLogger(NODES_NAME + ".setup") # A temporary logger for this message +_initial_setup_logger.debug(f"{NODES_NAME} logger level set to: {final_level_str}") diff --git a/utils/misc.py b/utils/misc.py new file mode 100644 index 0000000..4945c26 --- /dev/null +++ b/utils/misc.py @@ -0,0 +1,7 @@ +# Copyright (c) 2025 Salvador E. Tropea +# Copyright (c) 2025 Instituto Nacional de Tecnologïa Industrial +# License: GPL-3.0 +# Project: ComfyUI-AudioBatch + +NODES_NAME = "AudioBatch" +NODES_DEBUG_VAR = NODES_NAME.upper() + "_NODES_DEBUG"