From e0f6fd1622cf1fe9390beef5543b1be0d222df05 Mon Sep 17 00:00:00 2001 From: noembryo Date: Sat, 22 Aug 2026 00:14:38 +0300 Subject: [PATCH] 1.5.0 --- README.md | 36 ++++ __init__.py | 1 + nodes.py | 12 +- pyproject.toml | 17 +- stitcher.py | 453 +++++++++++++++++-------------------------------- 5 files changed, 208 insertions(+), 311 deletions(-) diff --git a/README.md b/README.md index 7ad5319..bc060a1 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,7 @@ It can load `.json` files from any directory, not just the node's directory. For the custom text integration, there is a variable (can be specified by the user), that can be used in the item's prompt text to insert the custom text anywhere in the body of the prompt. - **UI controls** + - **json_path** is the path to the `.json` file. It can be an absolute path of your hard drive, or a relative path to the ComfyUI's installation directory. - **selected_item** selects one of the `.json` file's items (prompts). @@ -57,6 +58,7 @@ A node lets you load an image from any path in your computer, either by pasting It also displays the image in the node. It's an enhanced version of the "[Load Image From Path (Enhanced)](https://github.com/Chaoses-Ib/ComfyUI_Ib_CustomNodes#load-image-from-path-enhanced)" from [`ComfyUI_Ib_CustomNodes`](ComfyUI_Ib_CustomNodes). +--- ## Resolution Scale ![ResolutionScale](https://raw.githubusercontent.com/noembryo/ComfyUI-noEmbryo/master/stuff/res_scale1.png) A simple node that outputs the resolution of an image using the dimensions of an input image or some custom user-defined dimensions, using a Scale Factor. @@ -68,6 +70,40 @@ If there is an input image connected, setting either `width` or `height` to 0 wi ![RegExChopper](https://raw.githubusercontent.com/noembryo/ComfyUI-noEmbryo/master/stuff/regex_text.png) A node that "chops" a text using a regular expression and outputs the chopped parts of the text. +--- +## H3 Motion Context Clip Stitcher + +![H3MotionContextClipStitcher](https://raw.githubusercontent.com/noembryo/ComfyUI-noEmbryo/master/stuff/H3MotionContextClipStitcher.png) + +Final assembly for [NikoDemon80's H3 Motion Context](https://github.com/NikoDemon80/ComfyUI-H3-Motion-Context) AV clip archives. + +It loads numbered h3_motion_context_av_v1 files (clip_xx.safetensors), decodes one clip at a time to avoid memory peaks, dissolves the overlap between adjacent clips (video + synchronized audio), and concatenates them to a final video and audio stream. +No quality loss, like when trying to concatenate encoded videos. + +- **Inputs** + - Video VAE + - Audio VAE + - (Optional), the current generated AV latent to be added as the last part of the stitching + +- **Controls** + + - **folder** is the folder containing clip_00001.safetensors, clip_00002.safetensors, etc. + Absolute paths and paths relative to ComfyUI/output are accepted. + - **pattern** is the filename glob. The final five-digit number is treated as the clip index. + - **first_clip**is the first approved clip to include. + - **last_clip** is the last clip to include. 0 = every clip from first_clip onward. + - **context_length** is the number of decoded frames to crossfade at each clip boundary. + Keep the same number here as in the H3 Motion Context nodes. + The normal setting is 22 frames. This is the overlap length that is dissolved between adjacent clips. + 5, 22, 39 or 56 are the lengths that are a whole number of latent steps, which is why other numbers aren't offered. + - **fps** is the H3 native output rate. Keep this at 24 unless your workflow deliberately changes it. +- **Outputs** + - **images**: The final stitched image stream to be saved + - **audio**: The final audio stream to be saved + - **frame_count**: The total number of frames + - **report**: Logging of some of the node's actions + + --- ## Auto Save Workflow ![AutoSaveWorkflow](https://raw.githubusercontent.com/noembryo/ComfyUI-noEmbryo/master/stuff/auto_save_workflow.png) diff --git a/__init__.py b/__init__.py index 68fb1b9..d5b8e35 100644 --- a/__init__.py +++ b/__init__.py @@ -26,6 +26,7 @@ NODE_DISPLAY_NAME_MAPPINGS = {} node_list = [ "nodes", # "load_image_from_path", + # "stitcher", ] for module_name in node_list: diff --git a/nodes.py b/nodes.py index 3fb36e6..83cf290 100644 --- a/nodes.py +++ b/nodes.py @@ -4,18 +4,18 @@ from os.path import realpath, join, dirname, isabs, splitext, basename from datetime import datetime import folder_paths from .load_image_from_path import LoadImageFromPathEnhanced +from .stitcher import H3MotionContextClipStitcher MANIFEST = {"name": "noEmbryo Nodes", - "version": (1, 0, 3), + "version": (1, 5, 0), "author": "noEmbryo", "project": "https://github.com/noembryo/ComfyUI-noEmbryo", "description": "Nodes for ComfyUI", "license": "MIT", } __author__ = "noEmbryo" -__version__ = "1.3.1" +__version__ = "1.5.0" -# LISTS_PATH = realpath("./custom_nodes/ComfyUI-noEmbryo/TermLists/") LISTS_PATH = join(dirname(realpath(__file__)), "TermLists") @@ -437,7 +437,9 @@ class AutoSaveWorkflow: CATEGORY = "utils" OUTPUT_NODE = True - def execute(self, trigger, save_directory, filename, prompt=None, extra_pnginfo=None): + # noinspection PyUnusedLocal + @staticmethod + def execute(trigger, save_directory, filename, prompt=None, extra_pnginfo=None): status = "Trigger disabled - workflow not saved" if trigger: @@ -478,6 +480,7 @@ NODE_CLASS_MAPPINGS = {f"JsonPromptLoader -{__author__}": JsonPromptLoader, f"Regex Text Chopper -{__author__}": RegExTextChopper, f"Auto Save Workflow -{__author__}": AutoSaveWorkflow, f"Load Image (from path) -{__author__}": LoadImageFromPathEnhanced, + f"H3MotionContextClipStitcher -{__author__}": H3MotionContextClipStitcher, "PromptTermList1": PromptTermList1, "PromptTermList2": PromptTermList2, "PromptTermList3": PromptTermList3, @@ -491,6 +494,7 @@ NODE_DISPLAY_NAME_MAPPINGS = {f"JsonPromptLoader -{__author__}": f"Json Prompt L f"Regex Text Chopper -{__author__}": f"Regex Text Chopper /{__author__}", f"Auto Save Workflow -{__author__}": f"Auto Save Workflow /{__author__}", f"Load Image (from path) -{__author__}": f"Load Image (from path) /{__author__}", + f"H3MotionContextClipStitcher -{__author__}": f"H3 Motion Context Clip Stitcher /{__author__}", "PromptTermList1": f"PromptTermList 1 /{__author__}", "PromptTermList2": f"PromptTermList 2 /{__author__}", "PromptTermList3": f"PromptTermList 3 /{__author__}", diff --git a/pyproject.toml b/pyproject.toml index 98013cb..10fb86a 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,15 +1,16 @@ [project] name = "comfyui-noembryo" description = """ -A collection of nodes for ComyUI. -* "Load Image (from path)" is a node lets you load an image from any path in your computer, and also displays the image in the node. -* "JsonPromptLoader" is a node that enhances a propt using a .json file with `item:prompt` pairs. -* "Resolution Scale" is a node that resizes an image using pixels or scale factor. -* "Regex Text Chopper" is a node that cuts text based on regex. -* "Auto Save Workflow" can save the current workflow to a .json file automatically. -* "PromptTermList (1-6)" nodes to help with the creation of Prompts. +A collection of nodes for ComyUI.\n +* "Load Image (from path)" is a node lets you load an image from any path in your computer, and also displays the image in the node.\n +* "H3 Motion Context Clip Stitcher" is a node that concatenates H3 Motion Context clip archives.\n +* "JsonPromptLoader" is a node that enhances a propt using a .json file with `item:prompt` pairs.\n +* "Resolution Scale" is a node that resizes an image using pixels or scale factor.\n +* "Regex Text Chopper" is a node that cuts text based on regex.\n +* "PromptTermList (1-6)" nodes to help with the creation of Prompts.\n +* "Auto Save Workflow" can save the current workflow to a .json file automatically.\n """ -version = "1.4.0" +version = "1.5.0" license = {text = "MIT License"} [project.urls] diff --git a/stitcher.py b/stitcher.py index 185bcbf..a3f5a0e 100644 --- a/stitcher.py +++ b/stitcher.py @@ -1,16 +1,14 @@ -"""MiniMax H3 Motion Context archive stitcher for ComfyUI. +"""H3 Motion Context clip stitcher for ComfyUI. -Loads NikoDemon80/ComfyUI-H3-Motion-Context v0.3.x archive files -(h3_motion_context_av_v1), decodes each approved clip once, removes the -carried Motion Context head from clips after the first, and concatenates the -remaining picture/audio into one IMAGE + AUDIO pair. +Loads NikoDemon80/ComfyUI-H3-Motion-Context clip archive files (h3_motion_context_av_v1), +decodes each approved clip once, crossfades the carried Motion Context head from +clips, and concatenates the picture/audio into one IMAGE + AUDIO pair. -This intentionally does NOT reconstruct a NestedTensor and feed the saved -files back into Motion Context. The archive format is the sampler output, -and this node is a final-media assembly tool. +This intentionally does NOT reconstruct a NestedTensor and feed the saved files back +into Motion Context. +The archive format is the sampler output, and this node is a final-media assembly tool. """ -import fnmatch import glob import logging import os @@ -18,7 +16,6 @@ import re import torch import torch.nn.functional as F - import folder_paths try: @@ -34,9 +31,6 @@ except Exception: log_ = logging.getLogger("h3_motion_context_archive") -INDEX_RE = re.compile(r"(?:^|_)(\d{5})(?:\.safetensors)$", re.IGNORECASE) - - def _resolve_folder(path): p = (path or "").strip().strip('"').strip("'") if not p: @@ -45,16 +39,16 @@ def _resolve_folder(path): for c in candidates: if os.path.isdir(c): return os.path.abspath(c) - raise FileNotFoundError( - "H3 Motion Context Archive Stitcher: folder not found: %s\n" - "You can use an absolute path or a path relative to ComfyUI's output folder." - % p - ) + raise FileNotFoundError("H3 Motion Context Clip Stitcher: folder not found: %s\n" + "You can use an absolute path or a path relative to " + "ComfyUI's output folder." % p) def _clip_number(path): name = os.path.basename(path) - m = INDEX_RE.search(name) + # noinspection RegExpUnnecessaryNonCapturingGroup + pat = re.compile(r"(?:^|_)(\d{5})(?:\.safetensors)$", re.IGNORECASE) + m = pat.search(name) return int(m.group(1)) if m else -1 @@ -71,15 +65,14 @@ def _find_files(folder, pattern, first_clip, last_clip): continue if idx < int(first_clip): continue - if int(last_clip) > 0 and idx > int(last_clip): + if 0 < int(last_clip) < idx: continue paths.append((idx, p)) paths.sort(key=lambda x: x[0]) if not paths: - raise FileNotFoundError( - "H3 Motion Context Archive Stitcher: no numbered .safetensors files " - "matched '%s' in %s." % (pattern, folder) - ) + raise FileNotFoundError("H3 Motion Context Clip Stitcher: " + "no numbered .safetensors files matched '%s' in %s." + % (pattern, folder)) # Do not silently skip a missing numbered clip. A gap usually means an # approved clip was not saved, and silently stitching around it would make @@ -87,111 +80,62 @@ def _find_files(folder, pattern, first_clip, last_clip): expected = paths[0][0] for idx, _ in paths: if idx != expected: - raise ValueError( - "H3 Motion Context Archive Stitcher: missing clip %05d between " - "the selected archive files." % expected - ) + raise ValueError("H3 Motion Context Clip Stitcher: missing clip %05d between " + "the selected archive files." % expected) expected += 1 return paths def _load_archive(path): if st_load is None: - raise RuntimeError( - "safetensors is unavailable in this ComfyUI Python environment." - ) + raise RuntimeError("safetensors is unavailable in this " + "ComfyUI Python environment.") + # noinspection PyCallingNonCallable data = st_load(path, device="cpu") if "video" not in data or "audio" not in data: - raise ValueError( - "%s is not an h3_motion_context_av_v1 archive: expected 'video' and 'audio'." - % path - ) + raise ValueError("%s is not an h3_motion_context_av_v1 archive: " + "expected 'video' and 'audio'." % path) video = data["video"] audio = data["audio"] if video.ndim != 5: - raise ValueError("%s: expected video [B,C,T,H,W], got %s" % (path, tuple(video.shape))) + raise ValueError("%s: expected video [B,C,T,H,W], got %s" + % (path, tuple(video.shape))) if audio.ndim != 4: - raise ValueError("%s: expected audio [B,C,2,T], got %s" % (path, tuple(audio.shape))) + raise ValueError("%s: expected audio [B,C,2,T], got %s" + % (path, tuple(audio.shape))) if video.shape[0] != 1 or audio.shape[0] != 1: raise ValueError("%s: only batch size 1 archive clips are supported." % path) return video, audio def _decode_video(vae, video_latent): - """Decode the H3 video stream and normalize to ComfyUI IMAGE format.""" + """ Decode the H3 video stream and normalize to ComfyUI IMAGE format. + """ images = vae.decode(video_latent) # H3's VAE normally returns [B,T,H,W,C]. Some VAE implementations can # return [T,H,W,C], so accept both. if images.ndim == 5: images = images.reshape(-1, *images.shape[-3:]) elif images.ndim != 4: - raise RuntimeError("H3 video VAE returned unexpected shape %s" % (tuple(images.shape),)) + raise RuntimeError("H3 video VAE returned unexpected shape %s" + % (tuple(images.shape),)) return images.to(torch.float32).clamp(0, 1).cpu() -def _decode_audio(audio_vae, audio_latent, normalize=True): - """Decode the H3 audio stream using the same convention as ComfyUI's VAEDecodeAudio.""" +def _decode_audio(audio_vae, audio_latent): + """ Decode the H3 audio stream using the same convention as ComfyUI's VAEDecodeAudio. + """ audio = audio_vae.decode(audio_latent) # Current ComfyUI audio VAE returns [B,L,C]. Convert to [B,C,L]. if audio.ndim != 3: - raise RuntimeError("H3 audio VAE returned unexpected shape %s" % (tuple(audio.shape),)) + raise RuntimeError( + "H3 audio VAE returned unexpected shape %s" % (tuple(audio.shape),)) audio = audio.movedim(-1, 1) - if normalize: - std = torch.std(audio, dim=[1, 2], keepdim=True) * 5.0 - std[std < 1.0] = 1.0 - audio = audio / std sr = int(getattr(audio_vae, "audio_sample_rate_output", getattr(audio_vae, "audio_sample_rate", 32000))) return {"waveform": audio.to(torch.float32).cpu(), "sample_rate": sr} -def _trim_clip(images, audio, trim_frames, fps, match_tail, trim_mode): - """Trim a Motion Context overlap from the requested side of a decoded clip.""" - n = int(trim_frames) - if n <= 0: - return images, audio - total = int(images.shape[0]) - if n >= total: - raise ValueError( - "Cannot trim %d frames from a decoded clip containing %d frames." - % (n, total) - ) - - if trim_mode == "TRIM_BACK": - out_images = images[:-n] - else: - out_images = images[n:] - - if audio is None: - return out_images, None - - waveform = audio["waveform"] - sr = int(audio["sample_rate"]) - cut = int(round((n / float(fps)) * sr)) - if cut >= waveform.shape[-1]: - raise ValueError( - "Audio is too short to remove the %d-frame (%0.4fs) Motion Context " - "%s." % (n, n / float(fps), - "tail" if trim_mode == "TRIM_BACK" else "head") - ) - - if trim_mode == "TRIM_BACK": - waveform = waveform[..., :-cut] - else: - waveform = waveform[..., cut:] - - if match_tail: - frames_left = total - n - want = int(round(frames_left / float(fps) * sr)) - have = int(waveform.shape[-1]) - if have > want: - waveform = waveform[..., :want] - elif have < want: - waveform = F.pad(waveform, (0, want - have)) - - return out_images, {"waveform": waveform, "sample_rate": sr} - - def _resample_audio(audio, target_sr): if audio is None: return None @@ -199,17 +143,16 @@ def _resample_audio(audio, target_sr): if sr == int(target_sr): return audio if torchaudio is None: - raise RuntimeError( - "Audio sample rates differ (%d vs %d), but torchaudio is unavailable " - "to resample them." % (sr, int(target_sr)) - ) + raise RuntimeError("Audio sample rates differ (%d vs %d), but torchaudio is " + "unavailable to resample them." % (sr, int(target_sr))) + # noinspection PyUnresolvedReferences waveform = torchaudio.functional.resample(audio["waveform"], sr, int(target_sr)) return {"waveform": waveform, "sample_rate": int(target_sr)} def _crossfade_boundary(prev_tail_images, cur_images, prev_tail_wave, cur_wave, overlap_frames, cross_samples): - """Crossfade the previous clip's tail with the current clip's head. + """ Crossfade the previous clip's tail with the current clip's head. prev_tail_images: [L,H,W,C] cur_images: [T,H,W,C] prev_tail_wave : [1,C,Ls] cur_wave: [1,C,Cs] (or None) @@ -236,8 +179,7 @@ def _crossfade_boundary(prev_tail_images, cur_images, prev_tail_wave, cur_wave, blend_wave = prev_tail_wave else: n = min(n, int(prev_tail_wave.shape[-1]), int(cur_wave.shape[-1])) - theta = torch.linspace(0.0, 1.5707963267948966, n, - dtype=prev_tail_wave.dtype, + theta = torch.linspace(0.0, 1.5707963267948966, n, dtype=prev_tail_wave.dtype, device=prev_tail_wave.device).view(1, 1, n) blend_wave = (prev_tail_wave[..., :n] * torch.cos(theta) + cur_wave[..., :n] * torch.sin(theta)) @@ -245,192 +187,143 @@ def _crossfade_boundary(prev_tail_images, cur_images, prev_tail_wave, cur_wave, def _av_from_live_latent(latent): - """Extract (video, audio) tensors from an in-memory H3 sampler LATENT, - matching the layout _load_archive() returns from a saved archive. - - NOTE: verify this against your installed Motion Context version before - relying on it -- run the debug snippet in the comment below once to - confirm the real attribute/key names, then adjust this function. + """ Extract (video, audio) tensors from an in-memory H3 AV LATENT, + using the same unpacking convention as NikoDemon80's own + _streams_from_latent()/save(): latent["samples"] is a NestedTensor + (or tuple/list) whose unbind() gives (video, audio) in that order. """ - # --- Debug helper: uncomment once to inspect the real object --- - # _LOG.warning("LIVE LATENT DEBUG: type=%s repr=%s", - # type(latent), - # getattr(latent, "__dict__", None) or - # (latent.keys() if isinstance(latent, dict) else dir(latent))) - - # Case 1: plain dict, e.g. {"samples": video_t, "audio_samples": audio_t} - if isinstance(latent, dict): - log_.info("1. dict") - video = latent.get("samples") or latent.get("video") - audio = latent.get("audio_samples") or latent.get("audio") - if video is not None and audio is not None: - return video, audio - - # Case 2: object with .video / .audio attributes (NestedTensor-style) - if hasattr(latent, "video") and hasattr(latent, "audio"): - log_.info("2. video/audio") - return latent.video, latent.audio - - # Case 3: tuple/list of two tensors (video, audio) - if isinstance(latent, (tuple, list)) and len(latent) == 2: - log_.info("3. tuple/list") - return latent[0], latent[1] - log_.info("4. unknown") - - raise ValueError( - "Could not extract video/audio from the connected live latent " - "(type=%s). Uncomment the debug line above, run once, check the " - "console output, and adjust _av_from_live_latent() accordingly." - % type(latent) - ) + if not isinstance(latent, dict) or "samples" not in latent: + raise ValueError("h3_motion_context: expected a MiniMax H3 AV latent dict with " + "a 'samples' key, got %r" % type(latent)) + samples = latent["samples"] + if hasattr(samples, "unbind"): + parts = list(samples.unbind()) + elif isinstance(samples, (tuple, list)): + parts = list(samples) + else: + raise ValueError("h3_motion_context: expected a MiniMax H3 AV latent (a nested " + "video/audio pair), got %r" % type(samples)) + if len(parts) < 2: + raise ValueError("h3_motion_context: latent has no audio stream; wire the " + "sampler output of an H3 AV graph.") + video = parts[0].cpu().contiguous() + audio = parts[1].cpu().contiguous() + return video, audio -class MiniMaxH3ContextStitcher: - """Load, decode, trim, and concatenate approved H3 Motion Context clips.""" - +class H3MotionContextClipStitcher: + """ Load, decode, and crossfade approved H3 Motion Context clips. + """ @classmethod def INPUT_TYPES(cls): - return { - "required": { - "folder": ("STRING", { - "default": "h3_context", - "tooltip": "Folder containing clip_00001.safetensors, clip_00002.safetensors, etc. " - "Absolute paths and paths relative to ComfyUI/output are accepted." - }), - "pattern": ("STRING", { - "default": "clip_*.safetensors", - "tooltip": "Filename glob. The final five-digit number is treated as the clip index." - }), - "first_clip": ("INT", { - "default": 1, "min": 1, "max": 9999, - "tooltip": "First approved clip to include." - }), - "last_clip": ("INT", { - "default": 0, "min": 0, "max": 9999, - "tooltip": "Last clip to include. 0 = every clip from first_clip onward." - }), - "context_length": ("INT", { - "default": 22, "min": 0, "max": 4096, - "tooltip": "Number of decoded frames to remove at each clip boundary. " - "For NikoDemon80 v0.3.1 the normal setting is 22 frames. " - "In CROSSFADE mode this is the overlap length that is dissolved " - "between adjacent clips instead of being removed." - }), - "trim_mode": (["TRIM_FRONT", "TRIM_BACK", "CROSSFADE"], { - "default": "TRIM_FRONT", - "tooltip": "TRIM_FRONT: remove the first context_length frames from clips 2..N.\n" - "TRIM_BACK: remove the last context_length frames from clips 1..N-1.\n" - "CROSSFADE: keep the context_length overlap and dissolve it between " - "adjacent clips (video + synchronized audio) instead of removing it." - }), - "fps": ("FLOAT", { - "default": 24.0, "min": 1.0, "max": 240.0, "step": 0.001, - "tooltip": "H3 native output rate. Keep this at 24 unless your workflow deliberately changes it." - }), - }, - "optional": { - "video_vae": ("VAE", { - "tooltip": "MiniMax H3 video VAE (FP16 or INT8 ConvRot)." - }), + return {"required": {"folder": ("STRING", {"default": "h3_context", + "tooltip": "Folder containing clip_00001.safetensors, " + "clip_00002.safetensors, etc.\nAbsolute paths and paths relative " + "to ComfyUI/output are accepted."}), + "pattern": ("STRING", {"default": "clip_*.safetensors", + "tooltip": "Filename glob. The final five-digit number is treated as " + "the clip index."}), + "first_clip": ("INT", {"default": 1, "min": 1, "max": 9999, + "tooltip": "First approved clip to include."}), + "last_clip": ("INT", {"default": 0, "min": 0, "max": 9999, + "tooltip": "Last clip to include. 0 = every clip from first_clip onward."}), + "context_length": (["5", "22", "39", "56"], {"default": "22", + "tooltip": "Number of decoded frames to crossfade at each clip boundary. " + "The normal setting is 22 frames.\n" + "This is the overlap length that is dissolved between " + "adjacent clips.\n" + "5, 22, 39 or 56 are the lengths that are a whole number of " + "latent steps, which is why other numbers aren't offered."}), + "fps": ("FLOAT", {"default": 24.0, "min": 1.0, "max": 240.0, "step": 0.001, + "tooltip": "H3 native output rate. Keep this at 24 unless your workflow " + "deliberately changes it."}), }, + "optional": {"video_vae": ("VAE", {"tooltip": "MiniMax H3 video VAE " + "(FP16 or INT8 ConvRot)."}), "audio_vae": ("VAE", { - "tooltip": "MiniMax H3 audio VAE FP32. Required for the AUDIO output." - }), - "match_audio_tail": ("BOOLEAN", { - "default": True, - "tooltip": "After removing the context head, force each remaining audio chunk to exactly " - "match its remaining picture duration. This follows NikoDemon80 v0.3.1's Trim node. " - "Ignored in CROSSFADE mode (no head/tail is removed)." - }), - "normalize_audio_per_clip": ("BOOLEAN", { - "default": True, - "tooltip": "Use ComfyUI's standard VAEDecodeAudio per-clip normalization. Disable if you want " - "raw VAE waveform levels before concatenation." - }), + "tooltip": "MiniMax H3 audio VAE FP32. Required for the AUDIO " + "output."}), "latent": ("LATENT", { "tooltip": "Optional: the currently-generated AV latent (from your " "H3 sampler), used in place of the highest-numbered file " - "on disk. Requires use_live_latent to be enabled."}), - "use_live_latent": ("BOOLEAN", { - "default": False, - "tooltip": "If enabled and 'latent' is connected, the " - "highest-indexed clip file on disk is skipped and " - "replaced with the live latent input instead -- " - "avoiding a race with the parallel SaveLatent node " - "and letting you preview the full video before that " - "file finishes writing."}), - }, - } + "on disk."}), + }, + } RETURN_TYPES = ("IMAGE", "AUDIO", "INT", "STRING") RETURN_NAMES = ("images", "audio", "frame_count", "report") FUNCTION = "stitch" CATEGORY = "video/minimax" - DESCRIPTION = ("Final assembly for NikoDemon80 H3 Motion Context AV archives. " + DESCRIPTION = ("Final assembly for NikoDemon80's H3 Motion Context AV clip archives.\n" "Loads numbered h3_motion_context_av_v1 files, decodes one clip at a " - "time, removes the carried context from the selected side of each " - "clip boundary, synchronizes audio, and concatenates.\n" - "CROSSFADE mode dissolves the overlap between adjacent clips (video " - "+ synchronized audio) instead of removing it.") + "time, dissolves the overlap between adjacent clips (video + synchronized audio), " + "and concatenates them to a final video and audio stream.") + # noinspection PyUnusedLocal @classmethod - def IS_CHANGED(cls, folder, pattern, first_clip, last_clip, context_length, trim_mode, fps, - video_vae=None, audio_vae=None, match_audio_tail=True, - normalize_audio_per_clip=True): + def IS_CHANGED(cls, folder, pattern, first_clip, last_clip, context_length, fps, + video_vae=None, audio_vae=None,): + # noinspection PyBroadException try: d = _resolve_folder(folder) files = _find_files(d, pattern, first_clip, last_clip) - return tuple((p, os.stat(p).st_mtime_ns, os.path.getsize(p)) for _, p in files) + ( - int(context_length), str(trim_mode), float(fps), bool(match_audio_tail), - bool(normalize_audio_per_clip), - ) + # noinspection PyTypeChecker + return tuple((p, os.stat(p).st_mtime_ns, os.path.getsize(p)) + for _, p in files) + (int(context_length), float(fps),) except Exception: return float("NaN") - def stitch(self, folder, pattern, first_clip, last_clip, context_length, trim_mode, fps, - video_vae=None, audio_vae=None, match_audio_tail=True, - normalize_audio_per_clip=True, latent=None, use_live_latent=False): + @staticmethod + def stitch(folder, pattern, first_clip, last_clip, context_length, fps, + video_vae=None, audio_vae=None, latent=None,): if video_vae is None: raise ValueError("Connect your MiniMax H3 video VAE to 'video_vae'.") if st_load is None: - raise RuntimeError("safetensors is not available in this ComfyUI environment.") + raise RuntimeError( + "safetensors is not available in this ComfyUI environment.") d = _resolve_folder(folder) files = _find_files(d, pattern, first_clip, last_clip) live_entry = None - if use_live_latent and latent is not None: + + if latent is not None: if files: files = files[:-1] # drop the presumed-duplicate on-disk file live_index = files[-1][0] + 1 if files else int(first_clip) else: live_index = int(first_clip) video_latent, audio_latent = _av_from_live_latent(latent) - live_entry = (live_index, None, video_latent, audio_latent) # path=None marks it as live - log_.info("H3 archive stitcher: %d clip(s) selected from %s", len(files), d) + live_entry = ( + live_index, None, video_latent, audio_latent) # path=None marks it as live image_parts = [] audio_parts = [] report_lines = [] target_sr = None - is_crossfade = trim_mode == "CROSSFADE" overlap = int(context_length) - # Degenerate crossfade (no overlap or a single clip) falls back to a - # plain concatenation, which is exactly what the trim modes would do. - crossfade_active = is_crossfade and overlap > 0 and len(files) > 1 prev_tail_img = None prev_tail_wave = None all_entries = [(idx, path, None, None) for idx, path in files] if live_entry is not None: + # noinspection PyTypeChecker all_entries.append(live_entry) + total_count = len(files) + (1 if live_entry is not None else 0) + log_.info("H3 archive stitcher: %d clip(s) selected from %s", total_count, d) + + # Degenerate crossfade (no overlap or a single clip) falls back to a + # plain concatenation, which is exactly what the trim modes would do. + crossfade_active = overlap > 0 and len(all_entries) > 1 + for pos, (idx, path, live_video, live_audio) in enumerate(all_entries): if path is not None: video_latent, audio_latent = _load_archive(path) else: video_latent, audio_latent = live_video, live_audio - log_.info("H3 archive stitcher: clip %05d taken from live latent input", idx) + log_.info("H3 archive stitcher: clip %05d taken from live latent input", + idx) # Decode one clip at a time. The decoded result is immediately moved # to CPU, so a long chain does not keep every VAE result on VRAM. @@ -439,7 +332,9 @@ class MiniMaxH3ContextStitcher: audio = None if audio_vae is not None: - audio = _decode_audio(audio_vae, audio_latent, normalize=normalize_audio_per_clip) + audio = _decode_audio(audio_vae, audio_latent, + # normalize=normalize_audio_per_clip + ) del audio_latent decoded_frames = int(images.shape[0]) @@ -447,10 +342,9 @@ class MiniMaxH3ContextStitcher: if crossfade_active: if decoded_frames < 2 * overlap: - raise ValueError( - "CROSSFADE requires each clip to have at least 2*context_length " - "(%d) frames; clip %05d has %d." % (overlap, idx, decoded_frames) - ) + raise ValueError("Crossfade requires each clip to have at least " + "2*context_length (%d) frames; clip %05d has %d." + % (overlap, idx, decoded_frames)) # Resample this clip's audio to the shared target rate before # splitting, so the head/tail sample counts line up across clips. @@ -468,10 +362,9 @@ class MiniMaxH3ContextStitcher: if n <= 0: n = 1 if n >= audio["waveform"].shape[-1]: - raise ValueError( - "Audio is too short to extract a %d-frame (%0.4fs) " - "crossfade head/tail for clip %05d." % (overlap, overlap / float(fps), idx) - ) + raise ValueError("Audio is too short to extract a %d-frame " + "(%0.4fs) crossfade head/tail for clip %05d." + % (overlap, overlap / float(fps), idx)) head_img = images[:overlap] body_img = images[overlap:-overlap] @@ -486,21 +379,18 @@ class MiniMaxH3ContextStitcher: tail_wave = {"waveform": wave[..., -n:], "sample_rate": sr} if pos == 0: - # First clip: emit head+body raw, buffer the tail for the - # next boundary. + # First clip: emit head+body raw, buffer the tail for the next boundary. image_parts.append(torch.cat([head_img, body_img], dim=0)) if audio is not None: - audio_parts.append(torch.cat( - [head_wave["waveform"], body_wave["waveform"]], dim=-1)) + audio_parts.append(torch.cat([head_wave["waveform"], + body_wave["waveform"]], dim=-1)) prev_tail_img = tail_img prev_tail_wave = tail_wave else: - blend_img, blend_wave = _crossfade_boundary( - prev_tail_img, images, - prev_tail_wave["waveform"] if prev_tail_wave is not None else None, - audio["waveform"] if audio is not None else None, - overlap, n - ) + blend_img, blend_wave = _crossfade_boundary(prev_tail_img, images, + prev_tail_wave[ + "waveform"] if prev_tail_wave is not None else None, + audio["waveform"] if audio is not None else None, overlap, n) image_parts.append(blend_img) if audio is not None: audio_parts.append(blend_wave) @@ -516,47 +406,15 @@ class MiniMaxH3ContextStitcher: prev_tail_wave = tail_wave kept_frames = decoded_frames - (overlap if not is_last else 0) - audio_sec = 0.0 if audio is None else audio["waveform"].shape[-1] / float(audio["sample_rate"]) - report_lines.append( - "clip_%05d: decoded=%d frames, crossfade=%d frames (%.4fs), kept=%d, audio=%.4fs" % - (idx, decoded_frames, overlap, overlap / float(fps), kept_frames, audio_sec) - ) - + audio_sec = (0.0 if audio is None else + audio["waveform"].shape[-1] / float(audio["sample_rate"])) + report_lines.append("clip_%05d: decoded=%d frames, crossfade=%d frames " + "(%.4fs), kept=%d, audio=%.4fs" + % (idx, decoded_frames, overlap, overlap / float(fps), + kept_frames, audio_sec)) del images if audio is not None: del audio - continue - - # --- TRIM_FRONT / TRIM_BACK path --- - if trim_mode == "TRIM_FRONT": - # Preserve the first clip; remove the carried context head from clips 2..N. - should_trim = pos > 0 - else: - # Preserve the final clip; remove the trailing context from clips 1..N-1. - should_trim = pos < len(files) - 1 - trim = int(context_length) if should_trim else 0 - images, audio = _trim_clip( - images, audio, trim, fps, match_audio_tail, trim_mode - ) - - image_parts.append(images) - if audio is not None: - if target_sr is None: - target_sr = int(audio["sample_rate"]) - audio = _resample_audio(audio, target_sr) - audio_parts.append(audio["waveform"]) - - kept_frames = int(images.shape[0]) - audio_sec = 0.0 if audio is None else audio["waveform"].shape[-1] / float(audio["sample_rate"]) - report_lines.append( - "clip_%05d: decoded=%d frames, trimmed=%d, kept=%d, audio=%.4fs" % - (idx, decoded_frames, trim, kept_frames, audio_sec) - ) - - # Explicitly drop local references before the next VAE decode. - del images - if audio is not None: - del audio final_images = torch.cat(image_parts, dim=0).contiguous() del image_parts @@ -569,25 +427,22 @@ class MiniMaxH3ContextStitcher: frame_count = int(final_images.shape[0]) video_seconds = frame_count / float(fps) - audio_seconds = (final_audio["waveform"].shape[-1] / float(final_audio["sample_rate"]) - if final_audio is not None else 0.0) + audio_seconds = (final_audio["waveform"].shape[-1] / float( + final_audio["sample_rate"]) if final_audio is not None else 0.0) - report_lines.append( - "TOTAL: %d frames = %.4fs at %.3f fps; audio=%.4fs%s" % - (frame_count, video_seconds, float(fps), audio_seconds, - "" if final_audio is not None else " (no audio_vae connected)") - ) + report_lines.append("TOTAL: %d frames = %.4fs at %.3f fps; audio=%.4fs%s" + % (frame_count, video_seconds, float(fps), audio_seconds, + "" if final_audio is not None + else " (no audio_vae connected)")) report = "\n".join(report_lines) log_.info("H3 archive stitcher finished: %d frames (%.3fs), audio %.3fs", frame_count, video_seconds, audio_seconds) - return (final_images, final_audio, frame_count, report) + return final_images, final_audio, frame_count, report NODE_CLASS_MAPPINGS = { - "MiniMaxH3MotionContextArchiveStitcher": MiniMaxH3ContextStitcher, -} + "H3MotionContextClipStitcher": H3MotionContextClipStitcher, } NODE_DISPLAY_NAME_MAPPINGS = { - "MiniMaxH3MotionContextArchiveStitcher": "MiniMax H3 Motion Context Archive Stitcher", -} \ No newline at end of file + "H3MotionContextClipStitcher": "H3 Motion Context Clip Stitcher", }