From 9e8fc5832dd5f99e549be9f98c2d447786ae89eb Mon Sep 17 00:00:00 2001 From: IAMCCS Date: Sun, 9 Aug 2026 02:09:44 +0200 Subject: [PATCH] added minimax h3 utilities --- README.md | 1 + SUPERNODES_REQUIREMENTS.md | 3 + docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md | 143 ++++ iamccs_minimax_h3_atomic_backend.py | 826 ++++++++++++++++++++--- iamccs_minimax_h3_cine_info.py | 490 ++++++++++++++ iamccs_minimax_h3_shotboard.py | 341 ++++++++-- iamccs_minimax_h3_shotboard_core.py | 166 ++++- iamccs_prompter.py | 212 +++++- iamccs_rtx_vfx.py | 93 ++- iamccs_shotboarder_exporter_pro.py | 2 + web/iamccs_minimax_h3_shotboard_ui.js | 500 ++++++++++++-- web/iamccs_prompter_ui.js | 188 +++++- 12 files changed, 2711 insertions(+), 254 deletions(-) create mode 100644 docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md create mode 100644 iamccs_minimax_h3_cine_info.py diff --git a/README.md b/README.md index 66e103c..8e52758 100644 --- a/README.md +++ b/README.md @@ -47,6 +47,7 @@ audio preprocessing, VAE decode, and video combine nodes. Before sharing or testing a SuperNode workflow, check the dedicated requirements document: - [IAMCCS SuperNodes Requirements](SUPERNODES_REQUIREMENTS.md) +- [MiniMax H3 Shotboard Workflow Requirements](docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md) - [AudioBoard + BusOut Guide](AUDIOBOARD_BUSOUT_GUIDE.md) ## 🆕 Added new LTX-2.3 nodes for v2v, au+img2vid (instructions: patreon.com/IAMCCS) diff --git a/SUPERNODES_REQUIREMENTS.md b/SUPERNODES_REQUIREMENTS.md index b4f2389..28f873d 100644 --- a/SUPERNODES_REQUIREMENTS.md +++ b/SUPERNODES_REQUIREMENTS.md @@ -1,5 +1,8 @@ # IAMCCS SuperNodes Requirements +> MiniMax H3 Shotboard, Turbo, LTX/Wan finishing, RIFE and RTX VSR have their +> own dependency matrix: [MiniMax H3 Shotboard Workflow Requirements](docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md). + The IAMCCS SuperNodes are workflow wrappers. They do not replace the underlying ComfyUI, LTXV, audio, VAE, stitching, and helper nodes; they orchestrate them. If one dependency is missing or outdated, the SuperNode may load in the graph but diff --git a/docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md b/docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md new file mode 100644 index 0000000..16dd3e1 --- /dev/null +++ b/docs/MINIMAX_H3_WORKFLOW_REQUIREMENTS.md @@ -0,0 +1,143 @@ +# IAMCCS MiniMax H3 Shotboard Workflow Requirements + +This document covers the IAMCCS MiniMax H3 Shotboard production workflows, +including the native H3 route, Turbo sampling, live preview, LTX/Wan finishing, +RIFE interpolation and optional RTX Video Super Resolution delivery. + +## Base runtime + +- A current ComfyUI build with native MiniMax H3 AV conditioning/sampling and + current LTX audio-video nodes. +- A recent NVIDIA driver and a PyTorch/CUDA build compatible with the selected + attention extensions. Reinstall compiled attention wheels after changing the + PyTorch/CUDA build. +- `IAMCCS-nodes` installed once in `custom_nodes`. Remove or move duplicate and + backup copies outside `custom_nodes`, otherwise ComfyUI can register stale + classes or report import failures. +- NVIDIA CUDA GPU for the supplied accelerated graphs. Twelve GB VRAM is + supported through dynamic weight offload; more VRAM reduces offload and wait + time. At least 32 GB system RAM is practical, while 64 GB is recommended for + H3 plus an LTX/Wan finishing pass. + +## Required node packs for the supplied H3 graphs + +### IAMCCS-nodes + +Provides the Shotboard and its workflow-facing wrappers: + +- `IAMCCS_MiniMaxH3ShotPlanner` +- `IAMCCS_MiniMaxH3AtomicModelRouter` +- `IAMCCS_MiniMaxH3AtomicConditioningBackend` +- `IAMCCS_MiniMaxH3GenerationBackendV2` +- `IAMCCS_MiniMaxH3PostUpscaleControlV2` +- `IAMCCS_MiniMaxH3DeliveryRouterV2` +- `IAMCCS_MiniMaxH3SegmentQueueLoop` +- `IAMCCS_MiniMaxH3SequentialLTXLoaderV2` +- `IAMCCS_MiniMaxH3OptionalLTXDetailerLoRA` +- `IAMCCS_MiniMaxH3RTX4KPost` +- `IAMCCS_Prompter` in the Prompter editions + +Repository: https://github.com/IAMCCS/IAMCCS-nodes + +### ComfyUI-GGUF + +Required when either the H3 diffusion model or Qwen3-VL text encoder is loaded +from GGUF. The supplied graphs use `UnetLoaderGGUFAdvanced` and +`CLIPLoaderGGUF`. + +Repository: https://github.com/city96/ComfyUI-GGUF + +### ComfyUI-KJNodes + +Used by the supplied graphs for MiniMax H3 Sage/low-VRAM patches, image resize, +TAEH3 preview override, KJ VAE loading and the LTX spatiotemporal tiled decode. + +Repository: https://github.com/kijai/ComfyUI-KJNodes + +### MiniMax H3 Turbo + +Required only when the Shotboard Turbo route is enabled. It supplies the Turbo +LoRA loader and `MiniMaxH3TurboSampler` used by the reference Turbo graphs. + +Repository: https://github.com/Larryvrh/ComfyUI-MiniMax-H3-Turbo + +## Optional finishing and acceleration packs + +- LTX Video nodes: recent ComfyUI contains the native LTX path used by the + current graph. `ComfyUI-LTXVideo` remains useful for compatible extended LTX + workflows: https://github.com/Lightricks/ComfyUI-LTXVideo +- Wan finishing: install the node pack required by the selected Wan branch; + the IAMCCS reference environment uses + https://github.com/kijai/ComfyUI-WanVideoWrapper +- RIFE frame interpolation: + https://github.com/Fannovel16/ComfyUI-Frame-Interpolation +- Sol Attention: https://github.com/kijai/ComfyUI-SolAttn_triton +- Spectrum for MiniMax H3: + https://github.com/xmarre/ComfyUI-Spectrum-MiniMax-H3 +- MiniMax H3 Adaptive Cache: + https://github.com/FFFFFFpy/ComfyUI-MiniMaxH3-AdaptiveCache + +These accelerators are alternatives or composable options only where the +Shotboard/backend explicitly reports them as active. A node merely present in +the graph does not accelerate an execution path that is bypassed. + +## Optional RTX 4K delivery + +The RTX final pass uses `RTXVideoSuperResolution` from: + +https://github.com/BetaDoggo/comfyui-rtx-simple + +It additionally requires NVIDIA VFX. Install it in the exact Python environment +that launches ComfyUI: + +```text +python -m pip install nvidia-vfx --extra-index-url https://pypi.nvidia.com/ +``` + +RTX VSR is an optional final delivery stage. It does not replace the LTX +generative finishing pass, and it must stay bypassed when `RTX final 4K` is off +in the Shotboard settings. + +## Model families expected by the workflow + +- One MiniMax H3 T2VA/I2VA/FL2VA model and, for REF2VA, the compatible REF2VA + model. Full, pruned INT8 and GGUF variants can be routed when their loader is + compatible with ComfyUI's native H3 model type. +- The matching Qwen3-VL MiniMax H3 text/vision encoder. +- MiniMax H3 video VAE and audio VAE. +- TAEH3 decoder for the live denoise preview. +- Turbo LoRA matching the chosen base model when Turbo is enabled. +- For LTX finishing: the selected LTX diffusion model, Gemma text encoder, + text projection, video VAE and audio VAE. A finishing LoRA is optional; the + Shotboard selector intentionally exposes all compatible installed LTX LoRAs. +- The selected latent/upscale model when the LTX graph uses a latent upres stage. + +## Low-VRAM execution contract + +The H3 backend uses a phased memory contract: + +1. Qwen3-VL conditioning runs GPU-first. +2. On GPUs up to 17 GB, IAMCCS supplies a temporary activation reserve so + ComfyUI dynamically offloads some Qwen weights instead of filling VRAM with + the entire encoder. +3. FL2VA keyframes or REF2VA media are encoded by their VAE. +4. `unload_all_models()`, model cleanup and CUDA cache cleanup run before the H3 + sampler requests the diffusion model. +5. A full CPU conditioning retry is used only after a genuine CUDA OOM. + +On a cold run, look for a log line containing `dynamic_reserve`, followed by +`conditioning complete`, `pre-sampler barrier`, and only then +`Requested to load MiniMaxH3`. If a warm Queue reuses cached conditioning, the +text-encoder load lines may legitimately be absent. + +## Installation validation + +After restarting ComfyUI: + +1. Open the workflow and confirm there are no red or `UNKNOWN` nodes. +2. Queue a short native H3 take with upscale, RIFE and RTX disabled. +3. Confirm the preview tap updates and the sampler advances. +4. Test LTX/Wan, RIFE and RTX as separate finishing checks before combining + them in a long multi-segment render. +5. Confirm the final saver uses numbered filenames so no previous take is + overwritten. diff --git a/iamccs_minimax_h3_atomic_backend.py b/iamccs_minimax_h3_atomic_backend.py index 3324bf7..f7860bf 100644 --- a/iamccs_minimax_h3_atomic_backend.py +++ b/iamccs_minimax_h3_atomic_backend.py @@ -23,6 +23,7 @@ import torch import torch.nn.functional as F from PIL import Image, ImageOps +import comfy.utils import folder_paths @@ -82,6 +83,36 @@ def _task_family(task: str) -> str: return "ref2va" if str(task or "").lower().startswith("ref2va") else "fl2va" +def _cine_info_h3(cine_linx: Any) -> tuple[dict[str, Any], dict[str, Any]]: + if not isinstance(cine_linx, dict): + return {}, {} + resources = cine_linx.get("resources") if isinstance(cine_linx.get("resources"), dict) else {} + config = resources.get("iamccs_minimax_h3_cine_info") + return (config if isinstance(config, dict) else {}), resources + + +def _effective_task(cine_linx: Any, chunk: dict[str, Any]) -> str: + config, _ = _cine_info_h3(cine_linx) + override = str(config.get("task_override", "from_shotboard") or "from_shotboard").lower() + if override in {"t2va", "i2va", "fl2va", "ref2va"}: + return override + return str(chunk.get("task_mode", "t2va") or "t2va").lower() + + +def _effective_shotplan(cine_linx: Any, shotplan: dict[str, Any]) -> dict[str, Any]: + config, _ = _cine_info_h3(cine_linx) + if not config: + return shotplan + result = dict(shotplan) + for key in ("reference_roles", "reference_video_role", "reference_audio_role", "ref_image_size"): + if key in config: + result[key] = config[key] + if isinstance(config.get("reference_resize"), dict): + result["reference_resize"] = dict(config["reference_resize"]) + result["reference_source"] = str(config.get("reference_source", "cine_info_h3_only")) + return result + + def _resolve_image_path(value: str) -> Path | None: raw = str(value or "").strip() if not raw: @@ -206,6 +237,12 @@ def _audio_slice(audio: dict[str, Any] | None, start_seconds: float, duration_se def _unique_plan_image_paths(shotplan: dict[str, Any]) -> list[str]: result: list[str] = [] + for path in shotplan.get("reference_image_paths", []): + clean = str(path or "").strip() + if clean and clean not in result: + result.append(clean) + if len(result) >= 4: + return result for slot in shotplan.get("slots", []): if not isinstance(slot, dict): continue @@ -243,35 +280,157 @@ def _reference_header(items: list[dict[str, str]], prompt: str) -> str: def _place_text_encoder(clip, shotplan: dict[str, Any]): - """Retarget the very large Qwen3-VL encoder before it is evaluated. + """Keep Qwen3-VL on ComfyUI's automatic GPU-first placement. - The Q2 GGUF file is compact on disk, but individual layers are expanded - while ComfyUI encodes the prompt. Loading the complete encoder on a - A 12 GiB Low VRAM configuration leaves no room for those temporary tensors and fails before - H3 sampling starts. ComfyUI's native retargeter clones the CLIP wrapper - and pins both load and offload devices to CPU without mutating the loader. + ``cpu_safe_12gb`` is retained only as a legacy workflow value. It no + longer pins the encoder to CPU because doing so turns prompt encoding into + the dominant runtime cost. A CPU clone is created only by the OOM retry + path below, after a real CUDA allocation failure. """ - mode = str(shotplan.get("text_encoder_device", "cpu_safe_12gb") or "cpu_safe_12gb").lower() - if mode == "auto": - return clip, "auto" - if mode != "cpu_safe_12gb": + mode = str(shotplan.get("text_encoder_device", "auto") or "auto").lower() + if mode not in {"auto", "cpu_safe_12gb"}: raise ValueError(f"Unsupported MiniMax H3 text encoder device mode: {mode}") + if mode == "cpu_safe_12gb": + return clip, "auto(gpu-first; migrated legacy cpu_safe_12gb)" + return clip, "auto(gpu-first)" + + +def _text_encoder_dynamic_reserve_mb(clip) -> int: + """Leave activation headroom when the 32B multimodal encoder uses a small GPU. + + ComfyUI's generic CLIP loader has no MiniMax-H3-specific activation estimate. + With a 12 GB card it can therefore stage the complete ~9.6 GB Qwen3-VL GGUF, + leaving too little room for the vision tower and two FL2VA image streams. A + temporary memory estimate makes CoreModelPatcher keep only part of the + weights resident while the remaining layers stream from CPU. This is still + GPU-first execution; it is not the much slower all-CPU fallback. + """ + patcher = getattr(clip, "patcher", None) + device = getattr(patcher, "load_device", None) + if getattr(device, "type", str(device)) != "cuda" or not torch.cuda.is_available(): + return 0 + try: + total_gib = torch.cuda.get_device_properties(device).total_memory / (1024 ** 3) + except Exception: + return 0 + if total_gib <= 13.0: + return 3584 + if total_gib <= 17.0: + return 2560 + return 0 + + +def _install_text_encoder_memory_estimate(clip, reserve_mb: int): + """Temporarily add/raise CLIP's activation estimate; return a restore callback.""" + stage = getattr(clip, "cond_stage_model", None) + if stage is None or reserve_mb <= 0: + return lambda: None + + instance_dict = getattr(stage, "__dict__", {}) + had_instance_value = "memory_estimation_function" in instance_dict + previous_instance_value = instance_dict.get("memory_estimation_function") + previous = getattr(stage, "memory_estimation_function", None) + reserve_bytes = int(reserve_mb) * 1024 * 1024 + + def estimate(tokens, device=None): + base = 0 + if callable(previous): + try: + base = int(previous(tokens, device=device)) + except TypeError: + base = int(previous(tokens)) + return max(base, reserve_bytes) + + stage.memory_estimation_function = estimate + + def restore(): + if had_instance_value: + stage.memory_estimation_function = previous_instance_value + else: + try: + delattr(stage, "memory_estimation_function") + except AttributeError: + pass + + return restore + + +def _is_cuda_oom(exc: BaseException) -> bool: + oom_type = getattr(torch, "OutOfMemoryError", None) + if oom_type is not None and isinstance(exc, oom_type): + return True + message = str(exc).lower() + return any( + marker in message + for marker in ( + "cuda out of memory", + "cuda error: out of memory", + "cudamalloc", + "cuda memory allocation", + ) + ) + + +def _clear_conditioning_cuda_after_oom() -> str: + """Release the failed GPU attempt before the one allowed CPU retry.""" + notes: list[str] = [] + try: + import comfy.model_management as mm + + mm.unload_all_models() + notes.append("unload_all_models") + try: + mm.cleanup_models() + notes.append("cleanup_models") + except Exception as exc: + notes.append(f"cleanup warning: {exc}") + mm.soft_empty_cache() + if torch.cuda.is_available(): + torch.cuda.empty_cache() + notes.append("empty CUDA cache") + except Exception as exc: + notes.append(f"cleanup warning: {exc}") + gc.collect() + return ", ".join(notes) + + +def _run_h3_conditioning_with_cpu_fallback(clip, shotplan: dict[str, Any], execute_fn): + """Encode GPU-first with activation headroom; retry on CPU only after CUDA OOM.""" + active_clip, placement_report = _place_text_encoder(clip, shotplan) + reserve_mb = _text_encoder_dynamic_reserve_mb(active_clip) + restore_estimate = _install_text_encoder_memory_estimate(active_clip, reserve_mb) + if reserve_mb: + placement_report += f"+dynamic_reserve={reserve_mb}MB" + LOG.info( + "MiniMax H3 low-VRAM text encode reserve active: %d MB kept for Qwen3-VL activations; weights remain GPU-first with dynamic offload", + reserve_mb, + ) + try: + try: + return execute_fn(active_clip), placement_report + finally: + restore_estimate() + except Exception as exc: + if not _is_cuda_oom(exc): + raise + + oom_exc = exc + cleanup_report = _clear_conditioning_cuda_after_oom() + LOG.warning( + "MiniMax H3 text conditioning exhausted CUDA memory; retrying once on CPU after %s", + cleanup_report, + ) from comfy_extras.nodes_multigpu import SelectCLIPDeviceNode - safe_clip = SelectCLIPDeviceNode.execute(clip=clip, device="cpu")[0] - load_device = getattr(getattr(safe_clip, "patcher", None), "load_device", None) + cpu_clip = SelectCLIPDeviceNode.execute(clip=clip, device="cpu")[0] + load_device = getattr(getattr(cpu_clip, "patcher", None), "load_device", None) if getattr(load_device, "type", str(load_device)) != "cpu": raise RuntimeError( - "MiniMax H3 CPU-safe mode could not place Qwen3-VL on CPU. " - "Update ComfyUI or insert Select CLIP Device=cpu before conditioning." - ) - return safe_clip, "cpu_safe_12gb(cpu)" - - -def _apply_sage(model): - sage_cls = _node_class("PathchSageAttentionKJ") - return sage_cls().patch(model=model, sage_attention="auto", allow_compile=False)[0] + "MiniMax H3 could not activate its CPU fallback after a CUDA OOM. " + "Update ComfyUI and retry." + ) from oom_exc + return execute_fn(cpu_clip), f"auto->cpu_fallback(cuda_oom; {cleanup_report})" def _apply_h3_memory_efficient_sage(model): @@ -279,6 +438,17 @@ def _apply_h3_memory_efficient_sage(model): return sage_cls.execute(model=model)[0] +def _apply_h3_low_vram_exact(model): + attention_cls = _node_class("MiniMaxLowVRAMAttention") + patched = attention_cls.execute(model=model, head_chunks=4)[0] + feed_forward_cls = _node_class("MiniMaxChunkFeedForward") + return feed_forward_cls.execute(model=patched, chunks=2, seq_threshold=4096)[0] + + +def _apply_h3_sage_low_vram(model): + return _apply_h3_low_vram_exact(_apply_h3_memory_efficient_sage(model)) + + def _apply_sol(model, conditioning_mode: str): sol_cls = _node_class("SolAttnPatch") return sol_cls.execute( @@ -298,27 +468,33 @@ def _apply_sol(model, conditioning_mode: str): )[0] +def _apply_adaptive_cache(model, preset: str): + cache_cls = _node_class("MiniMaxH3AdaptiveCache") + return cache_cls().patch(model=model, preset=preset, cache_device="auto")[0] + + def _apply_spectrum(model, profile: str): spectrum_cls = _node_class("SpectrumApplyMiniMaxH3") - profile = str(profile or "conservative_3060").lower() + profile = str(profile or "low_vram").lower() aggressive = profile == "aggressive" - quality = profile == "conservative_quality" - # max_history=5 is the minimum valid history for degree=4 and is the only - # practical default for a 1280x736 / 243-frame single branch on 32 GiB RAM. - max_history = 8 if quality else 5 + quality = profile in {"conservative_quality", "quality"} + degree = 4 if quality else 1 + warmup_steps = 5 if quality else 1 + max_history = 8 if quality else 2 return spectrum_cls().apply( model=model, enabled=True, blend_weight=0.75 if aggressive else 0.50, - degree=4, + degree=degree, ridge_lambda=0.10, window_size=2.0, flex_window=3.0 if aggressive else 0.75, - warmup_steps=5, + warmup_steps=warmup_steps, tail_actual_steps=1, max_history=max_history, debug=False, history_storage="system_ram", + bootstrap_first_forecast=not quality, )[0] @@ -391,30 +567,36 @@ def _turbo_sampler(shotplan: dict[str, Any]): def _accelerate(model, shotplan: dict[str, Any]): mode = str(shotplan.get("acceleration", "native") or "native").lower() - if mode == "auto_3060": + if mode in {"auto_3060", "low_vram_auto"}: try: - return _apply_h3_memory_efficient_sage(model), "Low VRAM Auto -> H3 Memory-Efficient Sage" + return _apply_h3_sage_low_vram(model), "Low VRAM Auto -> H3 Sage + exact attention/FFN chunks" except Exception as h3_exc: try: - return _apply_sage(model), f"Low VRAM Auto -> generic Sage (H3 patch unavailable: {h3_exc})" - except Exception as generic_exc: - return model, f"Low VRAM Auto -> native (H3 Sage: {h3_exc}; generic Sage: {generic_exc})" + return _apply_h3_memory_efficient_sage(model), f"Low VRAM Auto -> H3 Sage (exact chunks unavailable: {h3_exc})" + except Exception as sage_exc: + return model, f"Low VRAM Auto -> native (H3 stack: {h3_exc}; H3 Sage: {sage_exc})" if mode == "native": return model, "native" - if mode == "sage": - return _apply_sage(model), "SageAttention(auto)" - if mode == "h3_sage": - return _apply_h3_memory_efficient_sage(model), "MiniMax H3 Memory-Efficient Sage" - if mode == "sage_sol": - patched = _apply_sage(model) + if mode in {"sage", "h3_sage"}: + return _apply_h3_sage_low_vram(model), "MiniMax H3 Sage + exact attention/FFN chunks" + if mode in {"sage_sol", "sol_low_vram"}: + patched = _apply_h3_low_vram_exact(model) patched = _apply_sol(patched, str(shotplan.get("sol_conditioning", "exact_kv"))) - return patched, f"Sage+Sol({shotplan.get('sol_conditioning', 'exact_kv')})" + return patched, f"Sol({shotplan.get('sol_conditioning', 'exact_kv')}) + exact Low VRAM chunks" + if mode in {"adaptive_safe", "sol_adaptive_safe", "sol_adaptive_balanced"}: + patched = _apply_h3_low_vram_exact(model) + if mode.startswith("sol_"): + patched = _apply_sol(patched, str(shotplan.get("sol_conditioning", "exact_kv_and_rows"))) + preset = "balanced" if mode.endswith("balanced") else "safe" + patched = _apply_adaptive_cache(patched, preset) + prefix = "Sol + " if mode.startswith("sol_") else "" + return patched, f"{prefix}Adaptive Cache {preset} + exact Low VRAM chunks" if mode == "spectrum": - profile = str(shotplan.get("spectrum_profile", "conservative_3060")) + profile = str(shotplan.get("spectrum_profile", "low_vram")) return _apply_spectrum(model, profile), f"Spectrum({profile},system_ram)" if mode == "sage_spectrum": - profile = str(shotplan.get("spectrum_profile", "conservative_3060")) - sage_model = _apply_sage(model) + profile = str(shotplan.get("spectrum_profile", "low_vram")) + sage_model = _apply_h3_sage_low_vram(model) return _apply_spectrum(sage_model, profile), f"SageAttention + Spectrum({profile},system_ram)" raise ValueError(f"Unknown MiniMax H3 acceleration mode: {mode}") @@ -437,16 +619,13 @@ def _clean_vram_before_decode() -> str: return f"cleanup warning: {exc}" -def _release_cpu_conditioning_models(shotplan: dict[str, Any]) -> str: - """Drop CPU Qwen/VAE pages after conditioning, before H3 is loaded. +def _release_conditioning_models(shotplan: dict[str, Any]) -> str: + """Strict barrier after conditioning and immediately before H3 sampling. - The conditioning tensors are already materialized at this point. Keeping - the 9.6 GiB Qwen encoder resident would leave too little system RAM for a - Q4 H3 model and Spectrum history on a 32 GiB workstation. + Positive conditioning and the AV latent are already materialized when the + generation node runs. Qwen3-VL (and any conditioning-time VAE residency) + can therefore be unloaded before the H3 model is requested. """ - mode = str(shotplan.get("text_encoder_device", "cpu_safe_12gb") or "cpu_safe_12gb").lower() - if mode != "cpu_safe_12gb": - return "conditioning models kept" try: import comfy.model_management as mm @@ -456,6 +635,8 @@ def _release_cpu_conditioning_models(shotplan: dict[str, Any]) -> str: except Exception: pass mm.soft_empty_cache() + if torch.cuda.is_available(): + torch.cuda.empty_cache() except Exception as exc: LOG.warning("MiniMax H3 pre-sampler conditioning cleanup warning: %s", exc) return f"conditioning cleanup warning: {exc}" @@ -466,10 +647,14 @@ def _release_cpu_conditioning_models(shotplan: dict[str, Any]) -> str: handle = ctypes.windll.kernel32.GetCurrentProcess() ctypes.windll.psapi.EmptyWorkingSet(handle) - return "CPU text encoder/VAE unloaded; Windows working set trimmed" + report = "conditioning models unloaded; CUDA cache cleared; Windows working set trimmed" + LOG.info("MiniMax H3 pre-sampler barrier: %s", report) + return report except Exception as exc: LOG.warning("MiniMax H3 working-set trim warning: %s", exc) - return "CPU text encoder/VAE unloaded" + report = "conditioning models unloaded; CUDA cache cleared" + LOG.info("MiniMax H3 pre-sampler barrier: %s", report) + return report class IAMCCS_MiniMaxH3AtomicModelRouter: @@ -495,7 +680,8 @@ class IAMCCS_MiniMaxH3AtomicModelRouter: @staticmethod def _input_name(cine_linx, segment_index): - task = str(_chunk(cine_linx, segment_index).get("task_mode", "t2va")) + chunk = _chunk(cine_linx, segment_index) + task = _effective_task(cine_linx, chunk) return "ref2va_model" if _task_family(task) == "ref2va" else "fl2va_model" def check_lazy_status(self, cine_linx, segment_index, fl2va_model=None, ref2va_model=None, **kwargs): @@ -508,7 +694,7 @@ class IAMCCS_MiniMaxH3AtomicModelRouter: def select(self, cine_linx, segment_index, fl2va_model=None, ref2va_model=None): chunk = _chunk(cine_linx, segment_index) - task = str(chunk.get("task_mode", "t2va")) + task = _effective_task(cine_linx, chunk) family = _task_family(task) model = ref2va_model if family == "ref2va" else fl2va_model if model is None: @@ -582,10 +768,9 @@ class IAMCCS_MiniMaxH3AtomicConditioningBackend: ): from comfy_extras.nodes_minimax_h3 import MiniMaxH3ImageToVideo, MiniMaxH3ReferenceToVideo - shotplan = _resolve_shotplan(cine_linx) + shotplan = _effective_shotplan(cine_linx, _resolve_shotplan(cine_linx)) chunk = _chunk(shotplan, segment_index) - clip, text_encoder_report = _place_text_encoder(clip, shotplan) - task = str(chunk.get("task_mode", "t2va")).lower() + task = _effective_task(cine_linx, chunk) width = int(shotplan.get("width", 960)) height = int(shotplan.get("height", 544)) frames = int(chunk.get("frame_count", 124)) @@ -595,7 +780,19 @@ class IAMCCS_MiniMaxH3AtomicConditioningBackend: planned_last = _load_image(str(chunk.get("last_image", ""))) first = first_frame_override[:1] if torch.is_tensor(first_frame_override) else planned_first last = last_frame_override[:1] if torch.is_tensor(last_frame_override) else planned_last - external_images = [ref_image_1, ref_image_2, ref_image_3, ref_image_4] + h3_info, h3_resources = _cine_info_h3(cine_linx) + socket_images = [ref_image_1, ref_image_2, ref_image_3, ref_image_4] + resource_images = [h3_resources.get(f"iamccs_minimax_h3_ref_image_{index}") for index in range(1, 5)] + external_images = [ + socket if torch.is_tensor(socket) else resource + for socket, resource in zip(socket_images, resource_images) + ] + if not torch.is_tensor(ref_video): + ref_video = h3_resources.get("iamccs_minimax_h3_ref_video") + if not isinstance(ref_video_audio, dict): + ref_video_audio = h3_resources.get("iamccs_minimax_h3_ref_video_audio") + if not isinstance(ref_audio, dict): + ref_audio = h3_resources.get("iamccs_minimax_h3_ref_audio") if first is None and bool(chunk.get("uses_bridge_first_frame")) and torch.is_tensor(bridge_frame): first = bridge_frame[:1] @@ -621,27 +818,39 @@ class IAMCCS_MiniMaxH3AtomicConditioningBackend: manifest: list[dict[str, str]] = [] if task == "t2va": - result = MiniMaxH3ImageToVideo.execute( - clip=clip, vae=video_vae, prompt=prompt, - width=width, height=height, length=frames, - first_frame=None, last_frame=None, + result, text_encoder_report = _run_h3_conditioning_with_cpu_fallback( + clip, + shotplan, + lambda active_clip: MiniMaxH3ImageToVideo.execute( + clip=active_clip, vae=video_vae, prompt=prompt, + width=width, height=height, length=frames, + first_frame=None, last_frame=None, + ), ) elif task == "i2va": if first is None: raise ValueError("I2VA requires one opening image in the Shotboard or ref_image_1") - result = MiniMaxH3ImageToVideo.execute( - clip=clip, vae=video_vae, prompt=prompt, - width=width, height=height, length=frames, - first_frame=first, last_frame=None, + result, text_encoder_report = _run_h3_conditioning_with_cpu_fallback( + clip, + shotplan, + lambda active_clip: MiniMaxH3ImageToVideo.execute( + clip=active_clip, vae=video_vae, prompt=prompt, + width=width, height=height, length=frames, + first_frame=first, last_frame=None, + ), ) manifest.append({"label": "", "role": "opening_keyframe"}) elif task == "fl2va": if first is None or last is None: raise ValueError("FL2VA requires both opening and final images; connect ref_image_1/ref_image_2 or place two adjacent Shotboard keyframes") - result = MiniMaxH3ImageToVideo.execute( - clip=clip, vae=video_vae, prompt=prompt, - width=width, height=height, length=frames, - first_frame=first, last_frame=last, + result, text_encoder_report = _run_h3_conditioning_with_cpu_fallback( + clip, + shotplan, + lambda active_clip: MiniMaxH3ImageToVideo.execute( + clip=active_clip, vae=video_vae, prompt=prompt, + width=width, height=height, length=frames, + first_frame=first, last_frame=last, + ), ) manifest.extend([ {"label": "", "role": "opening_keyframe"}, @@ -649,7 +858,7 @@ class IAMCCS_MiniMaxH3AtomicConditioningBackend: ]) elif task.startswith("ref2va"): roles = _reference_roles(shotplan) - plan_paths = _unique_plan_image_paths(shotplan) + plan_paths = [] if str(h3_info.get("reference_source", "")) == "cine_info_h3_only" else _unique_plan_image_paths(shotplan) refs: dict[str, torch.Tensor] = {} for index in range(4): role = roles[index] @@ -699,29 +908,39 @@ class IAMCCS_MiniMaxH3AtomicConditioningBackend: if not refs and not ref_videos and not ref_audios: raise ValueError("REF2VA requires at least one enabled image, video, or audio reference") prompt = _reference_header(manifest, prompt) - result = MiniMaxH3ReferenceToVideo.execute( - clip=clip, - vae=video_vae, - audio_vae=audio_vae, - prompt=prompt, - width=width, - height=height, - length=frames, - ref_image_size=str(shotplan.get("ref_image_size", "match")), - ref_images=refs or None, - ref_videos=ref_videos, - ref_video_audios=ref_video_audios, - ref_audios=ref_audios, + result, text_encoder_report = _run_h3_conditioning_with_cpu_fallback( + clip, + shotplan, + lambda active_clip: MiniMaxH3ReferenceToVideo.execute( + clip=active_clip, + vae=video_vae, + audio_vae=audio_vae, + prompt=prompt, + width=width, + height=height, + length=frames, + ref_image_size=str(shotplan.get("ref_image_size", "match")), + ref_images=refs or None, + ref_videos=ref_videos, + ref_video_audios=ref_video_audios, + ref_audios=ref_audios, + ), ) else: raise ValueError(f"Unsupported atomic H3 task: {task}") positive, latent = result[0], result[1] + LOG.info( + "MiniMax H3 conditioning complete | task=%s | text_encoder=%s | pre-sampler unload barrier is next", + task, + text_encoder_report, + ) report = ( f"Atomic H3 conditioning | segment={int(segment_index) + 1}/{len(shotplan['chunks'])} | " f"task={task} | frames={frames} | first={'yes' if first is not None else 'no'} | " f"last={'yes' if last is not None else 'no'} | refs={len(manifest)} | " f"ref_size={shotplan.get('ref_image_size', 'match')} | " + f"ref_source={shotplan.get('reference_source', 'backend_sockets_or_legacy_timeline')} | " f"pre_resize={';'.join(resize_reports) if resize_reports else 'none'} | " f"text_encoder={text_encoder_report}" ) @@ -818,7 +1037,7 @@ class IAMCCS_MiniMaxH3GenerationBackendV2: shift_audio = float(sampling.get("shift_audio", shift_audio)) sampling_source = str(sampling.get("source", "backend_legacy_fallback")) actual_seed = (int(seed) + int(chunk_index) * int(seed_stride)) & 0xFFFFFFFFFFFFFFFF - conditioning_cleanup = _release_cpu_conditioning_models(shotplan) + conditioning_cleanup = _release_conditioning_models(shotplan) turbo = _turbo_settings(shotplan) turbo_requested = str(turbo.get("mode", "off") or "off").lower() != "off" and bool(turbo.get("enabled", True)) @@ -1003,6 +1222,360 @@ class IAMCCS_MiniMaxH3SequentialLTXLoaderV2: return model, clip, video_vae, audio_vae, report +def _materialize_conditioning_on_cpu(value): + if torch.is_tensor(value): + return value.detach().to(device="cpu", copy=True) + if isinstance(value, dict): + return {key: _materialize_conditioning_on_cpu(item) for key, item in value.items()} + if isinstance(value, list): + return [_materialize_conditioning_on_cpu(item) for item in value] + if isinstance(value, tuple): + return tuple(_materialize_conditioning_on_cpu(item) for item in value) + return value + + +def _release_ltx_text_stage() -> str: + import comfy.model_management as model_management + + if torch.cuda.is_available(): + torch.cuda.synchronize() + model_management.unload_all_models() + try: + model_management.cleanup_models() + except Exception: + pass + model_management.soft_empty_cache() + if torch.cuda.is_available(): + torch.cuda.empty_cache() + gc.collect() + if os.name == "nt": + try: + import ctypes + + handle = ctypes.windll.kernel32.GetCurrentProcess() + ctypes.windll.psapi.EmptyWorkingSet(handle) + return "unload_all_models + cleanup_models + CUDA cache + Windows working-set trim" + except Exception as exc: + LOG.warning("LTX text-stage working-set trim warning: %s", exc) + return "unload_all_models + cleanup_models + CUDA cache" + + +class IAMCCS_MiniMaxH3LTXConditioningStageV3: + """Materialize LTX text conditioning, then destroy Gemma and projection.""" + + @classmethod + def INPUT_TYPES(cls): + input_spec = IAMCCS_MiniMaxH3SequentialLTXLoaderV2._input_spec + return { + "required": { + "trigger_frames": ("IMAGE",), + "positive_text": ("STRING", {"default": "high quality 4k", "multiline": True}), + "negative_text": ( + "STRING", + { + "default": "pc game, console game, video game, cartoon, childish, ugly", + "multiline": True, + }, + ), + "text_encoder_name": input_spec( + "DualCLIPLoader", "clip_name1", "gemma_3_12B_it_fp8_e4m3fn.safetensors" + ), + "text_projection_name": input_spec( + "DualCLIPLoader", "clip_name2", "ltx-2.3_text_projection_bf16.safetensors" + ), + "text_encoder_device": (["default", "cpu"], {"default": "default"}), + } + } + + RETURN_TYPES = ("CONDITIONING", "CONDITIONING", "STRING") + RETURN_NAMES = ("positive", "negative", "report") + FUNCTION = "encode_and_release" + CATEGORY = CATEGORY + + def encode_and_release( + self, + trigger_frames, + positive_text, + negative_text, + text_encoder_name, + text_projection_name, + text_encoder_device="default", + ): + if not torch.is_tensor(trigger_frames) or trigger_frames.ndim != 4 or trigger_frames.shape[0] < 1: + raise ValueError("LTX conditioning stage requires decoded native H3 frames") + + _release_ltx_text_stage() + LOG.info( + "LTX phase A start | native_frames=%d | Gemma/projection only | device=%s", + int(trigger_frames.shape[0]), + text_encoder_device, + ) + clip = _node_class("DualCLIPLoader")().load_clip( + text_encoder_name, + text_projection_name, + "ltxv", + text_encoder_device, + )[0] + encoder = _node_class("CLIPTextEncode")() + try: + positive = encoder.encode(clip=clip, text=str(positive_text or ""))[0] + negative = encoder.encode(clip=clip, text=str(negative_text or ""))[0] + positive = _materialize_conditioning_on_cpu(positive) + negative = _materialize_conditioning_on_cpu(negative) + finally: + del encoder, clip + cleanup_report = _release_ltx_text_stage() + + report = ( + f"LTX phase A complete | conditioning=materialized on CPU | " + f"Gemma/projection destroyed=yes | {cleanup_report}" + ) + LOG.info(report) + return positive, negative, report + + +class IAMCCS_MiniMaxH3LTXDenoiseStackLoaderV3: + """Load LTXAV, VAEs and latent upsampler after Phase A has released text models.""" + + @classmethod + def INPUT_TYPES(cls): + input_spec = IAMCCS_MiniMaxH3SequentialLTXLoaderV2._input_spec + upscale_models = folder_paths.get_filename_list("latent_upscale_models") + upscale_meta = {} + preferred_upscaler = "ltx-2.3-spatial-upscaler-x2-1.1.safetensors" + if preferred_upscaler in upscale_models: + upscale_meta["default"] = preferred_upscaler + upscale_spec = (upscale_models, upscale_meta) if upscale_meta else (upscale_models,) + return { + "required": { + "conditioning_ready": ("CONDITIONING",), + "trigger_frames": ("IMAGE",), + "unet_name": input_spec( + "UnetLoaderGGUFAdvanced", "unet_name", "ltx-2.3-22b-dev-Q4_K_S.gguf" + ), + "video_vae_name": input_spec( + "VAELoaderKJ", "vae_name", "ltx-2.3-22b-dev_video_vae.safetensors" + ), + "audio_vae_name": input_spec( + "VAELoaderKJ", "vae_name", "ltx-2.3-22b-dev_audio_vae.safetensors" + ), + "upscale_model_name": upscale_spec, + } + } + + RETURN_TYPES = ("MODEL", "VAE", "VAE", "LATENT_UPSCALE_MODEL", "STRING") + RETURN_NAMES = ("model", "video_vae", "audio_vae", "upscale_model", "report") + FUNCTION = "load_after_conditioning" + CATEGORY = CATEGORY + + def load_after_conditioning( + self, + conditioning_ready, + trigger_frames, + unet_name, + video_vae_name, + audio_vae_name, + upscale_model_name, + video_vae_device="main_device", + video_vae_dtype="bf16", + audio_vae_device="cpu", + audio_vae_dtype="bf16", + ): + if not isinstance(conditioning_ready, list) or not conditioning_ready: + raise ValueError("LTX denoise stage requires materialized conditioning from Phase A") + if not torch.is_tensor(trigger_frames) or trigger_frames.ndim != 4 or trigger_frames.shape[0] < 1: + raise ValueError("LTX denoise stage requires decoded native H3 frames") + + cleanup_report = _release_ltx_text_stage() + LOG.info("LTX phase B start | Gemma/projection absent | loading denoise stack") + model = _node_class("UnetLoaderGGUFAdvanced")().load_unet( + unet_name, + dequant_dtype="default", + patch_dtype="default", + patch_on_device=False, + )[0] + vae_loader = _node_class("VAELoaderKJ")() + video_vae = vae_loader.load_vae(video_vae_name, video_vae_device, video_vae_dtype)[0] + audio_vae = vae_loader.load_vae(audio_vae_name, audio_vae_device, audio_vae_dtype)[0] + upscale_model = _node_class("LatentUpscaleModelLoader").execute(upscale_model_name)[0] + report = ( + f"LTX phase B ready | text models absent=yes | unet={unet_name} | " + f"video_vae={video_vae_name} | audio_vae={audio_vae_name} | " + f"upscaler={upscale_model_name} | pre_load_cleanup={cleanup_report}" + ) + LOG.info(report) + return model, video_vae, audio_vae, upscale_model, report + + +class IAMCCS_MiniMaxH3OptionalLTXDetailerLoRA: + """Apply one user-selected LTX finishing LoRA without hardcoded filenames.""" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "model": ("MODEL",), + "enabled": ("BOOLEAN", {"default": False}), + "lora_name": ("STRING", {"default": ""}), + "strength": ("FLOAT", {"default": 0.6, "min": 0.0, "max": 2.0, "step": 0.05}), + } + } + + RETURN_TYPES = ("MODEL", "STRING") + RETURN_NAMES = ("model", "report") + FUNCTION = "apply" + CATEGORY = CATEGORY + + def apply(self, model, enabled=False, lora_name="", strength=0.6): + name = str(lora_name or "").strip() + strength = min(2.0, max(0.0, float(strength or 0.0))) + if not bool(enabled) or not name or strength == 0.0: + return model, "LTX detailer LoRA off" + try: + path = folder_paths.get_full_path("loras", name) + except Exception: + path = None + if not path: + LOG.warning("Optional LTX detailer LoRA is unavailable: %s", name) + return model, f"LTX detailer unavailable ({name}); base LTX path retained" + + import nodes as comfy_nodes + + patched = comfy_nodes.LoraLoaderModelOnly().load_lora_model_only( + model=model, + lora_name=name, + strength_model=strength, + )[0] + compatibility_note = "" + if "ic-lora-detailer" in name.lower(): + compatibility_note = ( + " | compatibility load only: the official IC Detailer reaches its full effect " + "with the LTX IC conditioning/guiding-latent workflow" + ) + LOG.warning( + "LTX IC Detailer %s was loaded as a model-only finishing LoRA; " + "use the dedicated IC-guided LTX pipeline for its full behavior", + name, + ) + return patched, f"LTX detailer LoRA {name}@{strength:.2f}{compatibility_note}" + + +class IAMCCS_MiniMaxH3RTX4KPost: + """Optional final RTX VSR pass, requested lazily by the delivery router.""" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "images": ("IMAGE",), + "cine_linx": (SUPERNODE_LINX_TYPE,), + } + } + + RETURN_TYPES = ("IMAGE", "STRING") + RETURN_NAMES = ("images_4k", "report") + FUNCTION = "upscale" + CATEGORY = CATEGORY + + def upscale(self, images, cine_linx): + if not torch.is_tensor(images) or images.ndim != 4 or images.shape[0] < 1: + raise ValueError("RTX 4K post expects an IMAGE frame batch") + shotplan = _resolve_shotplan(cine_linx) + settings = shotplan.get("upscale_settings") if isinstance(shotplan.get("upscale_settings"), dict) else {} + if not bool(settings.get("ltx_4k_enabled", False)): + return images, "RTX VSR 4K off" + + quality = str(settings.get("ltx_4k_quality", "ULTRA") or "ULTRA").upper() + if quality not in {"ULTRA", "HIGH", "MEDIUM", "LOW"}: + quality = "ULTRA" + target_width = max(256, int(settings.get("target_width", 3840) or 3840)) + target_height = max(256, int(settings.get("target_height", 2160) or 2160)) + source_height = int(images.shape[1]) + source_width = int(images.shape[2]) + scale = max(target_width / max(1, source_width), target_height / max(1, source_height)) + if not 1.0 <= scale <= 4.0: + raise ValueError( + f"RTX VSR scale {scale:.3f} is outside the installed node's 1x-4x range " + f"({source_width}x{source_height} -> {target_width}x{target_height})" + ) + + # NVIDIA Video Effects accepts RGB float32 frames only. Never convert + # the full video at once: a 241-frame 1080p input plus its 4K float32 + # output can exceed 28 GiB before LTX caches are counted. Feed small + # batches to NvVFX and retain the completed 4K video as CPU float16. + source_device = str(images.device) + source_dtype = str(images.dtype) + if int(images.shape[-1]) < 3: + raise ValueError( + f"RTX 4K post expects at least three RGB channels, got shape {tuple(images.shape)}" + ) + + import comfy.model_management as model_management + + model_management.unload_all_models() + try: + model_management.cleanup_models() + except Exception: + pass + model_management.soft_empty_cache() + if torch.cuda.is_available(): + torch.cuda.empty_cache() + gc.collect() + + source = images[..., :3].detach().to(device="cpu") + frame_count = int(source.shape[0]) + chunk_size = min(8, frame_count) + upscaled = torch.empty( + (frame_count, target_height, target_width, 3), + device="cpu", + dtype=torch.float16, + ) + LOG.info( + "MiniMax H3 RTX VSR chunked start | %s/%s -> cpu/float16 4K | frames=%d | chunk=%d", + source_device, + source_dtype, + frame_count, + chunk_size, + ) + rtx_cls = _node_class("RTXVideoSuperResolution") + progress = comfy.utils.ProgressBar(frame_count) + for start in range(0, frame_count, chunk_size): + end = min(frame_count, start + chunk_size) + rtx_input = source[start:end].to(dtype=torch.float32).contiguous() + rtx_input = torch.nan_to_num( + rtx_input, + nan=0.0, + posinf=1.0, + neginf=0.0, + ).clamp_(0.0, 1.0) + chunk_output = rtx_cls.execute( + images=rtx_input, + scale=float(scale), + quality=quality, + )[0] + if int(chunk_output.shape[2]) != target_width or int(chunk_output.shape[1]) != target_height: + chunk_output = F.interpolate( + chunk_output.permute(0, 3, 1, 2), + size=(target_height, target_width), + mode="bicubic", + align_corners=False, + antialias=True, + ).permute(0, 2, 3, 1).clamp(0.0, 1.0) + upscaled[start:end].copy_(chunk_output.to(device="cpu", dtype=torch.float16)) + del rtx_input, chunk_output + if torch.cuda.is_available(): + torch.cuda.empty_cache() + progress.update_absolute(end, frame_count) + LOG.info("MiniMax H3 RTX VSR progress | %d/%d frames", end, frame_count) + del source + gc.collect() + return ( + upscaled, + f"RTX VSR {quality} | {source_width}x{source_height} -> {target_width}x{target_height} | " + f"scale={scale:.3f} | chunk={chunk_size} | output=cpu/float16", + ) + + class IAMCCS_MiniMaxH3PostUpscaleControlV2: """Resolve the selected post-upscale branch from the Shotboard CineLinX. @@ -1021,18 +1594,33 @@ class IAMCCS_MiniMaxH3PostUpscaleControlV2: }, } - RETURN_TYPES = ("IMAGE", "STRING", "INT", "INT", "FLOAT", "INT", "BOOLEAN", "FLOAT", "STRING", "STRING") + RETURN_TYPES = ( + "IMAGE", "STRING", "INT", "INT", "FLOAT", "INT", "BOOLEAN", "FLOAT", "STRING", "STRING", + "BOOLEAN", "STRING", "FLOAT", "BOOLEAN", "STRING", "BOOLEAN", + "INT", "INT", "INT", "INT", "INT", + ) RETURN_NAMES = ( "native_frames", "upscale_prompt", - "target_width", - "target_height", + "stage_target_width", + "stage_target_height", "duration_seconds", "upscale_seed", "sage_enabled", "wan_denoise", "selected_mode", "report", + "ltx_detailer_enabled", + "ltx_detailer_lora_name", + "ltx_detailer_strength", + "ltx_4k_enabled", + "ltx_4k_quality", + "ltx_seam_safe", + "ltx_encode_temporal_size", + "ltx_encode_temporal_overlap", + "ltx_decode_temporal_size", + "ltx_decode_temporal_overlap", + "ltx_decode_spatial_overlap", ) FUNCTION = "prepare" CATEGORY = CATEGORY @@ -1057,8 +1645,11 @@ class IAMCCS_MiniMaxH3PostUpscaleControlV2: native_width = int(shotplan.get("width", int(native_frames.shape[2])) or int(native_frames.shape[2])) native_height = int(shotplan.get("height", int(native_frames.shape[1])) or int(native_frames.shape[1])) - target_width = max(256, int(settings.get("target_width", native_width * 2) or native_width * 2)) - target_height = max(256, int(settings.get("target_height", native_height * 2) or native_height * 2)) + delivery_target_width = max(256, int(settings.get("target_width", native_width * 2) or native_width * 2)) + delivery_target_height = max(256, int(settings.get("target_height", native_height * 2) or native_height * 2)) + ltx_4k_enabled = bool(settings.get("ltx_4k_enabled", False)) and selected_mode == "ltx23" + stage_target_width = max(256, int(round(delivery_target_width / 2.0))) if ltx_4k_enabled else delivery_target_width + stage_target_height = max(256, int(round(delivery_target_height / 2.0))) if ltx_4k_enabled else delivery_target_height prompt = str(settings.get("prompt") or chunk.get("prompt") or shotplan.get("global_prompt") or "high quality cinematic video").strip() duration_seconds = float( chunk.get("duration_seconds") @@ -1071,23 +1662,49 @@ class IAMCCS_MiniMaxH3PostUpscaleControlV2: upscale_seed = (base_seed + index * seed_stride + seed_offset) & 0xFFFFFFFFFFFFFFFF sage_enabled = bool(settings.get("sage", True)) wan_denoise = min(1.0, max(0.0, float(settings.get("wan_denoise", 0.2) or 0.0))) + ltx_detailer_enabled = bool(settings.get("ltx_detailer_enabled", False)) + ltx_detailer_lora_name = str(settings.get("ltx_detailer_lora_name", "") or "").strip() + ltx_detailer_strength = min(2.0, max(0.0, float(settings.get("ltx_detailer_strength", 0.6) or 0.0))) + ltx_4k_quality = str(settings.get("ltx_4k_quality", "ULTRA") or "ULTRA").upper() + if ltx_4k_quality not in {"ULTRA", "HIGH", "MEDIUM", "LOW"}: + ltx_4k_quality = "ULTRA" + ltx_seam_safe = bool(settings.get("ltx_seam_safe", True)) + encode_temporal_size = int(settings.get("ltx_vae_encode_temporal_size", 500 if ltx_seam_safe else 64)) + encode_temporal_overlap = int(settings.get("ltx_vae_encode_temporal_overlap", 4 if ltx_seam_safe else 8)) + decode_temporal_size = int(settings.get("ltx_vae_decode_temporal_size", 64 if ltx_seam_safe else 16)) + decode_temporal_overlap = int(settings.get("ltx_vae_decode_temporal_overlap", 4 if ltx_seam_safe else 1)) + decode_spatial_overlap = int(settings.get("ltx_vae_decode_spatial_overlap", 4 if ltx_seam_safe else 1)) report = ( f"H3 post-upscale control | selected={selected_mode} | chunk={index + 1}/{max(1, len(chunks))} | " - f"native={native_width}x{native_height} -> target={target_width}x{target_height} | " + f"native={native_width}x{native_height} -> LTX stage={stage_target_width}x{stage_target_height} " + f"-> delivery={delivery_target_width}x{delivery_target_height} | " f"duration={duration_seconds:.3f}s | seed={upscale_seed} | sage={'on' if sage_enabled else 'off'} | " + f"detailer={'on' if ltx_detailer_enabled else 'off'}:{ltx_detailer_lora_name or 'none'}@{ltx_detailer_strength:.2f} | " + f"seam_safe={'on' if ltx_seam_safe else 'off'} | rtx_4k={'on' if ltx_4k_enabled else 'off'}:{ltx_4k_quality} | " f"wan_denoise={wan_denoise:.2f} | lazy=yes" ) return ( native_frames, prompt, - target_width, - target_height, + stage_target_width, + stage_target_height, duration_seconds, upscale_seed, sage_enabled, wan_denoise, selected_mode, report, + ltx_detailer_enabled, + ltx_detailer_lora_name, + ltx_detailer_strength, + ltx_4k_enabled, + ltx_4k_quality, + ltx_seam_safe, + encode_temporal_size, + encode_temporal_overlap, + decode_temporal_size, + decode_temporal_overlap, + decode_spatial_overlap, ) @@ -1105,6 +1722,7 @@ class IAMCCS_MiniMaxH3DeliveryRouterV2: }, "optional": { "ltx23_upscaled_frames": ("IMAGE", {"lazy": True}), + "ltx23_4k_frames": ("IMAGE", {"lazy": True}), "wan22_upscaled_frames": ("IMAGE", {"lazy": True}), }, } @@ -1121,6 +1739,9 @@ class IAMCCS_MiniMaxH3DeliveryRouterV2: return None mode = str(shotplan.get("upscale_mode", "off") or "off").lower() if mode == "ltx23": + settings = shotplan.get("upscale_settings") if isinstance(shotplan.get("upscale_settings"), dict) else {} + if bool(settings.get("ltx_4k_enabled", False)): + return "ltx23_4k_frames" return "ltx23_upscaled_frames" if mode == "wan22_5b": return "wan22_upscaled_frames" @@ -1133,12 +1754,15 @@ class IAMCCS_MiniMaxH3DeliveryRouterV2: native_audio, bridge_last_frame, ltx23_upscaled_frames=None, + ltx23_4k_frames=None, wan22_upscaled_frames=None, **kwargs, ): selected = self._selected_upscale_input(cine_linx) if selected == "ltx23_upscaled_frames" and ltx23_upscaled_frames is None: return [selected] + if selected == "ltx23_4k_frames" and ltx23_4k_frames is None: + return [selected] if selected == "wan22_upscaled_frames" and wan22_upscaled_frames is None: return [selected] return [] @@ -1185,6 +1809,7 @@ class IAMCCS_MiniMaxH3DeliveryRouterV2: native_audio, bridge_last_frame, ltx23_upscaled_frames=None, + ltx23_4k_frames=None, wan22_upscaled_frames=None, ): shotplan = _resolve_shotplan(cine_linx) @@ -1194,6 +1819,11 @@ class IAMCCS_MiniMaxH3DeliveryRouterV2: raise ValueError("Shotboard enabled LTX 2.3 upscale, but the lazy LTX output is not connected") delivery = ltx23_upscaled_frames upscale_report = "LTX 2.3" + elif selected == "ltx23_4k_frames": + if not torch.is_tensor(ltx23_4k_frames): + raise ValueError("Shotboard enabled RTX VSR 4K after LTX, but the lazy 4K output is not connected") + delivery = ltx23_4k_frames + upscale_report = "LTX 2.3 + RTX VSR 4K" elif selected == "wan22_upscaled_frames": if not torch.is_tensor(wan22_upscaled_frames): raise ValueError("Shotboard enabled Wan 2.2 5B upscale, but the lazy Wan output is not connected") @@ -1230,6 +1860,10 @@ NODE_CLASS_MAPPINGS = { "IAMCCS_MiniMaxH3AtomicConditioningBackend": IAMCCS_MiniMaxH3AtomicConditioningBackend, "IAMCCS_MiniMaxH3GenerationBackendV2": IAMCCS_MiniMaxH3GenerationBackendV2, "IAMCCS_MiniMaxH3SequentialLTXLoaderV2": IAMCCS_MiniMaxH3SequentialLTXLoaderV2, + "IAMCCS_MiniMaxH3LTXConditioningStageV3": IAMCCS_MiniMaxH3LTXConditioningStageV3, + "IAMCCS_MiniMaxH3LTXDenoiseStackLoaderV3": IAMCCS_MiniMaxH3LTXDenoiseStackLoaderV3, + "IAMCCS_MiniMaxH3OptionalLTXDetailerLoRA": IAMCCS_MiniMaxH3OptionalLTXDetailerLoRA, + "IAMCCS_MiniMaxH3RTX4KPost": IAMCCS_MiniMaxH3RTX4KPost, "IAMCCS_MiniMaxH3PostUpscaleControlV2": IAMCCS_MiniMaxH3PostUpscaleControlV2, "IAMCCS_MiniMaxH3DeliveryRouterV2": IAMCCS_MiniMaxH3DeliveryRouterV2, } @@ -1240,6 +1874,10 @@ NODE_DISPLAY_NAME_MAPPINGS = { "IAMCCS_MiniMaxH3AtomicConditioningBackend": "MiniMax H3 Atomic Shotboard Conditioning", "IAMCCS_MiniMaxH3GenerationBackendV2": "MiniMax H3 Generation V2 (Acceleration + Clean Decode)", "IAMCCS_MiniMaxH3SequentialLTXLoaderV2": "MiniMax H3 -> LTX Sequential Loader V2", + "IAMCCS_MiniMaxH3LTXConditioningStageV3": "MiniMax H3 -> LTX Phase A - Gemma Conditioning + Destroy", + "IAMCCS_MiniMaxH3LTXDenoiseStackLoaderV3": "MiniMax H3 -> LTX Phase B - Denoise Stack After Barrier", + "IAMCCS_MiniMaxH3OptionalLTXDetailerLoRA": "MiniMax H3 Optional LTX Detailer LoRA", + "IAMCCS_MiniMaxH3RTX4KPost": "MiniMax H3 RTX VSR 4K Post", "IAMCCS_MiniMaxH3PostUpscaleControlV2": "MiniMax H3 Post-Upscale Control V2 (LTX / Wan)", "IAMCCS_MiniMaxH3DeliveryRouterV2": "MiniMax H3 Delivery V2 (Lazy Upscale + RIFE)", } diff --git a/iamccs_minimax_h3_cine_info.py b/iamccs_minimax_h3_cine_info.py new file mode 100644 index 0000000..a844cfa --- /dev/null +++ b/iamccs_minimax_h3_cine_info.py @@ -0,0 +1,490 @@ +# SPDX-FileCopyrightText: 2026 Carmine Cristallo Scalzi (IAMCCS) +# SPDX-License-Identifier: GPL-3.0-or-later + +"""MiniMax H3 reference transport for IAMCCS CineLinX. + +The node deliberately keeps REF2VA reference media outside the Shotboard +timeline. The Shotboard remains the source of prompt, duration and shot +timing, while this node publishes optional image/video/audio references and +their roles to the isolated H3 backend through the existing CineLinX cable. +""" + +from __future__ import annotations + +import copy +import json +from typing import Any + +import torch + +from .iamccs_supernodes_linx import SUPERNODE_LINX_TYPE, build_stage_linx_payload +from .iamccs_minimax_h3_shotboard_core import build_shotplan, plan_json + + +CATEGORY = "IAMCCS/MiniMax H3" +STAGE_NAME = "iamccs_cine_info_h3" +RESOURCE_PREFIX = "iamccs_minimax_h3_" + +IMAGE_ROLES = ["subject_identity", "keyframe", "composition", "style", "disabled"] +VIDEO_ROLES = ["off", "motion_camera", "temporal_structure", "video_edit", "continuation"] +AUDIO_ROLES = ["off", "voice_timbre", "rhythm_timing", "audio_reuse", "sound_reference"] +TASK_OVERRIDES = ["from_shotboard", "t2va", "i2va", "fl2va", "ref2va"] + + +def _resources(cine_linx: Any) -> dict[str, Any]: + if not isinstance(cine_linx, dict): + return {} + resources = cine_linx.get("resources") + return resources if isinstance(resources, dict) else {} + + +def _has_h3_plan(cine_linx: Any) -> bool: + resources = _resources(cine_linx) + for key in ("iamccs_minimax_h3_shotplan", "minimax_h3_shotplan", "shotplan"): + value = resources.get(key) + if isinstance(value, dict) and value.get("schema") == "iamccs.minimax_h3.shotplan": + return True + return False + + +def _h3_plan(cine_linx: Any) -> dict[str, Any]: + resources = _resources(cine_linx) + outputs = cine_linx.get("outputs", {}) if isinstance(cine_linx, dict) else {} + payload = resources.get("cine_payload") if isinstance(resources.get("cine_payload"), dict) else {} + for value in ( + resources.get("iamccs_minimax_h3_shotplan"), + resources.get("minimax_h3_shotplan"), + resources.get("shotplan"), + outputs.get("shotplan") if isinstance(outputs, dict) else None, + payload.get("minimax_h3_shotplan"), + payload.get("shotplan"), + ): + if isinstance(value, dict) and value.get("schema") == "iamccs.minimax_h3.shotplan": + return value + raise ValueError("IAMCCS Cine Info H3 did not find a valid MiniMax H3 shotplan") + + +def _json_dict(value: Any) -> dict[str, Any]: + if isinstance(value, dict): + return copy.deepcopy(value) + try: + parsed = json.loads(str(value or "{}")) + except (TypeError, ValueError, json.JSONDecodeError): + return {} + return parsed if isinstance(parsed, dict) else {} + + +def _routed_timeline(cine_linx: Any) -> dict[str, Any]: + """Return only a TakeRouter-owned timeline; never infer a different take.""" + resources = _resources(cine_linx) + for value in ( + resources.get("cine_take_router_timeline_data"), + resources.get("cine_take_router_timeline_json"), + ): + timeline = _json_dict(value) + if timeline: + return timeline + return {} + + +def _copy_runtime_contract(source: dict[str, Any], target: dict[str, Any]) -> None: + """Preserve non-timeline H3 controls while rebuilding the selected take.""" + for key in ( + "prompter_injection", + "performance_profile", + "sampling", + "turbo", + "reference_resize", + "upscale_settings", + "control_contract", + "edition", + ): + if key in source: + target[key] = copy.deepcopy(source[key]) + + previous_performance = source.get("performance") if isinstance(source.get("performance"), dict) else {} + max_frames = max( + [int(chunk.get("frame_count", 0) or 0) for chunk in target.get("chunks", []) if isinstance(chunk, dict)] + or [0] + ) + width = int(target.get("width", 960) or 960) + height = int(target.get("height", 544) or 544) + performance = copy.deepcopy(previous_performance) + performance["max_chunk_frames"] = max_frames + performance["relative_native_load_vs_960x544x124"] = round( + (float(width) * float(height) * max(1, max_frames)) / (960.0 * 544.0 * 124.0), + 3, + ) + if performance: + target["performance"] = performance + + +def _pack_recompiled_plan( + cine_linx: dict[str, Any], + plan: dict[str, Any], + timeline: dict[str, Any], + report: str, +) -> dict[str, Any]: + slots = plan.get("slots") if isinstance(plan.get("slots"), list) else [] + chunks = plan.get("chunks") if isinstance(plan.get("chunks"), list) else [] + audio_segments = timeline.get("audioSegments") + if not isinstance(audio_segments, list): + audio_segments = timeline.get("audio_segments") + if not isinstance(audio_segments, list): + audio_segments = [] + global_prompt = str(plan.get("global_prompt", "") or "") + local_prompts = " | ".join( + str(slot.get("prompt", "")).strip() + for slot in slots + if isinstance(slot, dict) and str(slot.get("prompt", "")).strip() + ) + segment_lengths = ",".join( + str(int(chunk.get("frame_count", 0) or 0)) + for chunk in chunks + if isinstance(chunk, dict) + ) + plan_text = plan_json(plan) + prompt_map_text = json.dumps(plan.get("prompt_map", []), ensure_ascii=False, indent=2) + timeline_text = json.dumps(timeline, ensure_ascii=False) + previous_payload = _resources(cine_linx).get("cine_payload") + payload = copy.deepcopy(previous_payload) if isinstance(previous_payload, dict) else {} + payload.update({ + "backend_mode": "minimax_h3_multitimeline", + "pipeline_kind": "minimax_h3", + "global_prompt": global_prompt, + "local_prompts": local_prompts, + "segment_lengths": segment_lengths, + "duration_seconds": float(plan.get("effective_duration_seconds", 0.0) or 0.0), + "effective_duration_seconds": float(plan.get("effective_duration_seconds", 0.0) or 0.0), + "frame_rate": int(plan.get("fps", 24) or 24), + "width": int(plan.get("width", 0) or 0), + "height": int(plan.get("height", 0) or 0), + "timeline_data": copy.deepcopy(timeline), + "visual_segments": copy.deepcopy(slots), + "audioSegments": copy.deepcopy(audio_segments), + "minimax_h3_shotplan": plan, + }) + outputs = { + "shotplan": plan, + "shotplan_json": plan_text, + "prompt_map_json": prompt_map_text, + "total_segments": int(plan.get("total_segments", len(chunks)) or 0), + "effective_duration": float(plan.get("effective_duration_seconds", 0.0) or 0.0), + "global_prompt": global_prompt, + "local_prompts": local_prompts, + "segment_lengths": segment_lengths, + "timeline_data": timeline_text, + "audio_timeline_json": json.dumps({"audioSegments": audio_segments}, ensure_ascii=False), + "report": report, + } + resources = { + "cine_payload": payload, + "cine_global_prompt": global_prompt, + "cine_local_prompts": local_prompts, + "cine_segment_lengths": segment_lengths, + "cine_duration_seconds": float(plan.get("effective_duration_seconds", 0.0) or 0.0), + "cine_frame_rate": int(plan.get("fps", 24) or 24), + "cine_width": int(plan.get("width", 0) or 0), + "cine_height": int(plan.get("height", 0) or 0), + "cine_timeline_data_json": timeline_text, + "cine_visual_segments_json": json.dumps(slots, ensure_ascii=False), + "cine_audio_timeline_json": json.dumps({"audioSegments": audio_segments}, ensure_ascii=False), + "iamccs_minimax_h3_shotplan": plan, + "iamccs_minimax_h3_shotplan_json": plan_text, + "iamccs_minimax_h3_prompt_map": copy.deepcopy(plan.get("prompt_map", [])), + "iamccs_minimax_h3_prompt_map_json": prompt_map_text, + "iamccs_minimax_h3_total_segments": int(plan.get("total_segments", len(chunks)) or 0), + "iamccs_minimax_h3_effective_duration": float(plan.get("effective_duration_seconds", 0.0) or 0.0), + "iamccs_minimax_h3_multitimeline_report": report, + } + out = build_stage_linx_payload( + cine_linx, + stage_name="MiniMax H3 MultiTimeline Recompile", + stage_kind="minimax_h3_take_router_recompile", + payload=payload, + report=report, + outputs=outputs, + resources=resources, + policies={ + "minimax_h3_timeline_truth": "cine_take_router_timeline_data", + "minimax_h3_take_fallback": "forbidden", + }, + downstream_stages=("IAMCCS Cine Info H3", "MiniMax H3 backend"), + requires={"resources": ["cine_take_router_timeline_data", "iamccs_minimax_h3_shotplan"]}, + ) + out["mode"] = "minimax_h3_multitimeline" + return out + + +def _recompile_routed_h3_plan(cine_linx: dict[str, Any]) -> tuple[dict[str, Any], str]: + """Rebuild the H3 plan only when a strict TakeRouter timeline is present.""" + timeline = _routed_timeline(cine_linx) + if not timeline: + return cine_linx, "single_timeline" + source = _h3_plan(cine_linx) + global_prompt = str(timeline.get("global_prompt", timeline.get("prompt", source.get("global_prompt", ""))) or "") + duration = float( + timeline.get("duration_seconds", timeline.get("duration", source.get("requested_duration_seconds", 10.0))) + or 10.0 + ) + rebuilt = build_shotplan( + timeline_data=timeline, + global_prompt=global_prompt, + duration_seconds=duration, + task_mode=str(source.get("task_mode", "auto_from_timeline") or "auto_from_timeline"), + audio_mode=str(source.get("audio_mode", "h3_native_generated") or "h3_native_generated"), + prompt_mapping=str(source.get("prompt_mapping", "global_plus_local") or "global_plus_local"), + upscale_mode=str(source.get("upscale_mode", "off") or "off"), + width=int(source.get("width", 960) or 960), + height=int(source.get("height", 544) or 544), + acceleration=str(source.get("acceleration", "native") or "native"), + ref_image_size=str(source.get("ref_image_size", "match") or "match"), + text_encoder_device=str(source.get("text_encoder_device", "auto") or "auto"), + reference_roles=copy.deepcopy(source.get("reference_roles", [])), + reference_video_role=str(source.get("reference_video_role", "off") or "off"), + reference_audio_role=str(source.get("reference_audio_role", "off") or "off"), + sol_conditioning=str(source.get("sol_conditioning", "exact_kv") or "exact_kv"), + spectrum_profile=str(source.get("spectrum_profile", "conservative_3060") or "conservative_3060"), + vram_clean_before_decode=bool(source.get("vram_clean_before_decode", True)), + rife_mode=str(source.get("rife_mode", "off") or "off"), + upscale_enabled=bool(source.get("upscale_enabled", False)), + ) + _copy_runtime_contract(source, rebuilt) + multi = timeline.get("multiGeneration") if isinstance(timeline.get("multiGeneration"), dict) else {} + timeline_id = str(multi.get("activeTimelineId") or timeline.get("activeTimelineId") or "unknown") + take_index = int(multi.get("activeTake") or timeline.get("activeTake") or 0) + rebuilt["multitimeline"] = { + "routed": True, + "timeline_id": timeline_id, + "take_index": take_index, + "source": "IAMCCS_TakeRouter", + "fallback": "forbidden", + } + report = ( + "MiniMax H3 MultiTimeline recompiled | " + f"take={take_index} | timeline={timeline_id} | chunks={rebuilt.get('total_segments', 0)} | " + f"duration={float(rebuilt.get('effective_duration_seconds', 0.0)):.3f}s | " + "sampler/Turbo/resolution preserved" + ) + return _pack_recompiled_plan(cine_linx, rebuilt, timeline, report), report + + +def _shape(value: Any) -> list[int]: + if not torch.is_tensor(value): + return [] + return [int(item) for item in value.shape] + + +def _audio_meta(value: Any) -> dict[str, Any]: + if not isinstance(value, dict) or not torch.is_tensor(value.get("waveform")): + return {"connected": False} + waveform = value["waveform"] + sample_rate = int(value.get("sample_rate", 32000) or 32000) + samples = int(waveform.shape[-1]) if waveform.ndim else 0 + return { + "connected": True, + "shape": [int(item) for item in waveform.shape], + "sample_rate": sample_rate, + "duration_seconds": round(samples / max(1, sample_rate), 4), + } + + +def _clean_previous_h3_info(cine_linx: dict[str, Any]) -> dict[str, Any]: + """Remove a previous H3-info stage without copying tensor payloads.""" + cleaned = dict(cine_linx) + resources = dict(_resources(cine_linx)) + for key in list(resources): + if key.startswith(RESOURCE_PREFIX) and ( + key.startswith(f"{RESOURCE_PREFIX}ref_") + or key in { + f"{RESOURCE_PREFIX}cine_info", + f"{RESOURCE_PREFIX}reference_manifest", + f"{RESOURCE_PREFIX}reference_manifest_json", + } + ): + resources.pop(key, None) + cleaned["resources"] = resources + return cleaned + + +class IAMCCS_CineInfoH3: + """Attach MiniMax H3 REF2VA media to CineLinX, outside the timeline.""" + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "cine_linx": (SUPERNODE_LINX_TYPE,), + "task_override": (TASK_OVERRIDES, {"default": "from_shotboard"}), + "reference_role_1": (IMAGE_ROLES, {"default": "subject_identity"}), + "reference_role_2": (IMAGE_ROLES, {"default": "subject_identity"}), + "reference_role_3": (IMAGE_ROLES, {"default": "composition"}), + "reference_role_4": (IMAGE_ROLES, {"default": "style"}), + "reference_video_role": (VIDEO_ROLES, {"default": "off"}), + "reference_audio_role": (AUDIO_ROLES, {"default": "off"}), + "ref_image_size": (["match", "max"], {"default": "match"}), + "reference_resize_policy": ( + ["canvas_crop", "canvas_pad", "total_pixels", "off"], + {"default": "canvas_crop"}, + ), + "reference_resize_megapixels": ( + "FLOAT", + {"default": 0.5, "min": 0.1, "max": 2.0, "step": 0.05}, + ), + "reference_resize_filter": ( + ["area", "bilinear", "bicubic", "nearest-exact"], + {"default": "area"}, + ), + }, + "optional": { + "reference_image_1": ("IMAGE",), + "reference_image_2": ("IMAGE",), + "reference_image_3": ("IMAGE",), + "reference_image_4": ("IMAGE",), + "reference_video": ("IMAGE",), + "reference_video_audio": ("AUDIO",), + "reference_audio": ("AUDIO",), + }, + } + + RETURN_TYPES = (SUPERNODE_LINX_TYPE,) + RETURN_NAMES = ("cine_linx",) + FUNCTION = "attach" + CATEGORY = CATEGORY + + def attach( + self, + cine_linx, + task_override, + reference_role_1, + reference_role_2, + reference_role_3, + reference_role_4, + reference_video_role, + reference_audio_role, + ref_image_size, + reference_resize_policy, + reference_resize_megapixels, + reference_resize_filter, + reference_image_1=None, + reference_image_2=None, + reference_image_3=None, + reference_image_4=None, + reference_video=None, + reference_video_audio=None, + reference_audio=None, + ): + if not isinstance(cine_linx, dict): + raise ValueError("IAMCCS Cine Info H3 requires a valid cine_linx input") + if not _has_h3_plan(cine_linx): + raise ValueError( + "IAMCCS Cine Info H3 did not find a MiniMax H3 shotplan. " + "Connect MiniMax H3 Shotboard to IAMCCS CineInfo, then connect its cine_linx output here." + ) + cine_linx, timeline_mode = _recompile_routed_h3_plan(cine_linx) + + images = [reference_image_1, reference_image_2, reference_image_3, reference_image_4] + roles = [reference_role_1, reference_role_2, reference_role_3, reference_role_4] + image_items = [] + for index, (image, role) in enumerate(zip(images, roles), start=1): + image_items.append({ + "slot": index, + "label": f"", + "role": str(role), + "connected": bool(torch.is_tensor(image)), + "shape": _shape(image), + }) + + manifest = { + "schema": "iamccs.minimax_h3.cine_info", + "schema_version": 1, + "reference_source": "cine_info_h3_only", + "task_override": str(task_override), + "image_references": image_items, + "video_reference": { + "connected": bool(torch.is_tensor(reference_video)), + "shape": _shape(reference_video), + "role": str(reference_video_role), + "audio": _audio_meta(reference_video_audio), + }, + "audio_reference": { + **_audio_meta(reference_audio), + "role": str(reference_audio_role), + }, + "ref_image_size": str(ref_image_size), + "reference_resize": { + "policy": str(reference_resize_policy), + "megapixels": float(reference_resize_megapixels), + "filter": str(reference_resize_filter), + "multiple_of": 32, + "downscale_only": True, + }, + } + active_images = sum(1 for item in image_items if item["connected"] and item["role"] != "disabled") + active_video = bool(torch.is_tensor(reference_video) and str(reference_video_role) != "off") + active_audio = bool( + (isinstance(reference_audio, dict) and str(reference_audio_role) != "off") + or isinstance(reference_video_audio, dict) + ) + manifest["active_reference_count"] = active_images + int(active_video) + int(active_audio) + manifest_json = json.dumps(manifest, ensure_ascii=False, indent=2) + report = ( + "IAMCCS Cine Info H3 | references outside Shotboard timeline | " + f"task={task_override} | images={active_images}/4 | video={'on' if active_video else 'off'} | " + f"audio={'on' if active_audio else 'off'} | ref_size={ref_image_size} | " + f"resize={reference_resize_policy}:{float(reference_resize_megapixels):.2f}MP/{reference_resize_filter} | " + f"timeline={timeline_mode}" + ) + + config = { + "schema": manifest["schema"], + "schema_version": manifest["schema_version"], + "reference_source": manifest["reference_source"], + "task_override": str(task_override), + "reference_roles": [str(item) for item in roles], + "reference_video_role": str(reference_video_role), + "reference_audio_role": str(reference_audio_role), + "ref_image_size": str(ref_image_size), + "reference_resize": dict(manifest["reference_resize"]), + } + base = _clean_previous_h3_info(cine_linx) + resources = { + f"{RESOURCE_PREFIX}cine_info": config, + f"{RESOURCE_PREFIX}reference_manifest": manifest, + f"{RESOURCE_PREFIX}reference_manifest_json": manifest_json, + f"{RESOURCE_PREFIX}ref_image_1": reference_image_1, + f"{RESOURCE_PREFIX}ref_image_2": reference_image_2, + f"{RESOURCE_PREFIX}ref_image_3": reference_image_3, + f"{RESOURCE_PREFIX}ref_image_4": reference_image_4, + f"{RESOURCE_PREFIX}ref_video": reference_video, + f"{RESOURCE_PREFIX}ref_video_audio": reference_video_audio, + f"{RESOURCE_PREFIX}ref_audio": reference_audio, + } + out_linx = build_stage_linx_payload( + base, + stage_name=STAGE_NAME, + stage_kind="minimax_h3_reference_transport", + payload=config, + report=report, + slot_map={"cine_linx": "MiniMax H3 backend cine_linx"}, + downstream_stages=("IAMCCS MiniMax H3 Atomic Model Router", "IAMCCS MiniMax H3 Atomic Conditioning"), + policies={ + "reference_media_location": "cine_info_h3_not_shotboard_timeline", + "shotboard_owns": "prompt_duration_timeline", + "reference_precedence": "explicit_backend_socket_then_cine_info_h3", + }, + outputs={"minimax_h3_reference_manifest_json": manifest_json, "report": report}, + resources=resources, + requires={"resources": ["iamccs_minimax_h3_shotplan"]}, + ) + return (out_linx,) + + +NODE_CLASS_MAPPINGS = { + "IAMCCS_CineInfoH3": IAMCCS_CineInfoH3, +} + + +NODE_DISPLAY_NAME_MAPPINGS = { + "IAMCCS_CineInfoH3": "IAMCCS Cine Info H3 - REF2VA Inputs", +} diff --git a/iamccs_minimax_h3_shotboard.py b/iamccs_minimax_h3_shotboard.py index 908733b..ee1a552 100644 --- a/iamccs_minimax_h3_shotboard.py +++ b/iamccs_minimax_h3_shotboard.py @@ -408,16 +408,19 @@ def _encode_images(images: torch.Tensor, audio: dict[str, Any] | None, fps: floa raise RuntimeError("ffmpeg non trovato: impossibile salvare i segmenti MiniMax H3") if not torch.is_tensor(images) or images.ndim != 4 or images.shape[0] < 1: raise ValueError("images deve essere un batch IMAGE [T,H,W,C]") + if int(images.shape[-1]) < 3: + raise ValueError(f"images deve avere almeno tre canali RGB, shape ricevuta: {tuple(images.shape)}") output.parent.mkdir(parents=True, exist_ok=True) with tempfile.TemporaryDirectory(prefix="minimax_h3_segment_") as temp: temp_path = Path(temp) - for index, frame in enumerate(images): - _save_frame(temp_path / f"frame_{index:05d}.png", frame) wav_path = temp_path / "audio.wav" has_audio = _write_wav(audio, wav_path) + height = int(images.shape[1]) + width = int(images.shape[2]) command = [ - ffmpeg, "-nostdin", "-n", "-framerate", f"{float(fps):.6f}", - "-i", str(temp_path / "frame_%05d.png"), + ffmpeg, "-hide_banner", "-loglevel", "error", "-nostdin", "-n", + "-f", "rawvideo", "-pix_fmt", "rgb24", "-video_size", f"{width}x{height}", + "-framerate", f"{float(fps):.6f}", "-i", "pipe:0", ] if has_audio: command += ["-i", str(wav_path)] @@ -425,9 +428,42 @@ def _encode_images(images: torch.Tensor, audio: dict[str, Any] | None, fps: floa if has_audio: command += ["-c:a", "aac", "-b:a", "192k", "-ar", "48000", "-shortest"] command += ["-movflags", "+faststart", str(output)] - result = subprocess.run(command, capture_output=True, text=True, stdin=subprocess.DEVNULL) - if result.returncode != 0: - raise RuntimeError(f"ffmpeg segment encode failed: {result.stderr.strip() or result.stdout.strip()}") + process = subprocess.Popen( + command, + stdin=subprocess.PIPE, + stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE, + ) + try: + if process.stdin is None: + raise RuntimeError("ffmpeg raw-video stdin non disponibile") + total_frames = int(images.shape[0]) + for index, frame in enumerate(images): + rgb = ( + frame[..., :3] + .detach() + .to(device="cpu", dtype=torch.float32) + .nan_to_num(nan=0.0, posinf=1.0, neginf=0.0) + .clamp_(0.0, 1.0) + .mul_(255.0) + .round_() + .to(dtype=torch.uint8) + .contiguous() + .numpy() + ) + process.stdin.write(rgb.tobytes()) + if index == 0 or (index + 1) % 24 == 0 or index + 1 == total_frames: + LOG.info("MiniMax H3 streaming video encode | %d/%d frames", index + 1, total_frames) + process.stdin.close() + error_bytes = process.stderr.read() if process.stderr is not None else b"" + return_code = process.wait() + except Exception: + process.kill() + process.wait() + raise + if return_code != 0: + error = error_bytes.decode("utf-8", errors="replace").strip() + raise RuntimeError(f"ffmpeg segment encode failed: {error or f'exit {return_code}'}") def _concat_videos(paths: list[Path], output: Path) -> None: @@ -475,7 +511,7 @@ def _next_numbered_render_id(output_folder: Path, base_name: str, requested_rend escaped_base = re.escape(_safe_name(base_name, "segment")) escaped_root = re.escape(root) pattern = re.compile( - rf"^{escaped_base}_{escaped_root}(?:_(\d{{4,}}))?_(?:full|seg_\d{{4,}})\.mp4$", + rf"^{escaped_base}_{escaped_root}(?:_(\d{{4,}}))?_(?:native_)?(?:full|seg_\d{{4,}})\.mp4$", re.IGNORECASE, ) used_numbers: set[int] = set() @@ -495,9 +531,12 @@ def _next_numbered_render_id(output_folder: Path, base_name: str, requested_rend next_number = max(used_numbers, default=0) + 1 while True: candidate = f"{root}_{next_number:04d}" - segment_collision = any(output_folder.glob(f"{_safe_name(base_name, 'segment')}_{candidate}_seg_*.mp4")) - final_collision = (output_folder / f"{_safe_name(base_name, 'segment')}_{candidate}_full.mp4").exists() - if not segment_collision and not final_collision: + safe_base = _safe_name(base_name, "segment") + segment_collision = any(output_folder.glob(f"{safe_base}_{candidate}_seg_*.mp4")) + native_segment_collision = any(output_folder.glob(f"{safe_base}_{candidate}_native_seg_*.mp4")) + final_collision = (output_folder / f"{safe_base}_{candidate}_full.mp4").exists() + native_final_collision = (output_folder / f"{safe_base}_{candidate}_native_full.mp4").exists() + if not segment_collision and not native_segment_collision and not final_collision and not native_final_collision: return candidate next_number += 1 @@ -567,7 +606,7 @@ class IAMCCS_MiniMaxH3GGUFLoader: "spectrum_debug": ("BOOLEAN", {"default": False}), }, "optional": { - "text_encoder_device": (["cpu_safe_12gb", "auto"], {"default": "cpu_safe_12gb"}), + "text_encoder_device": (["auto", "cpu_safe_12gb"], {"default": "auto"}), }, } @@ -586,7 +625,7 @@ class IAMCCS_MiniMaxH3GGUFLoader: acceleration, spectrum_history, spectrum_debug, - text_encoder_device="cpu_safe_12gb", + text_encoder_device="auto", ): if str(unet_name).startswith("NO_") or str(clip_name).startswith("NO_"): raise FileNotFoundError("GGUF H3 UNET/CLIP non disponibili. Attendi la fine dei download e riavvia ComfyUI.") @@ -602,15 +641,10 @@ class IAMCCS_MiniMaxH3GGUFLoader: )[0] clip_cls = _node_class("CLIPLoaderGGUF") clip = clip_cls().load_clip(clip_name, type="minimax")[0] - if str(text_encoder_device) == "cpu_safe_12gb": - # Qwen3-VL 32B Q2 is ~8 GiB before temporary dequant buffers. - # Running it on a 12 GiB GPU fails before H3 sampling begins. - # Use ComfyUI's native device retargeter instead of mutating the - # GGUF patcher internals so future model-management changes remain - # compatible. - from comfy_extras.nodes_multigpu import SelectCLIPDeviceNode - - clip = SelectCLIPDeviceNode.execute(clip=clip, device="cpu")[0] + requested_text_encoder_device = str(text_encoder_device or "auto").lower() + text_encoder_device = "auto" + if requested_text_encoder_device == "cpu_safe_12gb": + text_encoder_device = "auto(gpu-first; migrated legacy cpu_safe_12gb)" vae_cls = _node_class("VAELoader") video_vae = vae_cls().load_vae(video_vae_name)[0] audio_vae = vae_cls().load_vae(audio_vae_name)[0] @@ -660,9 +694,27 @@ class IAMCCS_MiniMaxH3ShotPlanner: name for name in folder_paths.get_filename_list("loras") if "minimax" in name.lower() and "h3" in name.lower() and "turbo" in name.lower() ] + # The LTX finishing slot intentionally exposes the complete ComfyUI + # LoRA registry. Some useful LTX LoRAs (including community Crisp + # variants) do not carry reliable "ltx/detail/enhance" tokens in the + # filename or parent folder. Runtime compatibility remains the user's + # choice; keeping the historical field name preserves old workflows. + installed_ltx_detailer_loras = list(folder_paths.get_filename_list("loras")) + crisp_ltx_loras = sorted( + ( + name for name in installed_ltx_detailer_loras + if "ltx" in name.lower() and "crisp" in name.lower() + ), + key=lambda name: ( + 0 if Path(name).name.lower() == "ltx2.3_crisp_enhance.safetensors" else 1, + name.lower(), + ), + ) + preferred_ltx_detailer_lora = crisp_ltx_loras[0] if crisp_ltx_loras else "" # Only expose files that really exist. The web migration converts old # saved missing filenames to the empty/Base-H3 choice before validation. turbo_loras = list(dict.fromkeys(("", *installed_turbo_loras))) + ltx_detailer_loras = list(dict.fromkeys(("", *installed_ltx_detailer_loras))) if "res_multistep" in samplers: samplers.remove("res_multistep") samplers.insert(0, "res_multistep") @@ -700,7 +752,11 @@ class IAMCCS_MiniMaxH3ShotPlanner: "img_compression": ("INT", {"default": 0, "min": 0, "max": 100, "step": 1}), # H3-native backend controls. The dedicated Shotboard UI # renders these above the timeline and hides the raw widgets. - "acceleration": (["auto_3060", "native", "h3_sage", "sage", "sage_sol", "spectrum", "sage_spectrum"], {"default": "auto_3060"}), + "acceleration": ([ + "low_vram_auto", "native", "h3_sage", "sol_low_vram", "sol_adaptive_safe", + "sol_adaptive_balanced", "adaptive_safe", "spectrum", "sage_spectrum", + "auto_3060", "sage", "sage_sol", + ], {"default": "low_vram_auto"}), "ref_image_size": (["match", "max"], {"default": "match"}), "reference_role_1": (["subject_identity", "keyframe", "composition", "style", "disabled"], {"default": "subject_identity"}), "reference_role_2": (["subject_identity", "keyframe", "composition", "style", "disabled"], {"default": "subject_identity"}), @@ -708,8 +764,8 @@ class IAMCCS_MiniMaxH3ShotPlanner: "reference_role_4": (["subject_identity", "keyframe", "composition", "style", "disabled"], {"default": "style"}), "reference_video_role": (["off", "motion_camera", "temporal_structure", "video_edit", "continuation"], {"default": "off"}), "reference_audio_role": (["off", "voice_timbre", "rhythm_timing", "audio_reuse", "sound_reference"], {"default": "off"}), - "sol_conditioning": (["exact_kv", "exact_kv_and_rows"], {"default": "exact_kv"}), - "spectrum_profile": (["conservative_3060", "conservative_quality", "aggressive"], {"default": "conservative_3060"}), + "sol_conditioning": (["exact_kv_and_rows", "exact_kv"], {"default": "exact_kv_and_rows"}), + "spectrum_profile": (["low_vram", "quality", "aggressive", "conservative_3060", "conservative_quality"], {"default": "low_vram"}), "vram_clean_before_decode": ("BOOLEAN", {"default": True}), "rife_mode": (["off", "rife_48fps", "rife_60fps"], {"default": "off"}), "upscale_enabled": ("BOOLEAN", {"default": False}), @@ -723,14 +779,16 @@ class IAMCCS_MiniMaxH3ShotPlanner: "upscale_sage": ("BOOLEAN", {"default": True}), "upscale_seed_offset": ("INT", {"default": 10000, "min": 0, "max": 0xFFFFFFFFFFFFFFFF, "step": 1}), "wan_upscale_denoise": ("FLOAT", {"default": 0.2, "min": 0.0, "max": 1.0, "step": 0.01}), - # Qwen3-VL 32B GGUF temporarily expands quantized tensors while - # encoding. A 12 GiB GPU cannot hold the full encoder plus its - # dequantization buffers, so the atomic backend defaults to CPU. - "text_encoder_device": (["cpu_safe_12gb", "auto"], {"default": "cpu_safe_12gb"}), + # ComfyUI automatic placement is GPU-first. The atomic backend + # retries on CPU only after a genuine CUDA out-of-memory error. + "text_encoder_device": (["auto", "cpu_safe_12gb"], {"default": "auto"}), # These values are intentionally owned by the Shotboard. The # generation node keeps legacy widgets only as a compatibility # fallback for shotplans created before schema v3. - "performance_profile": (["rtx3060_draft", "rtx3060_balanced", "rtx3060_turbo", "h3_native_quality", "custom"], {"default": "rtx3060_balanced"}), + "performance_profile": ([ + "low_vram_draft", "low_vram_balanced", "low_vram_turbo", "h3_native_quality", "custom", + "rtx3060_draft", "rtx3060_balanced", "rtx3060_turbo", + ], {"default": "low_vram_balanced"}), "seed": ("INT", {"default": 42, "min": 0, "max": 0xFFFFFFFFFFFFFFFF}), "seed_stride": ("INT", {"default": 1, "min": 0, "max": 0xFFFFFFFFFFFFFFFF, "step": 1}), "steps": ("INT", {"default": 16, "min": 1, "max": 100, "step": 1}), @@ -751,6 +809,18 @@ class IAMCCS_MiniMaxH3ShotPlanner: "reference_resize_policy": (["canvas_crop", "canvas_pad", "total_pixels", "off"], {"default": "canvas_crop"}), "reference_resize_megapixels": ("FLOAT", {"default": 0.5, "min": 0.1, "max": 2.0, "step": 0.05}), "reference_resize_filter": (["area", "bilinear", "bicubic", "nearest-exact"], {"default": "area"}), + # Optional LTX finishing controls. All installed LoRAs remain + # selectable; an installed LTX Crisp variant is preselected. + # The enable switch still owns whether the LoRA is applied. + "ltx_detailer_enabled": ("BOOLEAN", {"default": False}), + "ltx_detailer_lora_name": ( + ltx_detailer_loras, + {"default": preferred_ltx_detailer_lora}, + ), + "ltx_detailer_strength": ("FLOAT", {"default": 0.6, "min": 0.0, "max": 2.0, "step": 0.05}), + "ltx_4k_enabled": ("BOOLEAN", {"default": False}), + "ltx_4k_quality": (["ULTRA", "HIGH", "MEDIUM", "LOW"], {"default": "ULTRA"}), + "ltx_seam_safe": ("BOOLEAN", {"default": True}), }, "optional": { "cine_linx": ( @@ -791,7 +861,7 @@ class IAMCCS_MiniMaxH3ShotPlanner: image_resize_method="crop", image_multiple_of=32, img_compression=0, - acceleration="auto_3060", + acceleration="low_vram_auto", ref_image_size="match", reference_role_1="subject_identity", reference_role_2="subject_identity", @@ -799,8 +869,8 @@ class IAMCCS_MiniMaxH3ShotPlanner: reference_role_4="style", reference_video_role="off", reference_audio_role="off", - sol_conditioning="exact_kv", - spectrum_profile="conservative_3060", + sol_conditioning="exact_kv_and_rows", + spectrum_profile="low_vram", vram_clean_before_decode=True, rife_mode="off", upscale_enabled=False, @@ -810,8 +880,8 @@ class IAMCCS_MiniMaxH3ShotPlanner: upscale_sage=True, upscale_seed_offset=10000, wan_upscale_denoise=0.2, - text_encoder_device="cpu_safe_12gb", - performance_profile="rtx3060_balanced", + text_encoder_device="auto", + performance_profile="low_vram_balanced", seed=42, seed_stride=1, steps=16, @@ -827,6 +897,12 @@ class IAMCCS_MiniMaxH3ShotPlanner: reference_resize_policy="canvas_crop", reference_resize_megapixels=0.5, reference_resize_filter="area", + ltx_detailer_enabled=False, + ltx_detailer_lora_name="", + ltx_detailer_strength=0.6, + ltx_4k_enabled=False, + ltx_4k_quality="ULTRA", + ltx_seam_safe=True, cine_linx=None, ): global_prompt, timeline_data, prompter_injection = apply_prompter_to_minimax( @@ -839,6 +915,14 @@ class IAMCCS_MiniMaxH3ShotPlanner: image_width = _h3_legal_dimension(image_width, width) image_height = _h3_legal_dimension(image_height, height) reference_resize_megapixels = _finite_float(reference_resize_megapixels, 0.5, 0.1, 2.0) + selected_ltx_detailer = str(ltx_detailer_lora_name or "").strip() + ltx_detailer_requested = bool(ltx_detailer_enabled) + ltx_detailer_available = _model_file_available("loras", selected_ltx_detailer) + effective_ltx_detailer = ltx_detailer_requested and ltx_detailer_available + effective_ltx_4k = bool(ltx_4k_enabled) and bool(upscale_enabled) and str(upscale_mode) == "ltx23" + ltx_4k_quality = str(ltx_4k_quality or "ULTRA").upper() + if ltx_4k_quality not in {"ULTRA", "HIGH", "MEDIUM", "LOW"}: + ltx_4k_quality = "ULTRA" requested_turbo_mode = str(turbo_mode or "off") selected_turbo_lora = str(turbo_lora_name or "").strip() @@ -912,17 +996,39 @@ class IAMCCS_MiniMaxH3ShotPlanner: "sage": bool(upscale_sage), "seed_offset": int(upscale_seed_offset), "wan_denoise": float(wan_upscale_denoise), + "ltx_detailer_requested": ltx_detailer_requested, + "ltx_detailer_enabled": effective_ltx_detailer, + "ltx_detailer_available": ltx_detailer_available, + "ltx_detailer_lora_name": selected_ltx_detailer, + "ltx_detailer_strength": float(ltx_detailer_strength), + "ltx_4k_enabled": effective_ltx_4k, + "ltx_4k_quality": ltx_4k_quality, + "ltx_seam_safe": bool(ltx_seam_safe), + "ltx_vae_encode_temporal_size": 500 if bool(ltx_seam_safe) else 64, + "ltx_vae_encode_temporal_overlap": 4 if bool(ltx_seam_safe) else 8, + "ltx_vae_decode_temporal_size": 64 if bool(ltx_seam_safe) else 16, + "ltx_vae_decode_temporal_overlap": 4 if bool(ltx_seam_safe) else 1, + "ltx_vae_decode_spatial_overlap": 4 if bool(ltx_seam_safe) else 1, "source": "shotboard", } chunk_frames = [int(chunk.get("frame_count", 0) or 0) for chunk in plan.get("chunks", [])] max_chunk_frames = max(chunk_frames, default=0) native_load = (float(width) * float(height) * max(1, max_chunk_frames)) / (960.0 * 544.0 * 124.0) warnings: list[str] = [] - if str(performance_profile).startswith("rtx3060") and max_chunk_frames > 124: + if ltx_detailer_requested and not ltx_detailer_available: + warnings.append(f"Optional LTX detailer unavailable ({selected_ltx_detailer or 'no LoRA selected'}); continuing without it") + if bool(ltx_4k_enabled) and not effective_ltx_4k: + warnings.append("RTX VSR 4K is available only when LTX 2.3 upscale is enabled") + if effective_ltx_4k: + warnings.append("4K delivery uses LTX at half delivery resolution, then NVIDIA RTX VSR 2x; expect high system-RAM usage") + if str(text_encoder_device).lower() == "cpu_safe_12gb": + warnings.append("Legacy CPU-safe text encoder setting migrated to GPU-first auto with CPU fallback only after CUDA OOM") + low_vram_profile = str(performance_profile).startswith(("low_vram", "rtx3060")) + if low_vram_profile and max_chunk_frames > 124: warnings.append("Low VRAM: trim this timeline box to 124 frames or less; use a following box for continuation") - if str(performance_profile).startswith("rtx3060") and int(width) * int(height) > 960 * 544: + if low_vram_profile and int(width) * int(height) > 960 * 544: warnings.append("Low VRAM: generate at 960x544 or below, then upscale for a 1280-class delivery") - if str(acceleration) == "sage_sol": + if str(acceleration) in {"sage_sol", "sol_low_vram", "sol_adaptive_safe", "sol_adaptive_balanced"}: warnings.append("Sol-Attn is experimental, has a slower first compile, and is not validated for every Low VRAM configuration") if str(acceleration) in {"spectrum", "sage_spectrum"} and effective_steps < 14: warnings.append("Spectrum saves few transformer calls below 14 steps because warmup and final native steps remain mandatory") @@ -936,6 +1042,8 @@ class IAMCCS_MiniMaxH3ShotPlanner: warnings.append("Early/non-ckpt500 Turbo is normally used at 8-10 steps") if effective_turbo_mode == "ckpt500_6_8" and not 6 <= effective_steps <= 8: warnings.append("Turbo ckpt500 is normally used at 6-8 steps") + if str(acceleration) in {"adaptive_safe", "sol_adaptive_safe", "sol_adaptive_balanced"}: + warnings.append("Adaptive Cache is approximate; use Safe for faces, hands, dialogue and lip sync") if turbo_enabled and str(acceleration) in {"spectrum", "sage_spectrum"}: warnings.append("Spectrum has little room to forecast at Turbo step counts; Sage-only is the Low VRAM default") if turbo_enabled and str(turbo_sampler_mode) == "res_multistep_stock" and int(steps) < 10: @@ -951,7 +1059,7 @@ class IAMCCS_MiniMaxH3ShotPlanner: "conditioning": ["width", "height", "timeline trim", "prompt mapping", "references", "audio mode"], "sampling": ["seed", "steps", "sampler", "scheduler", "denoise", "H3 shifts", "acceleration", "Turbo LoRA", "Turbo audio sampler"], "reference_preprocess": ["resize policy", "target megapixels", "filter", "multiple of 32"], - "delivery": ["VRAM clean", "RIFE", "upscale enabled", "upscale mode", "upscale target", "upscale prompt", "upscale seed"], + "delivery": ["VRAM clean", "RIFE", "upscale enabled", "upscale mode", "upscale target", "upscale prompt", "upscale seed", "LTX seam-safe VAE", "LTX detailer LoRA", "optional RTX VSR 4K"], "transport": "one IAMCCS_SUPERNODE_LINX cable; the private H3 plan stays inside CineLinX", } injection_summary = str(prompter_injection.get("actual_target", "none")) if prompter_injection.get("applied") else "none" @@ -963,9 +1071,11 @@ class IAMCCS_MiniMaxH3ShotPlanner: f"load={native_load:.2f}x | sampler={effective_steps}x{sampler_name}+{scheduler} | acceleration={acceleration} | " f"turbo={effective_turbo_mode}:{selected_turbo_lora or 'none'}@{float(turbo_strength):.2f}/{turbo_sampler_mode} | " f"ref_resize={reference_resize_policy}:{reference_resize_megapixels:.2f}MP/{reference_resize_filter} | " - f"ref_size={ref_image_size} | text_encoder={text_encoder_device} | " + f"ref_size={ref_image_size} | text_encoder={plan.get('text_encoder_device', 'auto')} | " f"RIFE={rife_mode} | upscale={'on' if upscale_enabled else 'off'}:{plan['upscale_mode']} " f"->{int(upscale_width)}x{int(upscale_height)} sage={'on' if upscale_sage else 'off'} " + f"ltx_detailer={'on' if effective_ltx_detailer else 'off'}:{selected_ltx_detailer or 'none'}@{float(ltx_detailer_strength):.2f} " + f"ltx_seam_safe={'on' if ltx_seam_safe else 'off'} ltx_4k={'on' if effective_ltx_4k else 'off'}:{ltx_4k_quality} " f"wan_denoise={float(wan_upscale_denoise):.2f} | " f"prompter={injection_summary} | warnings={'; '.join(warnings) if warnings else 'none'}" ) @@ -1398,6 +1508,133 @@ class IAMCCS_MiniMaxH3BridgeLoad: raise FileNotFoundError(f"MiniMax H3 bridge non trovato: {bridge_path}") +class IAMCCS_MiniMaxH3NativeCheckpointSave: + """Persist the native H3 result before any optional upscale branch. + + The node is deliberately a pass-through dependency. Downstream LTX, Wan, + or RTX processing cannot begin until the native segment has been encoded, + so an upscale failure never discards the expensive H3 render. + """ + + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "images": ("IMAGE",), + "audio": ("AUDIO",), + "current_segment": ("INT", {"forceInput": True}), + "total_segments": ("INT", {"forceInput": True}), + "fps": ("INT", {"forceInput": True}), + "trim_head_frames": ("INT", {"forceInput": True}), + }, + "optional": { + "filename_prefix": ("STRING", {"default": "IAMCCS/MiniMaxH3/segment"}), + "merge_segments": ("BOOLEAN", {"default": True}), + "keep_segments": ("BOOLEAN", {"default": True}), + "render_id": ("STRING", {"default": "minimax_h3_render"}), + }, + } + + RETURN_TYPES = ("IMAGE", "AUDIO", "STRING", "STRING") + RETURN_NAMES = ("native_frames", "native_audio", "resolved_render_id", "report") + FUNCTION = "checkpoint" + OUTPUT_NODE = True + CATEGORY = CATEGORY + + @classmethod + def IS_CHANGED(cls, *args, **kwargs): + # Saving is intentional on every queued render, even when ComfyUI can + # reuse the surrounding graph cache. + return float("nan") + + def checkpoint( + self, + images, + audio, + current_segment, + total_segments, + fps, + trim_head_frames, + filename_prefix="IAMCCS/MiniMaxH3/segment", + merge_segments=True, + keep_segments=True, + render_id="minimax_h3_render", + ): + if not torch.is_tensor(images) or images.ndim != 4 or int(images.shape[0]) < 1: + raise ValueError("MiniMax H3 native checkpoint expects a non-empty IMAGE frame batch") + if not isinstance(audio, dict): + raise ValueError("MiniMax H3 native checkpoint expects the H3 AUDIO output") + + current_segment = int(current_segment) + total_segments = int(total_segments) + fps = max(1, int(fps)) + if current_segment < 0 or total_segments < 1 or current_segment >= total_segments: + raise ValueError(f"Native checkpoint segment index is invalid: {current_segment + 1}/{total_segments}") + + output_folder, base_name = _output_location(filename_prefix) + requested_render_id = _safe_name(str(render_id or "").strip(), "minimax_h3_render") + active_render_id = ( + _next_numbered_render_id(output_folder, base_name, requested_render_id) + if current_segment == 0 + else requested_render_id + ) + + trim_count = max(0, int(trim_head_frames or 0)) + images_to_save = images + audio_to_save = audio + if trim_count and int(images.shape[0]) > trim_count: + images_to_save = images[trim_count:, ...] + audio_to_save = _trim_audio_frames(audio, trim_count, fps) + + segment_name = f"{base_name}_{active_render_id}_native_seg_{current_segment + 1:04d}.mp4" + segment_path = output_folder / segment_name + _require_new_output_path(segment_path) + _encode_images(images_to_save, audio_to_save, fps, segment_path) + + messages = [f"Native checkpoint saved: {segment_name}"] + preview_path = segment_path + if current_segment + 1 >= total_segments and bool(merge_segments): + segment_paths = [ + output_folder / f"{base_name}_{active_render_id}_native_seg_{index + 1:04d}.mp4" + for index in range(total_segments) + ] + final_name = f"{base_name}_{active_render_id}_native_full.mp4" + final_path = output_folder / final_name + _require_new_output_path(final_path) + _concat_videos(segment_paths, final_path) + preview_path = final_path + messages.append(f"Native full video saved: {final_name}") + if not bool(keep_segments): + for path in segment_paths: + path.unlink(missing_ok=True) + + LOG.info( + "MiniMax H3 native checkpoint complete | render=%s | segment=%d/%d | fps=%d", + active_render_id, + current_segment + 1, + total_segments, + fps, + ) + subfolder = os.path.relpath( + preview_path.parent, + folder_paths.get_output_directory(), + ).replace("\\", "/") + preview = { + "filename": preview_path.name, + "subfolder": "" if subfolder == "." else subfolder, + "type": "output", + } + report = " | ".join(messages) + return { + "ui": { + "text": messages, + "images": [preview], + "animated": (True,), + }, + "result": (images, audio, active_render_id, report), + } + + class IAMCCS_MiniMaxH3SegmentQueueLoop: @classmethod def INPUT_TYPES(cls): @@ -1418,6 +1655,7 @@ class IAMCCS_MiniMaxH3SegmentQueueLoop: "keep_segments": ("BOOLEAN", {"default": True}), "render_id": ("STRING", {"default": "minimax_h3_render"}), "segment_base_name": ("STRING", {"default": ""}), + "resolved_render_id": ("STRING", {"forceInput": True}), }, "hidden": {"prompt": "PROMPT", "unique_id": "UNIQUE_ID", "extra_pnginfo": "EXTRA_PNGINFO"}, } @@ -1442,6 +1680,7 @@ class IAMCCS_MiniMaxH3SegmentQueueLoop: keep_segments=True, render_id="minimax_h3_render", segment_base_name="", + resolved_render_id="", prompt=None, unique_id=None, extra_pnginfo=None, @@ -1464,11 +1703,15 @@ class IAMCCS_MiniMaxH3SegmentQueueLoop: ) output_folder, resolved_base_name = _output_location(filename_prefix) active_base_name = _safe_name(str(segment_base_name or "").strip(), resolved_base_name) - active_render_id = ( - _next_numbered_render_id(output_folder, active_base_name, requested_render_id) - if current_segment == 0 - else requested_render_id - ) + locked_render_id = str(resolved_render_id or "").strip() + if locked_render_id: + active_render_id = _safe_name(locked_render_id, requested_render_id) + else: + active_render_id = ( + _next_numbered_render_id(output_folder, active_base_name, requested_render_id) + if current_segment == 0 + else requested_render_id + ) if current_segment == 0: LOG.info("MiniMax H3 nuovo render numerato: %s", active_render_id) segment_name = f"{active_base_name}_{active_render_id}_seg_{current_segment + 1:04d}.mp4" @@ -1708,6 +1951,7 @@ NODE_CLASS_MAPPINGS = { "IAMCCS_MiniMaxH3Backend": IAMCCS_MiniMaxH3Backend, "IAMCCS_MiniMaxH3RenderBackend": IAMCCS_MiniMaxH3RenderBackend, "IAMCCS_MiniMaxH3BridgeLoad": IAMCCS_MiniMaxH3BridgeLoad, + "IAMCCS_MiniMaxH3NativeCheckpointSave": IAMCCS_MiniMaxH3NativeCheckpointSave, "IAMCCS_MiniMaxH3SegmentQueueLoop": IAMCCS_MiniMaxH3SegmentQueueLoop, "IAMCCS_MiniMaxH3AudioConcat": IAMCCS_MiniMaxH3AudioConcat, "IAMCCS_MiniMaxH3AudioPolicy": IAMCCS_MiniMaxH3AudioPolicy, @@ -1724,6 +1968,7 @@ NODE_DISPLAY_NAME_MAPPINGS = { "IAMCCS_MiniMaxH3Backend": "MiniMax H3 Shotboard Backend", "IAMCCS_MiniMaxH3RenderBackend": "MiniMax H3 Render Backend (Sampler + AV Decode)", "IAMCCS_MiniMaxH3BridgeLoad": "MiniMax H3 Last-Frame Bridge", + "IAMCCS_MiniMaxH3NativeCheckpointSave": "MiniMax H3 Native Checkpoint Save", "IAMCCS_MiniMaxH3SegmentQueueLoop": "MiniMax H3 Segment Queue + Concat", "IAMCCS_MiniMaxH3AudioConcat": "MiniMax H3 Audio Chunk Concat", "IAMCCS_MiniMaxH3AudioPolicy": "MiniMax H3 Audio Policy", @@ -1738,6 +1983,12 @@ from .iamccs_minimax_h3_atomic_backend import ( NODE_CLASS_MAPPINGS as _ATOMIC_NODE_CLASS_MAPPINGS, NODE_DISPLAY_NAME_MAPPINGS as _ATOMIC_NODE_DISPLAY_NAME_MAPPINGS, ) +from .iamccs_minimax_h3_cine_info import ( + NODE_CLASS_MAPPINGS as _CINE_INFO_H3_NODE_CLASS_MAPPINGS, + NODE_DISPLAY_NAME_MAPPINGS as _CINE_INFO_H3_NODE_DISPLAY_NAME_MAPPINGS, +) NODE_CLASS_MAPPINGS.update(_ATOMIC_NODE_CLASS_MAPPINGS) NODE_DISPLAY_NAME_MAPPINGS.update(_ATOMIC_NODE_DISPLAY_NAME_MAPPINGS) +NODE_CLASS_MAPPINGS.update(_CINE_INFO_H3_NODE_CLASS_MAPPINGS) +NODE_DISPLAY_NAME_MAPPINGS.update(_CINE_INFO_H3_NODE_DISPLAY_NAME_MAPPINGS) diff --git a/iamccs_minimax_h3_shotboard_core.py b/iamccs_minimax_h3_shotboard_core.py index bcd2cec..0ee4b8a 100644 --- a/iamccs_minimax_h3_shotboard_core.py +++ b/iamccs_minimax_h3_shotboard_core.py @@ -267,6 +267,118 @@ def _normalise_slots(timeline: dict[str, Any], duration_seconds: float, fallback ] +def _timeline_h3_bridges(timeline: dict[str, Any]) -> list[dict[str, Any]]: + """Return the dedicated MiniMax bridge contract, when the UI supplied it.""" + for key in ("h3_bridges", "h3Bridges"): + value = timeline.get(key) + if isinstance(value, list): + return [dict(item) for item in value if isinstance(item, dict)] + nested = timeline.get("timeline") + if isinstance(nested, dict): + return _timeline_h3_bridges(nested) + return [] + + +def _normalise_flf_bridge_slots( + timeline: dict[str, Any], + slots: list[dict[str, Any]], + duration_seconds: float, +) -> list[dict[str, Any]]: + """Convert N image anchors into N-1 MiniMax first/last-frame chunks. + + The Shotboard renders the local prompt from the centre of one image box to + the centre of the next. Those centre distances determine the *relative* + duration of the FLF chunks, while the first and last centres are normalised + to the full requested timeline duration. Consequently two image anchors + on a ten-second board still produce one ten-second FLF chunk; with three or + more anchors, resizing or moving a box changes the proportional timing of + the adjacent chunks without losing the requested total duration. + """ + anchors = [slot for slot in slots if _text(slot.get("image"))] + if len(anchors) < 2: + return slots + + ui_bridges = _timeline_h3_bridges(timeline) + bridge_by_pair: dict[tuple[str, str], dict[str, Any]] = {} + for bridge in ui_bridges: + pair = ( + _text(_first_value(bridge, ("from_segment_id", "fromSegmentId", "from_id"))), + _text(_first_value(bridge, ("to_segment_id", "toSegmentId", "to_id"))), + ) + if pair[0] and pair[1]: + bridge_by_pair[pair] = bridge + + centres = [ + float(slot["start_seconds"]) + float(slot["requested_frame_count"]) / H3_FPS / 2.0 + for slot in anchors + ] + gaps = [max(1.0 / H3_FPS, centres[index + 1] - centres[index]) for index in range(len(centres) - 1)] + gap_total = sum(gaps) or float(len(gaps)) + requested_total = max(H3_MIN_FRAMES, int(round(max(0.01, _float(duration_seconds, 10.0)) * H3_FPS))) + if requested_total > H3_MAX_TRAINED_FRAMES * len(gaps): + raise ValueError( + f"La timeline FLF richiede {requested_total} frame ma {len(gaps)} ponti H3 possono contenerne " + f"al massimo {H3_MAX_TRAINED_FRAMES * len(gaps)}. Aggiungi keyframe o riduci la durata." + ) + + raw_lengths = [requested_total * gap / gap_total for gap in gaps] + requested_lengths = [max(H3_MIN_FRAMES, int(math.floor(value))) for value in raw_lengths] + remainder = requested_total - sum(requested_lengths) + order = sorted( + range(len(raw_lengths)), + key=lambda index: raw_lengths[index] - math.floor(raw_lengths[index]), + reverse=remainder > 0, + ) + step = 1 if remainder > 0 else -1 + for offset in range(abs(remainder)): + index = order[offset % len(order)] + if step < 0 and requested_lengths[index] <= H3_MIN_FRAMES: + continue + requested_lengths[index] += step + + bridge_slots: list[dict[str, Any]] = [] + cursor = 0.0 + for index, (first, last) in enumerate(zip(anchors, anchors[1:])): + requested_frames = requested_lengths[index] + if requested_frames > H3_MAX_TRAINED_FRAMES: + raise ValueError( + f"Il ponte FLF '{first['label']} -> {last['label']}' richiede {requested_frames} frame: " + f"avvicina i centri dei box o aggiungi un keyframe (massimo {H3_MAX_TRAINED_FRAMES})." + ) + frame_count = align_h3_frames(requested_frames) + if frame_count > H3_MAX_TRAINED_FRAMES: + raise ValueError( + f"Il ponte FLF '{first['label']} -> {last['label']}' diventa {frame_count} frame dopo " + f"l'allineamento H3 17k+5: riduci leggermente la durata relativa del ponte." + ) + ui_bridge = bridge_by_pair.get((_text(first.get("id")), _text(last.get("id"))), {}) + local_prompt = _text(_first_value(ui_bridge, ("prompt", "local_prompt", "relay_prompt"))) or _text(first.get("prompt")) + audio_prompt = _text(_first_value(ui_bridge, ("audio_prompt", "sound_prompt"))) or _text(first.get("audio_prompt")) + bridge_slots.append( + { + "id": _text(ui_bridge.get("id")) or f"flf_bridge_{index + 1}", + "label": _text(ui_bridge.get("label")) or f"{first['label']} -> {last['label']}", + "type": "image", + "start_seconds": cursor, + "requested_frame_count": requested_frames, + "frame_count": frame_count, + "duration_seconds": frame_count / H3_FPS, + "image": _text(first.get("image")), + "explicit_last_image": _text(last.get("image")), + "prompt": local_prompt, + "audio_prompt": audio_prompt, + "transition": "start" if index == 0 else "h3_keyframe_chain", + "use_keyframe": True, + "from_anchor_id": _text(first.get("id")), + "to_anchor_id": _text(last.get("id")), + "visual_start_frame": int(round(centres[index] * H3_FPS)), + "visual_end_frame": int(round(centres[index + 1] * H3_FPS)), + } + ) + cursor += frame_count / H3_FPS + return bridge_slots + + def _compose_prompt( *, global_prompt: str, @@ -341,12 +453,12 @@ def build_shotplan( height: int = 768, acceleration: str = "native", ref_image_size: str = "match", - text_encoder_device: str = "cpu_safe_12gb", + text_encoder_device: str = "auto", reference_roles: list[str] | tuple[str, ...] | None = None, reference_video_role: str = "off", reference_audio_role: str = "off", - sol_conditioning: str = "exact_kv", - spectrum_profile: str = "conservative_3060", + sol_conditioning: str = "exact_kv_and_rows", + spectrum_profile: str = "low_vram", vram_clean_before_decode: bool = True, rife_mode: str = "off", upscale_enabled: bool = False, @@ -382,19 +494,26 @@ def build_shotplan( raise ValueError("aspect ratio H3 deve essere compreso tra 2:5 e 5:2") acceleration = _text(acceleration).lower() or "native" - if acceleration not in {"auto_3060", "native", "h3_sage", "sage", "sage_sol", "spectrum", "sage_spectrum"}: + if acceleration not in { + "auto_3060", "low_vram_auto", "native", "h3_sage", "sage", "sage_sol", "sol_low_vram", + "adaptive_safe", "sol_adaptive_safe", "sol_adaptive_balanced", "spectrum", "sage_spectrum", + }: raise ValueError(f"accelerazione H3 non valida: {acceleration}") ref_image_size = _text(ref_image_size).lower() or "match" if ref_image_size not in {"match", "max"}: raise ValueError(f"ref_image_size H3 non valido: {ref_image_size}") - text_encoder_device = _text(text_encoder_device).lower() or "cpu_safe_12gb" + text_encoder_device = _text(text_encoder_device).lower() or "auto" if text_encoder_device not in {"cpu_safe_12gb", "auto"}: raise ValueError(f"device text encoder H3 non valido: {text_encoder_device}") - sol_conditioning = _text(sol_conditioning).lower() or "exact_kv" + # Old boards remain loadable, but CPU is now an OOM-only fallback handled + # by the atomic conditioning backend rather than a forced placement mode. + if text_encoder_device == "cpu_safe_12gb": + text_encoder_device = "auto" + sol_conditioning = _text(sol_conditioning).lower() or "exact_kv_and_rows" if sol_conditioning not in {"exact_kv", "exact_kv_and_rows"}: raise ValueError(f"Sol-Attn conditioning non valido: {sol_conditioning}") - spectrum_profile = _text(spectrum_profile).lower() or "conservative_3060" - if spectrum_profile not in {"conservative_3060", "conservative_quality", "aggressive"}: + spectrum_profile = _text(spectrum_profile).lower() or "low_vram" + if spectrum_profile not in {"conservative_3060", "low_vram", "conservative_quality", "quality", "aggressive"}: raise ValueError(f"profilo Spectrum non valido: {spectrum_profile}") rife_mode = _text(rife_mode).lower() or "off" if rife_mode not in {"off", "rife_48fps", "rife_60fps"}: @@ -413,6 +532,20 @@ def build_shotplan( fallback_duration = min(H3_MAX_TRAINED_FRAMES / H3_FPS, max(H3_MIN_FRAMES / H3_FPS, 10.0)) slots = _normalise_slots(timeline, duration_seconds, fallback_duration) + requested_task_mode = _text(task_mode).lower() or "auto_from_timeline" + auto_task_mode = requested_task_mode in {"auto", "auto_from_timeline"} + explicit_flf_mode = requested_task_mode in {"flf", "fflf", "fl2va"} + explicit_i2v_mode = requested_task_mode in {"i2v", "i2va"} + image_slots = [slot for slot in slots if _text(slot.get("image"))] + legacy_explicit_last = len(image_slots) == 1 and bool(_text(image_slots[0].get("explicit_last_image"))) + flf_anchor_mode = bool( + (explicit_flf_mode and len(image_slots) >= 2) + or (auto_task_mode and len(image_slots) >= 2) + ) + if flf_anchor_mode: + timeline_duration = _float(timeline.get("duration_seconds"), duration_seconds) + slots = _normalise_flf_bridge_slots(timeline, slots, timeline_duration) + i2v_hard_cut_mode = bool(explicit_i2v_mode and len(image_slots) > 1) chunks: list[dict[str, Any]] = [] prompt_map: list[dict[str, Any]] = [] @@ -420,9 +553,9 @@ def build_shotplan( for slot_index, slot in enumerate(slots): frame_count = int(slot["frame_count"]) - hard_cut_start = slot_index > 0 and slot["transition"] == "hard_cut" + hard_cut_start = slot_index > 0 and (slot["transition"] == "hard_cut" or i2v_hard_cut_mode) next_slot = slots[slot_index + 1] if slot_index + 1 < len(slots) else None - next_is_cut = bool(next_slot and next_slot["transition"] == "hard_cut") + next_is_cut = bool(next_slot and (next_slot["transition"] == "hard_cut" or i2v_hard_cut_mode)) next_anchor = "" if next_slot and not next_is_cut: next_anchor = _text(next_slot.get("image")) @@ -491,23 +624,25 @@ def build_shotplan( ) unique_frames_total += frame_count - overlap - image_count = sum(1 for slot in slots if slot.get("image")) + reference_image_paths = _timeline_image_paths(timeline)[:4] + image_count = len(reference_image_paths) or sum(1 for slot in slots if slot.get("image")) return { "schema": "iamccs.minimax_h3.shotplan", - "schema_version": 5, + "schema_version": 6, "source_timeline_schema": _text(timeline.get("schema")), "fps": H3_FPS, "width": resolved_width, "height": resolved_height, "task_mode": task_mode, "generation_mode": task_mode, - "continuation_mode": "timeline_keyframe_adjacency", + "continuation_mode": "flf_image_center_bridges" if flf_anchor_mode else ("i2v_hard_cuts" if i2v_hard_cut_mode else "timeline_keyframe_adjacency"), "audio_mode": audio_mode, "prompt_mapping": prompt_mapping, "acceleration": acceleration, "ref_image_size": ref_image_size, "text_encoder_device": text_encoder_device, "reference_roles": roles, + "reference_image_paths": reference_image_paths, "reference_video_role": _text(reference_video_role).lower() or "off", "reference_audio_role": _text(reference_audio_role).lower() or "off", "sol_conditioning": sol_conditioning, @@ -516,7 +651,10 @@ def build_shotplan( "rife_mode": rife_mode, "upscale_enabled": bool(_bool(upscale_enabled, False)), "upscale_mode": active_upscale_mode, - "chunk_policy": "one_timeline_box_one_h3_chunk", + "chunk_policy": "n_keyframes_n_minus_one_flf_bridges" if flf_anchor_mode else ("one_i2v_box_one_hard_cut_chunk" if i2v_hard_cut_mode else "one_timeline_box_one_h3_chunk"), + "flf_anchor_mode": flf_anchor_mode, + "i2v_hard_cut_mode": i2v_hard_cut_mode, + "legacy_explicit_last": legacy_explicit_last, "chunk_max_frames": H3_MAX_TRAINED_FRAMES, "global_prompt": _text(global_prompt), "slots": slots, diff --git a/iamccs_prompter.py b/iamccs_prompter.py index a207c20..4557e76 100644 --- a/iamccs_prompter.py +++ b/iamccs_prompter.py @@ -2,10 +2,11 @@ """Structured MiniMax H3 prompt editor and CineLinX injection contract. -The browser editor stores only user-authored project data. This backend is -deliberately deterministic: it formats the selected MiniMax prompt structure, -validates the character budget, and carries an injection request through the -standard IAMCCS CineLinX socket. No API key or network service is required. +The browser editor stores only user-authored project data. The deterministic +path formats MiniMax prompt sections and carries an injection request through +CineLinX. The optional assistant is implemented locally in this module and can +call Ollama or a user-selected compatible provider without wrapping another +custom-node package. """ from __future__ import annotations @@ -24,8 +25,10 @@ from typing import Any SUPERNODE_LINX_TYPE = "IAMCCS_SUPERNODE_LINX" CATEGORY = "IAMCCS/MiniMax H3/Prompting" PROJECT_SCHEMA = "iamccs.minimax_h3.prompter_project" -PROJECT_VERSION = 1 +PROJECT_VERSION = 2 H3_ABSOLUTE_CHAR_LIMIT = 7000 +AI_IMAGE_LIMIT = 4 +AI_IMAGE_MAX_BYTES = 16 * 1024 * 1024 MODE_SECTIONS: dict[str, tuple[tuple[str, str], ...]] = { @@ -121,6 +124,9 @@ def default_project() -> dict[str, Any]: "injection_target": "global", "writing_mode": "guided", "merge_policy": "replace", + "ai_direction": "", + "ai_scope": "active_field", + "ai_visual_roles": {}, "sections": copy.deepcopy(DEFAULT_SECTIONS), } @@ -145,6 +151,10 @@ def _safe_project(value: Any) -> dict[str, Any]: sections = source.get("sections") if isinstance(sections, dict): project["sections"].update({str(key): str(value or "") for key, value in sections.items()}) + project["ai_direction"] = str(project.get("ai_direction") or "") + project["ai_scope"] = str(project.get("ai_scope") or "active_field") + visual_roles = project.get("ai_visual_roles") + project["ai_visual_roles"] = visual_roles if isinstance(visual_roles, dict) else {} project["schema"] = PROJECT_SCHEMA project["schema_version"] = PROJECT_VERSION return project @@ -227,24 +237,88 @@ def _merge_text(existing: str, incoming: str, policy: str) -> str: return f"{old}\n\n{new}" -def _assistant_instruction(task_mode: str, sections: dict[str, str]) -> tuple[str, str]: +def _normalise_ai_images(value: Any) -> list[dict[str, str]]: + images: list[dict[str, str]] = [] + for item in value if isinstance(value, list) else []: + if not isinstance(item, dict) or len(images) >= AI_IMAGE_LIMIT: + continue + data = str(item.get("data") or "").strip() + if data.startswith("data:") and "," in data: + header, data = data.split(",", 1) + guessed = header[5:].split(";", 1)[0] + else: + guessed = "" + data = re.sub(r"\s+", "", data) + if not data: + continue + estimated_bytes = (len(data) * 3) // 4 + if estimated_bytes > AI_IMAGE_MAX_BYTES: + raise ValueError(f"AI reference image exceeds {AI_IMAGE_MAX_BYTES // (1024 * 1024)} MB") + mime_type = str(item.get("mime_type") or guessed or "image/png").strip().lower() + if not mime_type.startswith("image/"): + mime_type = "image/png" + role = str(item.get("role") or "reference").strip().lower() + if role not in {"opening", "closing", "identity", "composition", "style", "reference"}: + role = "reference" + images.append({ + "data": data, + "mime_type": mime_type, + "name": str(item.get("name") or f"Picture {len(images) + 1}").strip(), + "role": role, + "slot": str(item.get("slot") or len(images) + 1), + }) + return images + + +def _assistant_instruction( + task_mode: str, + sections: dict[str, str], + user_direction: str = "", + target_keys: Any = None, + images: Any = None, +) -> tuple[str, str]: mode = str(task_mode or "t2va").lower() if mode not in MODE_SECTIONS: mode = "t2va" allowed = [key for key, _label in MODE_SECTIONS[mode]] - filled = {key: str(sections.get(key, "") or "").strip() for key in allowed} - filled = {key: value for key, value in filled.items() if value} + rough = {key: str(sections.get(key, "") or "").strip() for key in allowed} + filled = {key: value for key, value in rough.items() if value} + selected = [str(key) for key in (target_keys if isinstance(target_keys, list) else []) if str(key) in allowed] + if not selected: + selected = list(filled) + if not selected: + raise ValueError("Select a MiniMax prompt section or write a rough idea before calling the AI") + visuals = _normalise_ai_images(images) + mode_rules = { + "t2va": "Build the requested event from text. Keep the action chronological, filmable and compatible with one continuous audiovisual clip.", + "i2va": "Treat as the exact opening-frame authority. Animate from it without redesigning identity, wardrobe, composition or screen geography.", + "fl2va": "Treat the opening and closing pictures as exact boundary frames. Describe one physically continuous path from the first frame to the last; do not solve the transition with a cut, dissolve or unrelated redesign.", + "ref2va": "Use explicit ,