diff --git a/__init__.py b/__init__.py index 925904a..ee98d1f 100644 --- a/__init__.py +++ b/__init__.py @@ -73,6 +73,7 @@ from .iamccs_ltx2_extension_module import ( IAMCCS_LoadImagesFromDirLite, IAMCCS_SourceFramesToDisk, IAMCCS_StartDirToVideoLatent, + IAMCCS_StartImagesToVideoLatent, IAMCCS_VideoCombineFromDir, IAMCCS_LTX2_ExtensionModule_simple, IAMCCS_LTX2_GetImageFromBatch, @@ -318,6 +319,7 @@ NODE_CLASS_MAPPINGS = { "IAMCCS_LoadImagesFromDirLite": IAMCCS_LoadImagesFromDirLite, "IAMCCS_SourceFramesToDisk": IAMCCS_SourceFramesToDisk, "IAMCCS_StartDirToVideoLatent": IAMCCS_StartDirToVideoLatent, + "IAMCCS_StartImagesToVideoLatent": IAMCCS_StartImagesToVideoLatent, "IAMCCS_VideoCombineFromDir": IAMCCS_VideoCombineFromDir, "IAMCCS_LTX2_ExtensionModule_simple": IAMCCS_LTX2_ExtensionModule_simple, "IAMCCS_LTX2_GetImageFromBatch": IAMCCS_LTX2_GetImageFromBatch, @@ -500,6 +502,7 @@ NODE_DISPLAY_NAME_MAPPINGS = { "IAMCCS_LoadImagesFromDirLite": "Load Images From Dir (Lite) πŸ“", "IAMCCS_SourceFramesToDisk": "Source Frames To Disk πŸ“ΌπŸ’Ύ", "IAMCCS_StartDirToVideoLatent": "Start Dir To Video Latent πŸš€", + "IAMCCS_StartImagesToVideoLatent": "Start Images To Video Latent πŸš€", "IAMCCS_VideoCombineFromDir": "Video Combine From Dir 🎞️", "IAMCCS_LTX2_ExtensionModule_simple": "LTX-2 Extension Module (simple) 🎬", "IAMCCS_LTX2_GetImageFromBatch": "LTX-2 Get Images From Batch 🎞️", diff --git a/iamccs_ltx2_extension_module.py b/iamccs_ltx2_extension_module.py index 5407da4..16653e5 100644 --- a/iamccs_ltx2_extension_module.py +++ b/iamccs_ltx2_extension_module.py @@ -1422,6 +1422,98 @@ class IAMCCS_StartDirToVideoLatent: return ({"samples": samples, "noise_mask": conditioning_latent_frames_mask}, int(images.shape[0]), report) +class IAMCCS_StartImagesToVideoLatent: + @classmethod + def INPUT_TYPES(cls): + return { + "required": { + "start_images": ("IMAGE",), + "vae": ("VAE",), + "latent": ("LATENT",), + "mode": (["all", "from_start", "from_end"], {"default": "all"}), + "count": ("INT", {"default": 9, "min": 1, "max": 512, "step": 1}), + "insert_at_pixel_frame": ("INT", {"default": 0, "min": 0, "max": 100000, "step": 1}), + "strength": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "preprocess": ("BOOLEAN", {"default": True}), + "preprocess_crf": ("INT", {"default": 33, "min": 0, "max": 100, "step": 1}), + } + } + + RETURN_TYPES = ("LATENT", "INT", "STRING") + RETURN_NAMES = ("latent", "frames_loaded", "report") + FUNCTION = "inject" + CATEGORY = "IAMCCS/LTX-2" + + def inject(self, start_images, vae, latent, mode: str, count: int, insert_at_pixel_frame: int, strength: float, preprocess: bool, preprocess_crf: int): + images = start_images + if images is None or getattr(images, "shape", None) is None or int(images.shape[0]) == 0: + raise ValueError("No start images provided") + + count = max(1, int(count)) + if mode == "from_start": + images = images[:count] + elif mode == "from_end": + images = images[-count:] + + if preprocess: + try: + import comfy_extras.nodes_lt as nodes_lt # type: ignore + + images = nodes_lt.LTXVPreprocess().execute(images, int(preprocess_crf))[0] + except Exception as e: + _log.warning("[IAMCCS_StartImagesToVideoLatent] preprocess fallback: %s", e) + + samples = latent["samples"].clone() + scale_factors = getattr(vae, "downscale_index_formula", (8, 32, 32)) + time_scale_factor, height_scale_factor, width_scale_factor = scale_factors + batch, _, latent_frames, latent_height, latent_width = samples.shape + width = latent_width * width_scale_factor + height = latent_height * height_scale_factor + + if images.shape[1] != height or images.shape[2] != width: + try: + import comfy.utils # type: ignore + except Exception as e: + raise ImportError("comfy.utils is required for IAMCCS_StartImagesToVideoLatent") from e + + pixels = comfy.utils.common_upscale(images.movedim(-1, 1), width, height, "bilinear", "center").movedim(1, -1) + else: + pixels = images + + encoded = vae.encode(pixels[:, :, :, :3]) + if isinstance(encoded, dict): + encoded = encoded.get("samples", encoded) + if encoded.ndim == 4: + encoded = encoded.unsqueeze(2) + if encoded.ndim != 5: + raise ValueError(f"Unexpected encoded latent shape: {tuple(encoded.shape)}") + + if encoded.shape[0] != batch: + if encoded.shape[0] == 1 and batch == 1: + pass + elif batch == 1: + encoded = encoded[:1] + else: + raise ValueError("Encoded batch does not match target latent batch") + + if "noise_mask" in latent: + conditioning_latent_frames_mask = latent["noise_mask"].clone() + else: + conditioning_latent_frames_mask = torch.ones((batch, 1, latent_frames, 1, 1), dtype=torch.float32, device=samples.device) + + latent_idx = max(0, min(int(insert_at_pixel_frame) // max(1, int(time_scale_factor)), latent_frames - 1)) + end_index = min(latent_idx + int(encoded.shape[2]), latent_frames) + samples[:, :, latent_idx:end_index] = encoded[:, :, :end_index - latent_idx] + conditioning_latent_frames_mask[:, :, latent_idx:end_index] = 1.0 - float(max(0.0, min(1.0, strength))) + + report = ( + f"Loaded {int(images.shape[0])} start images from input | " + f"insert_pixel={int(insert_at_pixel_frame)} -> latent_idx={latent_idx} | " + f"encoded_t={int(encoded.shape[2])} | replaced={int(end_index - latent_idx)} latent slots" + ) + return ({"samples": samples, "noise_mask": conditioning_latent_frames_mask}, int(images.shape[0]), report) + + class IAMCCS_VideoCombineFromDir: @classmethod def INPUT_TYPES(cls): diff --git a/iamccs_ltx2_tools.py b/iamccs_ltx2_tools.py index 6e5a55d..29ea50c 100644 --- a/iamccs_ltx2_tools.py +++ b/iamccs_ltx2_tools.py @@ -774,6 +774,7 @@ class IAMCCS_SegmentPlanner: continuation_loops = max(0, estimated_segments - 1) remainder = total_frames - unique_segment_frames * max(0, estimated_segments - 1) last_segment_unique_frames = unique_segment_frames if remainder <= 0 else int(remainder) + last_segment_raw_frames = first_segment_raw_frames if estimated_segments <= 1 else self._fix_ltx_frames(last_segment_unique_frames + overlap_frames, str(ltx_round_mode)) clamped_segment_index = min(segment_index, max(0, estimated_segments - 1)) current_segment_start_frames = unique_segment_frames * clamped_segment_index @@ -783,7 +784,12 @@ class IAMCCS_SegmentPlanner: current_segment_start_frames = min(current_segment_start_frames, max(0, total_frames - current_segment_unique_frames)) current_segment_end_frames = min(total_frames, current_segment_start_frames + current_segment_unique_frames) current_remaining_frames_after = max(0, total_frames - current_segment_end_frames) - current_segment_raw_frames = first_segment_raw_frames if clamped_segment_index == 0 else continuation_raw_frames + if clamped_segment_index == 0: + current_segment_raw_frames = first_segment_raw_frames + elif clamped_segment_index >= estimated_segments - 1: + current_segment_raw_frames = last_segment_raw_frames + else: + current_segment_raw_frames = continuation_raw_frames current_segment_start_s = float(current_segment_start_frames) / float(fps) current_segment_end_s = float(current_segment_end_frames) / float(fps) recommended_overlap_frames = int(rec["overlap_frames"]) @@ -981,6 +987,7 @@ class IAMCCS_SegmentPlanFromPlanner: estimated_segments = max(1, int(estimated_segments)) last_segment_unique_frames = max(1, int(last_segment_unique_frames)) segment_index = max(0, min(int(segment_index), estimated_segments - 1)) + overlap_hint_frames = max(0, continuation_raw_frames - unique_segment_frames) current_segment_start_frames = unique_segment_frames * segment_index if segment_index >= estimated_segments - 1: @@ -990,7 +997,13 @@ class IAMCCS_SegmentPlanFromPlanner: current_segment_end_frames = min(total_frames, current_segment_start_frames + current_segment_unique_frames) current_remaining_frames_after = max(0, total_frames - current_segment_end_frames) - current_segment_raw_frames = first_segment_raw_frames if segment_index == 0 else continuation_raw_frames + if segment_index == 0: + current_segment_raw_frames = first_segment_raw_frames + elif segment_index >= estimated_segments - 1: + current_segment_raw_frames = max(1, current_segment_unique_frames + overlap_hint_frames) + current_segment_raw_frames = 1 + 8 * max(0, int(math.ceil(float(current_segment_raw_frames - 1) / 8.0))) + else: + current_segment_raw_frames = continuation_raw_frames current_segment_start_s = float(current_segment_start_frames) / float(fps) current_segment_end_s = float(current_segment_end_frames) / float(fps) current_segment_report = ( diff --git a/iamccs_supernodes_auimg2vid_exec_backend.py b/iamccs_supernodes_auimg2vid_exec_backend.py index 04c7e66..565e654 100644 --- a/iamccs_supernodes_auimg2vid_exec_backend.py +++ b/iamccs_supernodes_auimg2vid_exec_backend.py @@ -32,16 +32,50 @@ from .iamccs_supernodes_linx import SUPERNODE_LINX_TYPE, build_stage_linx_payloa _SAMPLER_NAMES = tuple(comfy.samplers.SAMPLER_NAMES) _REFERENCE_MANUAL_SIGMAS = "1., 0.99375, 0.9875, 0.98125, 0.975, 0.909375, 0.725, 0.421875, 0.0" +_PLANNER_AUDIO_MODES = ( + "melband_vocals_duration_math", + "raw_audio_only", +) +_RENDER_BACKEND_MODES = ( + "auto", + "single_best", + "two_segments_normal_vram", + "loop_normal_vram", + "loop_low_ram_disk", +) +_MODULAR_DECODE_MODES = ( + "inherit_render_backend", + "low_ram_disk", + "normal_tiled_vhs", + "high_vram", + "custom_mode", +) +_VAE_DECODE_MODES = ( + "inherit_render_backend", + "low_ram_disk", + "normal_tiled_vhs", + "high_vram", + "custom_mode", +) try: import folder_paths # type: ignore _LATENT_UPSCALE_MODEL_NAMES = tuple(folder_paths.get_filename_list("latent_upscale_models")) + _MELBAND_MODEL_NAMES = tuple( + name + for name in folder_paths.get_filename_list("diffusion_models") + if "melband" in str(name).lower() or "roformer" in str(name).lower() + ) except Exception: _LATENT_UPSCALE_MODEL_NAMES = ( "ltx-2.3-spatial-upscaler-x2-1.1.safetensors", "ltx-2.3-spatial-upscaler-x2-1.0.safetensors", ) + _MELBAND_MODEL_NAMES = ("MelBandRoformer_fp32.safetensors",) + +if not _MELBAND_MODEL_NAMES: + _MELBAND_MODEL_NAMES = ("MelBandRoformer_fp32.safetensors",) def _node_class(name): @@ -219,22 +253,93 @@ def _decode_settings(decode_backend): } +def _normalize_audio_preprocess_mode(mode): + mapping = { + "melband_vocals": "melband_vocals_duration_math", + "melband_vocals_duration_math": "melband_vocals_duration_math", + "raw_audio": "raw_audio_only", + "raw_audio_only": "raw_audio_only", + } + return mapping.get(str(mode or "melband_vocals_duration_math"), "melband_vocals_duration_math") + + +def _normalize_backend_mode(mode): + mapping = { + "auto_from_plan": "auto", + "auto": "auto", + "single_workflow1_best": "single_best", + "single_best": "single_best", + "two_segments_normal_vram": "two_segments_normal_vram", + "loop_normal_vram": "loop_normal_vram", + "loop_low_ram_disk": "loop_low_ram_disk", + } + value = mapping.get(str(mode or "auto"), "auto") + if value in _RENDER_BACKEND_MODES: + return value + return "auto" + + +def _normalize_modular_decode_mode(mode, backend_mode=None): + mapping = { + "inherit_render_backend": "inherit_from_backend", + "inherit_from_render": "inherit_from_backend", + "inherit_from_backend": "inherit_from_backend", + "low_ram": "low_ram_disk", + "low_ram_disk": "low_ram_disk", + "normal": "normal_tiled_vhs_ready", + "normal_tiled_vhs": "normal_tiled_vhs_ready", + "normal_tiled_vhs_ready": "normal_tiled_vhs_ready", + "high": "high_vram_direct", + "high_vram": "high_vram_direct", + "high_vram_direct": "high_vram_direct", + "custom_mode": "custom_mode", + } + normalized = mapping.get(str(mode or "inherit_from_backend"), "inherit_from_backend") + if normalized != "inherit_from_backend": + return normalized + if _normalize_backend_mode(backend_mode) == "loop_low_ram_disk": + return "low_ram_disk" + return "normal_tiled_vhs_ready" + + def _modular_decode_to_vae_mode(modular_decode): mapping = { + "inherit_from_backend": "normal_tiled", "low_ram": "low_ram_disk", + "low_ram_disk": "low_ram_disk", "normal": "normal_tiled", + "normal_tiled_vhs_ready": "normal_tiled", "high": "high_vram", + "high_vram_direct": "high_vram", "custom_mode": "custom_mode", } return mapping.get(str(modular_decode), "low_ram_disk") +def _resolve_backend_route(requested_backend_mode, planner_segment_count, modular_decode): + backend_mode = _normalize_backend_mode(requested_backend_mode) + planner_segment_count = max(1, int(planner_segment_count or 1)) + modular_decode = _normalize_modular_decode_mode(modular_decode, backend_mode) + if backend_mode == "single_best": + return backend_mode, 1, True, False, modular_decode + if backend_mode == "two_segments_normal_vram": + return backend_mode, 2, False, True, "normal_tiled_vhs_ready" + if backend_mode == "loop_normal_vram": + return backend_mode, max(2, planner_segment_count), False, True, "normal_tiled_vhs_ready" + if backend_mode == "loop_low_ram_disk": + return backend_mode, max(2, planner_segment_count), False, False, "low_ram_disk" + use_single_best = planner_segment_count <= 1 + use_in_memory_loop = (not use_single_best) and modular_decode != "low_ram_disk" + resolved_backend = "single_best" if use_single_best else ("loop_normal_vram" if use_in_memory_loop else "loop_low_ram_disk") + return resolved_backend, planner_segment_count, use_single_best, use_in_memory_loop, modular_decode + + def _render_status(duration_seconds, planner_data, modular_decode, continuity_mode, anti_drift_mode, second_stage_enabled, audio_concat_enabled): return ( f"duration {float(duration_seconds):.2f}s | fps {float(planner_data['fps']):.2f} | total {int(_to_int(planner_data, 'total_frames', 0))}f | " f"segments {int(_to_int(planner_data, 'segment_count', 0))} | first {int(_to_int(planner_data, 'first_segment_raw_frames', 0))}f | " f"loop {int(_to_int(planner_data, 'continuation_raw_frames', 0))}f | overlap {int(_to_int(planner_data, 'recommended_overlap_frames', 0))}f | " - f"decode {modular_decode} | stage2 {'on' if second_stage_enabled else 'off'} | audio_concat {'on' if audio_concat_enabled else 'off'} | continuity {continuity_mode} | anti_drift {anti_drift_mode}" + f"vae {modular_decode} | stage2 {'on' if second_stage_enabled else 'off'} | audio_concat {'on' if audio_concat_enabled else 'off'} | continuity {continuity_mode} | anti_drift {anti_drift_mode}" ) @@ -260,6 +365,173 @@ def _manual_sigmas(sigmas_text): return torch.tensor(values, dtype=torch.float32) +def _node_output_tuple(value): + current = value + while isinstance(current, tuple) and len(current) == 1: + current = current[0] + if isinstance(current, tuple): + return current + if isinstance(current, list): + return tuple(current) + args = getattr(current, "args", None) + if isinstance(args, (tuple, list)): + return tuple(args) + nested = getattr(current, "value", None) + if nested is not None and nested is not current: + return _node_output_tuple(nested) + return (current,) + + +def _invoke_node(node, method_names, positional_variants=None, keyword_variants=None): + positional_variants = positional_variants or [] + keyword_variants = keyword_variants or [] + last_error = None + for method_name in method_names: + fn = getattr(node, method_name, None) + if fn is None: + continue + for kwargs in keyword_variants: + try: + return _node_output_tuple(fn(**kwargs)) + except TypeError as exc: + last_error = exc + except Exception as exc: + last_error = exc + for args in positional_variants: + try: + return _node_output_tuple(fn(*args)) + except TypeError as exc: + last_error = exc + except Exception as exc: + last_error = exc + if last_error is not None: + raise last_error + raise RuntimeError(f"No callable method found on node {type(node).__name__}") + + +def _compute_audio_duration_seconds(audio): + try: + node = _node_class("Audio Duration (mtb)")() + outputs = _invoke_node( + node, + ("execute", "get_duration", "duration"), + positional_variants=[(audio,)], + keyword_variants=[{"audio": audio}], + ) + duration_ms = int(outputs[0]) + return float(duration_ms) * 0.001, "mtb" + except Exception: + return _audio_duration_seconds(audio), "waveform" + + +def _prepare_planner_audio(audio, audio_preprocess_mode, melband_model_name): + mode = _normalize_audio_preprocess_mode(audio_preprocess_mode) + model_name = str(melband_model_name or "MelBandRoformer_fp32.safetensors") + raw_seconds, raw_source = _compute_audio_duration_seconds(audio) + result = { + "raw_audio": audio, + "conditioning_audio_single": audio, + "conditioning_audio_segmented": audio, + "duration_audio": audio, + "duration_seconds": raw_seconds, + "duration_source": raw_source, + "preprocess_report": f"audio_preprocess=raw_audio_only | duration_source={raw_source}", + "melband_enabled": False, + } + if mode != "melband_vocals_duration_math": + return result + + try: + loader = _node_class("MelBandRoFormerModelLoader")() + melband_model = _invoke_node( + loader, + ("execute", "load", "load_model", "loadmodel"), + positional_variants=[(model_name,)], + keyword_variants=[{"model_name": model_name}, {"model": model_name}], + )[0] + sampler = _node_class("MelBandRoFormerSampler")() + vocals = _invoke_node( + sampler, + ("execute", "sample", "process"), + positional_variants=[(melband_model, audio)], + keyword_variants=[{"model": melband_model, "audio": audio}], + )[0] + duration_seconds, duration_source = _compute_audio_duration_seconds(vocals) + result.update({ + "conditioning_audio_single": vocals, + "duration_audio": vocals, + "duration_seconds": duration_seconds, + "duration_source": duration_source, + "preprocess_report": f"audio_preprocess=melband_vocals_duration_math | model={model_name} | duration_source={duration_source}", + "melband_enabled": True, + }) + except Exception as exc: + result["preprocess_report"] = f"audio_preprocess=raw_audio_fallback | reason={exc} | duration_source={raw_source}" + return result + + +def _scheduler_sigmas(model, scheduler_name="simple", steps=8, denoise=1.0): + try: + node = _node_class("BasicScheduler")() + outputs = _invoke_node( + node, + ("get_sigmas", "execute"), + positional_variants=[ + (model, scheduler_name, int(steps), float(denoise)), + (scheduler_name, int(steps), float(denoise), model), + ], + keyword_variants=[ + {"model": model, "scheduler": scheduler_name, "steps": int(steps), "denoise": float(denoise)}, + {"scheduler": scheduler_name, "steps": int(steps), "denoise": float(denoise), "model": model}, + ], + ) + return outputs[0] + except Exception: + return _manual_sigmas(_REFERENCE_MANUAL_SIGMAS) + + +def _decode_images_in_memory(video_latent, vae, tile_size=512, overlap=64): + decode_node = _node_class("IAMCCS_VAEDecodeTiledSafe")() + return decode_node.decode( + video_latent, + vae, + True, + "auto", + int(tile_size), + int(overlap), + 256, + 32, + False, + False, + 0, + )[0] + + +def _load_images_from_dir_for_output(frames_dir): + import numpy as np + from PIL import Image + + path = str(frames_dir or "").strip() + if not path or not os.path.isdir(path): + return torch.zeros((1, 1, 1, 3), dtype=torch.float32) + + files = [] + for name in sorted(os.listdir(path)): + lower = name.lower() + if lower.endswith((".png", ".jpg", ".jpeg", ".webp")): + files.append(os.path.join(path, name)) + if not files: + return torch.zeros((1, 1, 1, 3), dtype=torch.float32) + + images = [] + for file_path in files: + with Image.open(file_path) as image: + rgb = image.convert("RGB") + array = np.asarray(rgb).astype("float32") / 255.0 + images.append(torch.from_numpy(array)) + return torch.stack(images, dim=0) + + def _resize_image_to(image, width, height): if image is None: raise ValueError("image is required for resize") @@ -401,14 +673,35 @@ def _load_guidance_image_from_dir(path, fallback_image=None, pick_mode="latest") def _resolve_stage2_model(primary_model, second_stage_model, second_stage_payload): data = _parse_payload(second_stage_payload) - policy = _to_text(data, "stage2_model_policy", "replace_stage1_if_connected") + policy = _to_text(data, "stage2_model_policy", "stage2_model_if_connected") if second_stage_model is None: return primary_model - if policy in {"replace_stage1_if_connected", "prefer_stage2_else_primary"}: + if policy in {"stage2_model_if_connected", "replace_stage1_if_connected", "prefer_stage2_else_primary"}: return second_stage_model return primary_model +def _stage2_payload_from_exec_widgets( + second_stage_mode, + stage2_model_policy, + second_stage_upscale_model, + second_stage_reinject_strength, + second_stage_cfg, + second_stage_manual_sigmas, +): + scale_mode = "x2_latent_upscale_beta" if str(second_stage_mode) == "latent_upscale_refine_x2_beta" else "same_resolution_refine" + return ( + f"second_stage_mode={second_stage_mode}; " + f"stage2_model_policy={stage2_model_policy}; " + f"second_stage_upscale_model={second_stage_upscale_model}; " + f"second_stage_scale_mode={scale_mode}; " + f"second_stage_reinject_strength={float(second_stage_reinject_strength)}; " + f"second_stage_cfg={float(second_stage_cfg)}; " + f"second_stage_manual_sigmas={second_stage_manual_sigmas}; " + f"second_stage_steps={max(0, len([x for x in str(second_stage_manual_sigmas).replace(chr(10), ',').split(',') if x.strip()]) - 1)}" + ) + + def _continuity_mode_from_payload(continuity_payload, fallback_mode, fallback_interval, fallback_strength): data = _parse_payload(continuity_payload) if not data: @@ -540,6 +833,28 @@ def _hard_unload_all_models(): pass +def _accelerate_exec_model_if_available(model): + try: + accelerator = _node_class("IAMCCS_GGUF_accelerator")() + except Exception as exc: + return model, f"gguf_accelerator=unavailable reason={exc}" + + try: + accelerated_model, accelerator_report = accelerator.accelerate( + model, + "auto_oom_safe", + True, + True, + 1500, + True, + "all_or_nothing", + 1024, + ) + return accelerated_model, str(accelerator_report) + except Exception as exc: + return model, f"gguf_accelerator=fallback reason={exc}" + + class IAMCCS_GC_AUIMG2VIDExecutablePlanner: CATEGORY = "IAMCCS/GoyAIcanvas/TestBackends" FUNCTION = "plan" @@ -567,6 +882,8 @@ class IAMCCS_GC_AUIMG2VIDExecutablePlanner: "segment_preset": (["10sec", "15sec", "20sec"], {"default": "15sec"}), "overlap_frames": ("INT", {"default": 9, "min": 0, "max": 4096, "step": 1}), "ltx_round_mode": (["up", "nearest", "down"], {"default": "up"}), + "audio_preprocess_mode": (_PLANNER_AUDIO_MODES, {"default": "melband_vocals_duration_math"}), + "melband_model_name": (_MELBAND_MODEL_NAMES, {"default": _MELBAND_MODEL_NAMES[0] if _MELBAND_MODEL_NAMES else "MelBandRoformer_fp32.safetensors"}), } , "optional": { @@ -579,12 +896,14 @@ class IAMCCS_GC_AUIMG2VIDExecutablePlanner: } } - def plan(self, audio, fps, segment_seconds, planning_mode, segment_preset, overlap_frames, ltx_round_mode, model=None, clip=None, vae=None, audio_vae=None, linx=None, audio_concat_payload=""): + def plan(self, audio, fps, segment_seconds, planning_mode, segment_preset, overlap_frames, ltx_round_mode, audio_preprocess_mode, melband_model_name, model=None, clip=None, vae=None, audio_vae=None, linx=None, audio_concat_payload=""): planning_mode = _normalize_planner_mode(planning_mode) segment_preset = _normalize_segment_preset(segment_preset) + audio_preprocess_mode = _normalize_audio_preprocess_mode(audio_preprocess_mode) + audio_plan = _prepare_planner_audio(audio, audio_preprocess_mode, melband_model_name) duration_seconds = _duration_hint_from_payload(audio_concat_payload) if duration_seconds is None: - duration_seconds = _audio_duration_seconds(audio) + duration_seconds = float(audio_plan["duration_seconds"]) planner = _node_class("IAMCCS_SegmentPlanner")() planned = planner.plan( duration_seconds, @@ -609,7 +928,9 @@ class IAMCCS_GC_AUIMG2VIDExecutablePlanner: report = ( f"Executable planner. duration={duration_seconds:.3f}s @ {float(fps):.3f}fps | " f"segments={int(planned[4])} | first_raw={int(planned[2])}f | continuation_raw={int(planned[3])}f | " - f"recommended_overlap={int(planned[18])}f | left_context={float(planned[19]):.3f}s" + f"recommended_overlap={int(planned[18])}f | left_context={float(planned[19]):.3f}s | " + f"audio_preprocess_mode={audio_preprocess_mode} | melband_model={melband_model_name} | " + f"melband_enabled={bool(audio_plan['melband_enabled'])} | {audio_plan['preprocess_report']}" ) linx = build_stage_linx_payload( linx, @@ -624,6 +945,8 @@ class IAMCCS_GC_AUIMG2VIDExecutablePlanner: "content_profile": str(segment_preset), "recommended_extension_preset": str(planned[20]), "audio_concat_duration_override": float(duration_seconds), + "audio_preprocess_mode": str(audio_preprocess_mode), + "melband_model_name": str(melband_model_name), }, report, slot_map={ @@ -649,12 +972,18 @@ class IAMCCS_GC_AUIMG2VIDExecutablePlanner: }, resources={ "audio": audio, + "audio_raw": audio_plan["raw_audio"], + "audio_conditioning_single": audio_plan["conditioning_audio_single"], + "audio_conditioning_segmented": audio_plan["conditioning_audio_segmented"], + "audio_duration_source": audio_plan["duration_audio"], "model": model, "clip": clip, "vae": vae, "audio_vae": audio_vae, "fps": float(fps), "planner_payload": payload, + "audio_preprocess_report": audio_plan["preprocess_report"], + "melband_enabled": bool(audio_plan["melband_enabled"]), }, ) return ( @@ -681,7 +1010,8 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: return { "required": { "image": ("IMAGE",), - "ui_preset": (["low_ram_safe", "balanced", "high_quality", "fast_preview", "custom"], {"default": "balanced"}), + "generation_mode": (["img2vid", "t2v"], {"default": "img2vid"}), + "backend_mode": (_RENDER_BACKEND_MODES, {"default": "auto"}), "positive_text": ("STRING", {"default": "cinematic motion, detailed scene", "multiline": True}), "negative_text": ("STRING", {"default": "blurry, low quality, artifacts", "multiline": True}), "width": ("INT", {"default": 768, "min": 64, "max": 8192, "step": 32}), @@ -709,11 +1039,17 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: "anti_drift_mode": (["off", "rolling_adain", "dual_reference_adain"], {"default": "off"}), "anti_drift_strength": ("FLOAT", {"default": 0.18, "min": 0.0, "max": 1.0, "step": 0.01}), "identity_persistence_strength": ("FLOAT", {"default": 0.08, "min": 0.0, "max": 1.0, "step": 0.01}), - "modular_decode": (["low_ram", "normal", "high", "custom_mode"], {"default": "low_ram"}), + "vae_mode": (_MODULAR_DECODE_MODES, {"default": "inherit_render_backend"}), "downstream_stage_mode": (["finalize_only", "upscale_ready", "detailer_ready", "upscale_then_detailer"], {"default": "finalize_only"}), "output_root": ("STRING", {"default": "iamccs_gc_auimg2vid/exec_run"}), "segment_overlay_mode": (["off", "segment_label", "custom_text"], {"default": "off"}), "segment_overlay_text": ("STRING", {"default": "seg {segment_number}/{segment_count}", "multiline": True}), + "second_stage_mode": (["off", "latent_refine_3step", "latent_upscale_refine_x2_beta"], {"default": "off"}), + "stage2_model_policy": (["stage2_model_if_connected", "keep_stage1_model"], {"default": "stage2_model_if_connected"}), + "second_stage_upscale_model": (_LATENT_UPSCALE_MODEL_NAMES, {"default": _LATENT_UPSCALE_MODEL_NAMES[0] if _LATENT_UPSCALE_MODEL_NAMES else "ltx-2.3-spatial-upscaler-x2-1.1.safetensors"}), + "second_stage_reinject_strength": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.01}), + "second_stage_cfg": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 30.0, "step": 0.1}), + "second_stage_manual_sigmas": ("STRING", {"default": "0.909375, 0.725, 0.421875, 0.0", "multiline": True}), }, "optional": { "linx": (SUPERNODE_LINX_TYPE,), @@ -724,6 +1060,8 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: "audio_vae": ("VAE",), "plan_payload": ("STRING",), "refresh_image": ("IMAGE",), + "second_stage_linx": (SUPERNODE_LINX_TYPE,), + "stage2_model": ("MODEL",), }, "hidden": { "unique_id": "UNIQUE_ID", @@ -760,7 +1098,8 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: def render( self, image, - ui_preset, + generation_mode, + backend_mode, positive_text, negative_text, width, @@ -788,11 +1127,17 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: anti_drift_mode, anti_drift_strength, identity_persistence_strength, - modular_decode, + vae_mode, downstream_stage_mode, output_root, segment_overlay_mode, segment_overlay_text, + second_stage_mode, + stage2_model_policy, + second_stage_upscale_model, + second_stage_reinject_strength, + second_stage_cfg, + second_stage_manual_sigmas, linx=None, audio=None, model=None, @@ -801,9 +1146,10 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: audio_vae=None, plan_payload="", refresh_image=None, + second_stage_linx=None, + stage2_model=None, unique_id=None, ): - del ui_preset audio = _require_runtime_value(_input_or_linx(audio, linx, "audio"), "audio") model = _require_runtime_value(_input_or_linx(model, linx, "model"), "model") clip = _require_runtime_value(_input_or_linx(clip, linx, "clip"), "clip") @@ -813,13 +1159,35 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: if not plan_payload: raise ValueError("plan_payload is required either as a local input or via linx inheritance") + raw_audio = _input_or_linx(audio, linx, "audio_raw") or audio + single_conditioning_audio = _input_or_linx(None, linx, "audio_conditioning_single") or raw_audio + segmented_audio = _input_or_linx(None, linx, "audio_conditioning_segmented") or raw_audio + audio_preprocess_report = str(linx_resource(linx, "audio_preprocess_report", "audio_preprocess=unknown") or "audio_preprocess=unknown") + melband_enabled = bool(linx_resource(linx, "melband_enabled", False)) + fps_value = float(linx_resource(linx, "fps", 24.0) or 24.0) - modular_decode = str(_inherit_widget_value(modular_decode, "low_ram", linx, "decode_mode")) + backend_mode = _normalize_backend_mode(backend_mode) + modular_decode = _normalize_modular_decode_mode( + _inherit_widget_value(vae_mode, "inherit_render_backend", linx, "decode_mode"), + backend_mode, + ) output_root = str(_inherit_widget_value(output_root, "iamccs_gc_auimg2vid/exec_run", linx, "output_root")) audio_concat_payload = str(linx_output(linx, "audio_concat_payload", "") or "") continuity_payload = str(linx_output(linx, "continuity_payload", "") or "") - second_stage_payload = str(linx_output(linx, "second_stage_payload", "") or "") - second_stage_model = linx_resource(linx, "second_stage_model", None) + generation_mode = str(generation_mode or "img2vid") + second_stage_payload = _stage2_payload_from_exec_widgets( + second_stage_mode, + stage2_model_policy, + second_stage_upscale_model, + second_stage_reinject_strength, + second_stage_cfg, + second_stage_manual_sigmas, + ) + second_stage_model = stage2_model + if second_stage_model is None and isinstance(second_stage_linx, dict): + second_stage_model = linx_resource(second_stage_linx, "second_stage_model", None) + if second_stage_model is None: + second_stage_model = linx_resource(linx, "second_stage_model", None) planner_settings = self._planner_settings( plan_payload, @@ -879,10 +1247,15 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: planner_settings["ltx_round_mode"], 0, ) - segment_count = int(planner_head[4]) + planner_segment_count = int(planner_head[4]) recommended_left_context = float(planner_head[19]) if float(audio_left_context_s) <= 0.0: audio_left_context_s = recommended_left_context + backend_mode, segment_count, use_single_best, use_in_memory_loop, modular_decode = _resolve_backend_route( + backend_mode, + planner_segment_count, + modular_decode, + ) planner_report_line = ( f"Planner settings used. mode={planner_settings['planning_mode']} | " @@ -911,17 +1284,180 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: stage2_model_active = _resolve_stage2_model(model, second_stage_model, second_stage_payload) stage2_data = _parse_payload(second_stage_payload) second_stage_mode = _to_text(stage2_data, "second_stage_mode", "off") + second_stage_scale_mode = _to_text(stage2_data, "second_stage_scale_mode", "same_resolution_refine") second_stage_upscale_model_name = _to_text( stage2_data, "second_stage_upscale_model", _LATENT_UPSCALE_MODEL_NAMES[0] if _LATENT_UPSCALE_MODEL_NAMES else "ltx-2.3-spatial-upscaler-x2-1.1.safetensors", ) - second_stage_reinject_strength = _to_float(stage2_data, "second_stage_reinject_strength", 1.0) + second_stage_reinject_strength = _to_float(stage2_data, "second_stage_reinject_strength", 0.0) second_stage_cfg = _to_float(stage2_data, "second_stage_cfg", 1.0) second_stage_manual_sigmas = _to_text(stage2_data, "second_stage_manual_sigmas", "0.909375, 0.725, 0.421875, 0.0") + second_stage_step_count = _to_int(stage2_data, "second_stage_steps", 3) + second_stage_model_source = "stage2_model" if second_stage_model is not None and stage2_model_active is second_stage_model else "stage1_model" identity_reference_latent = None rolling_reference_latent = None + if use_single_best: + total_frames = max(1, _to_int(_parse_payload(plan_payload), "total_frames", int(planner_head[0]) or 1)) + video_latent = EmptyLTXVLatentVideo.execute(int(width), int(height), total_frames, 1)[0] + if str(generation_mode) != "t2v": + preprocessed_image = LTXVPreprocess.execute(image, int(image_compression))[0] + video_latent = LTXVImgToVideoInplace.execute(vae, preprocessed_image, video_latent, float(image_strength), False)[0] + + audio_latent = LTXVAudioVAEEncode.execute(single_conditioning_audio, audio_vae)[0] + audio_mask = SolidMask.execute(0.0, 1024, 1024)[0] + audio_latent = comfy_nodes.SetLatentNoiseMask().set_mask(audio_latent, audio_mask)[0] + _soft_cleanup() + av_latent = LTXVConcatAVLatent.execute(video_latent, audio_latent)[0] + accelerated_model, accelerator_report = _accelerate_exec_model_if_available(model) + model_for_segment = accelerated_model + guider = CFGGuider.execute(model_for_segment, conditioned_positive, conditioned_negative, float(cfg))[0] + sampler = KSamplerSelect.execute("lcm")[0] + sigmas = _scheduler_sigmas(model_for_segment, "simple", 8, 1.0) + noise = RandomNoise.execute(int(seed))[0] + sampled_av = _node_class("IAMCCS_SamplerAdvancedVersion1")().sample( + noise, + guider, + sampler, + sigmas, + av_latent, + True, + True, + )[0] + sampled_video, sampled_audio_latent = LTXVSeparateAVLatent.execute(sampled_av) + sampled_video = LTXVCropGuides.execute(conditioned_positive, conditioned_negative, sampled_video)[2] + + stage2_model_active = _resolve_stage2_model(model, second_stage_model, second_stage_payload) + stage2_data = _parse_payload(second_stage_payload) + second_stage_mode = _to_text(stage2_data, "second_stage_mode", "off") + second_stage_scale_mode = _to_text(stage2_data, "second_stage_scale_mode", "same_resolution_refine") + second_stage_upscale_model_name = _to_text( + stage2_data, + "second_stage_upscale_model", + _LATENT_UPSCALE_MODEL_NAMES[0] if _LATENT_UPSCALE_MODEL_NAMES else "ltx-2.3-spatial-upscaler-x2-1.1.safetensors", + ) + second_stage_reinject_strength = _to_float(stage2_data, "second_stage_reinject_strength", 0.0) + second_stage_cfg = _to_float(stage2_data, "second_stage_cfg", 1.0) + second_stage_manual_sigmas = _to_text(stage2_data, "second_stage_manual_sigmas", "0.909375, 0.725, 0.421875, 0.0") + second_stage_step_count = _to_int(stage2_data, "second_stage_steps", 3) + second_stage_model_source = "stage2_model" if second_stage_model is not None and stage2_model_active is second_stage_model else "stage1_model" + + if str(second_stage_mode) in {"latent_refine_3step", "latent_upscale_refine", "latent_upscale_refine_x2_beta"}: + stage2_positive, stage2_negative, cropped_video_latent = LTXVCropGuides.execute( + conditioned_positive, + conditioned_negative, + sampled_video, + ) + if str(second_stage_scale_mode) == "x2_latent_upscale_beta": + upscale_model = LatentUpscaleModelLoader.execute(str(second_stage_upscale_model_name))[0] + stage2_video_latent = LTXVLatentUpsampler().upsample_latent( + cropped_video_latent, + upscale_model, + vae, + )[0] + else: + stage2_video_latent = cropped_video_latent + if float(second_stage_reinject_strength) > 0.0 and str(generation_mode) != "t2v": + resized_guidance_image = _resize_image_to(image, int(width), int(height)) + preprocessed_guidance = LTXVPreprocess.execute(resized_guidance_image, int(image_compression))[0] + reinjected_video_latent = LTXVImgToVideoInplace.execute( + vae, + preprocessed_guidance, + stage2_video_latent, + float(second_stage_reinject_strength), + False, + )[0] + else: + reinjected_video_latent = stage2_video_latent + latent_stage2 = LTXVConcatAVLatent.execute(reinjected_video_latent, sampled_audio_latent)[0] + model_stage2 = ModelSamplingLTXV.execute(stage2_model_active, float(max_shift), float(base_shift), latent_stage2)[0] + guider_stage2 = CFGGuider.execute(model_stage2, stage2_positive, stage2_negative, float(second_stage_cfg))[0] + sampler_stage2 = KSamplerSelect.execute("euler")[0] + sigmas_stage2 = _manual_sigmas(second_stage_manual_sigmas) + noise_stage2 = RandomNoise.execute(int(seed))[0] + sampled_stage2_av = SamplerCustomAdvanced.sample( + noise_stage2, + guider_stage2, + sampler_stage2, + sigmas_stage2, + latent_stage2, + )[0] + sampled_video = LTXVSeparateAVLatent.execute(sampled_stage2_av)[0] + + report = ( + f"duration {float(total_duration_seconds):.2f}s | fps {float(fps_value):.2f} | total {int(total_frames)}f | segments 1 | " + f"backend_requested={backend_mode} | backend_resolved=single_best | generation_mode {generation_mode} | " + f"stage2 {'on' if str(second_stage_mode) != 'off' else 'off'} | decode_mode={modular_decode} | " + f"conditioning {'melband_vocals_duration_math' if single_conditioning_audio is not raw_audio else 'raw_audio_only'} | " + f"melband_enabled={melband_enabled}\n" + f"Planner settings used. mode={planner_settings['planning_mode']} | segment_preset={planner_settings['segment_preset']} | " + f"segment_seconds={float(planner_settings['segment_seconds']):.3f}s | overlap={int(planner_settings['overlap_frames'])}f | " + f"ltx_round={planner_settings['ltx_round_mode']}\n" + f"Audio preprocess. {audio_preprocess_report}\n" + f"Single route details. sampler=lcm | scheduler=simple(steps=8, denoise=1.0) | sampler_node=IAMCCS_SamplerAdvancedVersion1 | cleanup_before_sampling=soft_cleanup | model_sampling=workflow_single_match(no_extra_ModelSamplingLTXV) | {accelerator_report}\n" + f"Executable AU+IMG2VID render completed. single generation backend=workflow1_best | latent handed to VAE stage" + ) + render_linx = build_stage_linx_payload( + linx, + "exec_render", + "render", + { + "pipeline_kind": "au_img2vid_exec", + "backend_mode": str(backend_mode), + "generation_mode": str(generation_mode), + "fps": float(fps_value), + "modular_decode": str(modular_decode), + "segment_count": 1, + "segments_rendered": 1, + "second_stage_mode": str(second_stage_mode), + "second_stage_scale_mode": str(second_stage_scale_mode), + "second_stage_steps": int(second_stage_step_count), + "second_stage_model_source": str(second_stage_model_source), + }, + report, + unique_id=unique_id, + slot_map={ + "frames_dir": {"type": "STRING", "role": "rendered_frames_dir"}, + "start_dir": {"type": "STRING", "role": "rendered_start_dir"}, + "linx": {"type": SUPERNODE_LINX_TYPE, "role": "stage_linx"}, + }, + downstream_stages=_downstream_stage_hints(downstream_stage_mode), + policies={ + "decode_mode": str(modular_decode), + "second_stage_mode": str(second_stage_mode), + }, + outputs={ + "frames_dir": "", + "start_dir": "", + "segments_rendered": 1, + "estimated_duration_seconds": float(total_duration_seconds), + "render_status": _first_line(report), + }, + resources={ + "audio": raw_audio, + "model": model, + "clip": clip, + "vae": vae, + "audio_vae": audio_vae, + "fps": float(fps_value), + "decode_mode": str(modular_decode), + "output_root": str(output_root), + "planner_payload": str(plan_payload), + "generation_mode": str(generation_mode), + "video_latent": sampled_video, + "rendered_images": None, + "second_stage_model": stage2_model_active, + "second_stage_payload": second_stage_payload, + }, + ) + return ("", "", 1, float(total_duration_seconds), render_linx, report) + + extension_node_mem = _node_class("IAMCCS_LTX2_ExtensionModule")() if use_in_memory_loop else None + start_inject_images_node = _node_class("IAMCCS_StartImagesToVideoLatent")() if use_in_memory_loop else None + current_extended_images = None + current_start_images = None + for segment_index in range(segment_count): plan_segment = planner_node.plan( total_duration_seconds, @@ -954,7 +1490,7 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: is_last_segment = int(math_out[8]) conditioning_audio = audio_extender_node.slice_segment( - audio, + segmented_audio, fps_value, audio_context_mode, float(audio_left_context_s), @@ -974,7 +1510,9 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: )[0] video_latent = EmptyLTXVLatentVideo.execute(int(width), int(height), current_segment_raw_frames, 1)[0] - uses_source_anchor = _use_source_anchor(segment_index, continuity_settings["mode"], continuity_settings["interval"]) + is_t2v = generation_mode == "t2v" + uses_source_anchor = False if is_t2v else _use_source_anchor(segment_index, continuity_settings["mode"], continuity_settings["interval"]) + init_mode = "t2v_empty" if is_t2v and segment_index == 0 else "tail" if uses_source_anchor: anchor_image = refresh_source_image if segment_index > 0 and continuity_settings["mode"] == "periodic_source_refresh": @@ -982,18 +1520,32 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: preprocessed_image = LTXVPreprocess.execute(anchor_image, int(image_compression))[0] source_strength = float(image_strength if segment_index == 0 else continuity_settings["strength"]) video_latent = LTXVImgToVideoInplace.execute(vae, preprocessed_image, video_latent, source_strength, False)[0] - else: - video_latent = start_inject_node.inject( - start_dir, - vae, - video_latent, - "all", - max(1, overlap_frames_value), - 0, - float(image_strength), - True, - int(image_compression), - )[0] + init_mode = "source" + elif segment_index > 0: + if use_in_memory_loop: + video_latent = start_inject_images_node.inject( + current_start_images, + vae, + video_latent, + "all", + max(1, overlap_frames_value), + 0, + float(image_strength), + True, + int(image_compression), + )[0] + else: + video_latent = start_inject_node.inject( + start_dir, + vae, + video_latent, + "all", + max(1, overlap_frames_value), + 0, + float(image_strength), + True, + int(image_compression), + )[0] audio_latent = LTXVAudioVAEEncode.execute(conditioning_audio, audio_vae)[0] audio_mask = SolidMask.execute(0.0, 1024, 1024)[0] @@ -1004,10 +1556,10 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: av_latent = LTXVConcatAVLatent.execute(video_latent, audio_latent)[0] model_for_segment = ModelSamplingLTXV.execute(model, float(max_shift), float(base_shift), av_latent)[0] guider = CFGGuider.execute(model_for_segment, segment_positive, segment_negative, float(cfg))[0] - sampler = KSamplerSelect.execute(str(sampler_name))[0] - sigmas = _manual_sigmas(manual_sigmas) + sampler = KSamplerSelect.execute("lcm")[0] + sigmas = _scheduler_sigmas(model_for_segment, "simple", 8, 1.0) noise = RandomNoise.execute(int(seed) + segment_index)[0] - sampled_av = SamplerCustomAdvanced.sample(noise, guider, sampler, sigmas, av_latent)[1] + sampled_av = SamplerCustomAdvanced.sample(noise, guider, sampler, sigmas, av_latent)[0] sampled_video, sampled_audio_latent = LTXVSeparateAVLatent.execute(sampled_av) segment_positive, segment_negative, sampled_video = LTXVCropGuides.execute( segment_positive, @@ -1017,7 +1569,7 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: stage_mode = str(second_stage_mode) stage2_applied = False - if stage_mode == "latent_upscale_refine": + if stage_mode in {"latent_refine_3step", "latent_upscale_refine", "latent_upscale_refine_x2_beta"}: guidance_source = "refresh" guidance_image = refresh_source_image if not uses_source_anchor: @@ -1028,25 +1580,31 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: conditioned_negative, sampled_video, ) - upscale_model = LatentUpscaleModelLoader.execute(str(second_stage_upscale_model_name))[0] - upsampled_video_latent = LTXVLatentUpsampler().upsample_latent( - cropped_video_latent, - upscale_model, - vae, - )[0] - upsample_width, upsample_height = _pixel_dims_from_latent(upsampled_video_latent, vae) - resized_guidance_image = _resize_image_to(guidance_image, upsample_width, upsample_height) - preprocessed_guidance = LTXVPreprocess.execute(resized_guidance_image, int(image_compression))[0] + if str(second_stage_scale_mode) == "x2_latent_upscale_beta": + upscale_model = LatentUpscaleModelLoader.execute(str(second_stage_upscale_model_name))[0] + stage2_video_latent = LTXVLatentUpsampler().upsample_latent( + cropped_video_latent, + upscale_model, + vae, + )[0] + else: + stage2_video_latent = cropped_video_latent reinject_strength = float(second_stage_reinject_strength) if uses_source_anchor and segment_index > 0: reinject_strength = min(reinject_strength, float(continuity_settings["strength"])) - reinjected_video_latent = LTXVImgToVideoInplace.execute( - vae, - preprocessed_guidance, - upsampled_video_latent, - reinject_strength, - False, - )[0] + if reinject_strength > 0.0 and not (is_t2v and segment_index == 0): + stage2_width, stage2_height = _pixel_dims_from_latent(stage2_video_latent, vae) + resized_guidance_image = _resize_image_to(guidance_image, stage2_width, stage2_height) + preprocessed_guidance = LTXVPreprocess.execute(resized_guidance_image, int(image_compression))[0] + reinjected_video_latent = LTXVImgToVideoInplace.execute( + vae, + preprocessed_guidance, + stage2_video_latent, + reinject_strength, + False, + )[0] + else: + reinjected_video_latent = stage2_video_latent latent_stage2 = LTXVConcatAVLatent.execute(reinjected_video_latent, sampled_audio_latent)[0] model_stage2 = ModelSamplingLTXV.execute(stage2_model_active, float(max_shift), float(base_shift), latent_stage2)[0] guider_stage2 = CFGGuider.execute(model_stage2, stage2_positive, stage2_negative, float(second_stage_cfg))[0] @@ -1059,11 +1617,16 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: sampler_stage2, sigmas_stage2, latent_stage2, - )[1] + )[0] sampled_video = LTXVSeparateAVLatent.execute(sampled_stage2_av)[0] stage2_applied = True else: guidance_source = "none" + stage2_segment_report = ( + f"on({int(second_stage_step_count)}step,{second_stage_model_source},{second_stage_scale_mode})" + if stage2_applied + else "off" + ) anti_drift_report = "off" if int(segment_index) > 0 and anti_drift_mode != "off": @@ -1096,82 +1659,133 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: rolling_reference_latent = _clone_latent(sampled_video) _hard_unload_all_models() - segment_dir = os.path.join(segments_dir, f"seg_{segment_index:03d}") - decode_dir, frames_saved, _ = vae_decode_node.decode_to_disk( - sampled_video, - vae, - segment_dir, - "frame", - internal_decode_image_format, - int(internal_decode_jpg_quality), - bool(decode_settings["tile"]), - decode_settings["tiling_mode"], - int(decode_settings["tile_size"]), - int(decode_settings["overlap"]), - False, - os.path.join(run_dir, "seam_debug"), - bool(decode_settings["cleanup_between_frames"]), - True, - 0, - ) + if use_in_memory_loop: + decoded_images = _decode_images_in_memory(sampled_video, vae, 512, 64) + frames_saved = int(decoded_images.shape[0]) + overlay_report = "overlay=off(in-memory)" + if segment_index == 0: + ext_out = extension_node_mem.process_extension( + decoded_images, + overlap_frames_value, + overlap_side, + overlap_mode, + True, + "none", + "none", + start_frames_rule, + "none", + 0.25, + 8, + "best_of_k", + 16, + 1.0, + 0.5, + stitch_preset, + None, + 1, + ) + current_extended_images = decoded_images + current_start_images = ext_out[1] + else: + ext_out = extension_node_mem.process_extension( + current_extended_images, + overlap_frames_value, + overlap_side, + overlap_mode, + True, + "none", + "none", + start_frames_rule, + "none", + 0.25, + 8, + "best_of_k", + 16, + 1.0, + 0.5, + stitch_preset, + decoded_images, + 1, + ) + current_extended_images = ext_out[2] + current_start_images = ext_out[1] + else: + segment_dir = os.path.join(segments_dir, f"seg_{segment_index:03d}") + decode_dir, frames_saved, _ = vae_decode_node.decode_to_disk( + sampled_video, + vae, + segment_dir, + "frame", + internal_decode_image_format, + int(internal_decode_jpg_quality), + bool(decode_settings["tile"]), + decode_settings["tiling_mode"], + int(decode_settings["tile_size"]), + int(decode_settings["overlap"]), + False, + os.path.join(run_dir, "seam_debug"), + bool(decode_settings["cleanup_between_frames"]), + True, + 0, + ) - overlay_report = "overlay=off" - if str(segment_overlay_mode) != "off": - if str(segment_overlay_mode) == "custom_text": - overlay_text = _format_segment_overlay_text( - segment_overlay_text, - segment_index, - segment_count, - current_segment_raw_frames, - current_segment_unique_frames, - effective_unique_frames, - total_duration_seconds, + overlay_report = "overlay=off" + if str(segment_overlay_mode) != "off": + if str(segment_overlay_mode) == "custom_text": + overlay_text = _format_segment_overlay_text( + segment_overlay_text, + segment_index, + segment_count, + current_segment_raw_frames, + current_segment_unique_frames, + effective_unique_frames, + total_duration_seconds, + ) + else: + overlay_text = _format_segment_overlay_text( + "seg {segment_number}/{segment_count}\nraw {raw_frames}f uniq {unique_frames}f eff {effective_frames}f", + segment_index, + segment_count, + current_segment_raw_frames, + current_segment_unique_frames, + effective_unique_frames, + total_duration_seconds, + ) + overlay_frames = _overlay_text_on_frame_dir(decode_dir, overlay_text) + overlay_report = f"overlay={overlay_frames} frames" + + if segment_index == 0: + ext_out = extension_node.process_extension_disk( + decode_dir, + extended_dir, + start_dir, + overlap_frames_value, + overlap_side, + overlap_mode, + True, + "none", + "none", + start_frames_rule, + stitch_preset, + "", + 1, ) else: - overlay_text = _format_segment_overlay_text( - "seg {segment_number}/{segment_count}\nraw {raw_frames}f uniq {unique_frames}f eff {effective_frames}f", - segment_index, - segment_count, - current_segment_raw_frames, - current_segment_unique_frames, - effective_unique_frames, - total_duration_seconds, + ext_out = extension_node.process_extension_disk( + extended_dir, + extended_dir, + start_dir, + overlap_frames_value, + overlap_side, + overlap_mode, + True, + "none", + "none", + start_frames_rule, + stitch_preset, + decode_dir, + 1, ) - overlay_frames = _overlay_text_on_frame_dir(decode_dir, overlay_text) - overlay_report = f"overlay={overlay_frames} frames" - - if segment_index == 0: - ext_out = extension_node.process_extension_disk( - decode_dir, - extended_dir, - start_dir, - overlap_frames_value, - overlap_side, - overlap_mode, - True, - "none", - "none", - start_frames_rule, - stitch_preset, - "", - 1, - ) - else: - ext_out = extension_node.process_extension_disk( - extended_dir, - extended_dir, - start_dir, - overlap_frames_value, - overlap_side, - overlap_mode, - True, - "none", - "none", - start_frames_rule, - stitch_preset, - decode_dir, - 1, - ) gate_out = audio_gate_node.decide( remaining_frames_after, @@ -1184,12 +1798,16 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: cursor_frames = cursor_frames_out rendered_segments += 1 segment_reports.append( - f"seg{segment_index}: raw={current_segment_raw_frames}f unique={current_segment_unique_frames}f effective={effective_unique_frames}f saved={int(frames_saved)}f anchor={'src' if uses_source_anchor else 'tail'} stage2={'on' if stage2_applied else 'off'} guidance={guidance_source} anti_drift={anti_drift_report} {overlay_report} | {ext_out[5]}" + f"seg{segment_index}: raw={current_segment_raw_frames}f unique={current_segment_unique_frames}f effective={effective_unique_frames}f saved={int(frames_saved)}f init={init_mode} stage2={stage2_segment_report} guidance={guidance_source} anti_drift={anti_drift_report} {overlay_report} | {ext_out[5]}" ) _soft_cleanup() if int(gate_out[0]) == 0: break + final_frames_dir = extended_dir if not use_in_memory_loop else "" + final_start_dir = start_dir if not use_in_memory_loop else "" + final_report_hint = final_frames_dir if final_frames_dir else "(in-memory images)" + report = ( _render_status( total_duration_seconds, @@ -1210,8 +1828,10 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: + "\n" + planner_report_line + "\n" + + f"Audio preprocess. melband_enabled={melband_enabled} | {audio_preprocess_report}\n" + + f"Render route. backend_requested={backend_mode} | backend_resolved={backend_mode} | decode_mode={modular_decode} | generation_mode={generation_mode}\n" + f"Executable AU+IMG2VID render completed. segments_rendered={rendered_segments}/{segment_count} | " - f"frames_dir={extended_dir} | start_dir={start_dir}\n" + f"frames_dir={final_report_hint} | start_dir={final_start_dir or '(in-memory start_images)'}\n" + "\n".join(segment_reports) ) render_linx = build_stage_linx_payload( @@ -1220,6 +1840,8 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: "render", { "pipeline_kind": "au_img2vid_exec", + "backend_mode": str(backend_mode), + "generation_mode": str(generation_mode), "fps": fps_value, "modular_decode": str(modular_decode), "continuity_anchor_mode": str(continuity_settings["mode"]), @@ -1229,7 +1851,10 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: "anti_drift_strength": float(anti_drift_strength), "identity_persistence_strength": float(identity_persistence_strength), "second_stage_mode": str(second_stage_mode), + "second_stage_scale_mode": str(second_stage_scale_mode), "second_stage_upscale_model_name": str(second_stage_upscale_model_name), + "second_stage_steps": int(second_stage_step_count), + "second_stage_model_source": str(second_stage_model_source), "downstream_stage_mode": str(downstream_stage_mode), "segment_count": int(segment_count), "segments_rendered": int(rendered_segments), @@ -1243,36 +1868,43 @@ class IAMCCS_GC_AUIMG2VIDExecutableRender: }, downstream_stages=_downstream_stage_hints(downstream_stage_mode), policies={ - "decode_mode": _modular_decode_to_vae_mode(modular_decode), + "decode_mode": str(modular_decode), "stitch_preset": str(stitch_preset), "audio_context_mode": str(audio_context_mode), "continuity_anchor_mode": str(continuity_settings["mode"]), "anchor_refresh_interval": int(continuity_settings["interval"]), "anti_drift_mode": str(anti_drift_mode), "second_stage_mode": str(second_stage_mode), + "second_stage_scale_mode": str(second_stage_scale_mode), + "second_stage_steps": int(second_stage_step_count), + "second_stage_model_source": str(second_stage_model_source), }, outputs={ - "frames_dir": extended_dir, - "start_dir": start_dir, + "frames_dir": final_frames_dir, + "start_dir": final_start_dir, "segments_rendered": int(rendered_segments), "estimated_duration_seconds": float(total_duration_seconds), "render_status": _first_line(report), }, resources={ - "audio": audio, + "audio": raw_audio, "model": model, "clip": clip, "vae": vae, "audio_vae": audio_vae, "fps": float(fps_value), - "decode_mode": _modular_decode_to_vae_mode(modular_decode), + "decode_mode": str(modular_decode), "output_root": str(output_root), "planner_payload": str(plan_payload), + "generation_mode": str(generation_mode), "second_stage_model": stage2_model_active, + "second_stage_payload": second_stage_payload, "anti_drift_mode": str(anti_drift_mode), + "rendered_images": current_extended_images if use_in_memory_loop else None, + "start_images": current_start_images if use_in_memory_loop else None, }, ) - return (extended_dir, start_dir, int(rendered_segments), float(total_duration_seconds), render_linx, report) + return (final_frames_dir, final_start_dir, int(rendered_segments), float(total_duration_seconds), render_linx, report) class IAMCCS_GC_AUIMG2VIDExecutableFinalize: @@ -1336,17 +1968,16 @@ class IAMCCS_GC_AUIMG2VIDExecutableFinalize: class IAMCCS_GC_AUIMG2VIDExecutableVAE: CATEGORY = "IAMCCS/GoyAIcanvas/TestBackends" FUNCTION = "decode_and_combine" - RETURN_TYPES = ("STRING", SUPERNODE_LINX_TYPE, "STRING") - RETURN_NAMES = ("video_path", "linx", "report") + RETURN_TYPES = ("STRING", SUPERNODE_LINX_TYPE, "STRING", "IMAGE", "AUDIO", "STRING") + RETURN_NAMES = ("video_path", "linx", "report", "images", "audio_passthrough", "frames_dir_out") OUTPUT_NODE = True @classmethod def INPUT_TYPES(cls): return { "required": { - "ui_preset": (["low_ram_safe", "balanced", "high_quality", "fast_preview", "custom"], {"default": "balanced"}), "frame_rate": ("FLOAT", {"default": 24.0, "min": 1.0, "max": 240.0, "step": 0.01}), - "decode_mode": (["low_ram", "normal", "high", "custom_mode"], {"default": "low_ram"}), + "decode_mode": (_VAE_DECODE_MODES, {"default": "inherit_render_backend"}), "filename_prefix": ("STRING", {"default": "IAMCCS/GC_AUIMG2VID_EXEC"}), "output_root": ("STRING", {"default": "iamccs_gc_auimg2vid/final_vae"}), "frames_subdir": ("STRING", {"default": "frames"}), @@ -1374,7 +2005,6 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: def decode_and_combine( self, - ui_preset, frame_rate, decode_mode, filename_prefix, @@ -1396,19 +2026,37 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: prompt=None, extra_pnginfo=None, ): - del ui_preset audio = _require_runtime_value(_input_or_linx(audio, linx, "audio"), "audio") vae = _input_or_linx(vae, linx, "vae") + if video_latent is None: + video_latent = _input_or_linx(None, linx, "video_latent") + rendered_images = _input_or_linx(None, linx, "rendered_images") frame_rate = float(_inherit_widget_value(frame_rate, 24.0, linx, "fps")) - decode_mode = str(_inherit_widget_value(decode_mode, "low_ram", linx, "decode_mode")) + decode_mode = _normalize_modular_decode_mode( + _inherit_widget_value(decode_mode, "inherit_render_backend", linx, "decode_mode"), + linx_output(linx, "backend_mode", "auto"), + ) output_root = str(_inherit_widget_value(output_root, "iamccs_gc_auimg2vid/final_vae", linx, "output_root")) run_root = _resolve_output_path(output_root) target_frames_dir = os.path.join(run_root, str(frames_subdir or "frames")) actual_frames_dir = str(frames_dir or "").strip() decode_report = "" resolved_decode_mode = _modular_decode_to_vae_mode(decode_mode) + render_backend_mode = str(linx_output(linx, "backend_mode", "auto") or "auto") + images_out = None - if video_latent is not None and vae is not None: + if rendered_images is not None: + images_out = rendered_images + actual_frames_dir, frames_saved = _images_to_dir( + rendered_images, + target_frames_dir, + "frame", + image_format, + int(jpg_quality), + True, + ) + decode_report = f"decode_mode={decode_mode} used in-memory rendered images -> saved {frames_saved} frames to {actual_frames_dir}" + elif video_latent is not None and vae is not None: if str(resolved_decode_mode) == "low_ram_disk": vae_decode_node = _node_class("IAMCCS_VAEDecodeToDisk")() actual_frames_dir, _, _ = vae_decode_node.decode_to_disk( @@ -1429,14 +2077,16 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: 0, ) decode_report = f"decode_mode=low_ram -> {actual_frames_dir}" + images_out = _load_images_from_dir_for_output(actual_frames_dir) else: actual_decode_mode = str(resolved_decode_mode) if actual_decode_mode == "custom_mode": - actual_decode_mode = "high_vram" + actual_decode_mode = "normal_tiled" if actual_decode_mode == "normal_tiled": - decoded_images = comfy_nodes.VAEDecodeTiled().decode(vae, video_latent, int(tiled_tile_size), int(tiled_overlap))[0] + decoded_images = _decode_images_in_memory(video_latent, vae, int(tiled_tile_size), int(tiled_overlap)) else: decoded_images = comfy_nodes.VAEDecode().decode(vae, video_latent)[0] + images_out = decoded_images actual_frames_dir, frames_saved = _images_to_dir( decoded_images, target_frames_dir, @@ -1448,6 +2098,7 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: decode_report = f"decode_mode={decode_mode} -> saved {frames_saved} frames to {actual_frames_dir}" elif actual_frames_dir: decode_report = f"decode_mode={decode_mode} bypassed because frames_dir was provided directly: {actual_frames_dir}" + images_out = _load_images_from_dir_for_output(actual_frames_dir) else: raise ValueError("Executable VAE requires either video_latent+vae or frames_dir") @@ -1489,7 +2140,11 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: metadata_path = _save_video_metadata_sidecar(video_path, metadata_payload) if metadata_path: metadata_report = f"metadata={metadata_path}" - report = f"{decode_report} | {combine_report} | {metadata_report}" + report = ( + f"Executable VAE. backend_mode={render_backend_mode} | decode_requested={decode_mode} | " + f"decode_resolved={resolved_decode_mode} | frame_rate={float(frame_rate):.3f}\n" + f"{decode_report} | {combine_report} | {metadata_report}" + ) vae_linx = build_stage_linx_payload( linx, "exec_vae", @@ -1519,6 +2174,9 @@ class IAMCCS_GC_AUIMG2VIDExecutableVAE: "fps": float(frame_rate), "decode_mode": str(resolved_decode_mode), "output_root": str(output_root), + "rendered_images": images_out, }, ) - return (video_path, vae_linx, report) \ No newline at end of file + if images_out is None: + images_out = _load_images_from_dir_for_output(actual_frames_dir) + return (video_path, vae_linx, report, images_out, audio, str(actual_frames_dir)) diff --git a/web/iamccs_ltx2_time_length_sync.js b/web/iamccs_ltx2_time_length_sync.js index 10ea51d..463ecae 100644 --- a/web/iamccs_ltx2_time_length_sync.js +++ b/web/iamccs_ltx2_time_length_sync.js @@ -204,6 +204,48 @@ function ensurePlannerSettingsReportWidget(node) { return widget; } +function setWidgetVisibility(widget, visible) { + if (!widget || widget.type === "converted-widget") return; + + widget.hidden = !visible; + widget.disabled = !visible; + + if (widget.element) { + widget.element.style.display = visible ? "" : "none"; + } + if (widget.inputEl) { + widget.inputEl.style.display = visible ? "" : "none"; + } + + if (visible) { + if (Object.prototype.hasOwnProperty.call(widget, "__iamccsOrigComputeSize")) { + widget.computeSize = widget.__iamccsOrigComputeSize; + } else { + delete widget.computeSize; + } + } else { + if (!Object.prototype.hasOwnProperty.call(widget, "__iamccsOrigComputeSize")) { + widget.__iamccsOrigComputeSize = widget.computeSize; + } + widget.computeSize = () => [0, -4]; + widget.y = undefined; + widget.last_y = undefined; + } +} + +function applySegmentPlannerSettingsVisibility(node) { + const wPlanning = getWidget(node, "planning_mode"); + const wSeg = getWidget(node, "segment_duration_s"); + const wPreset = getWidget(node, "segment_preset") || getWidget(node, "content_profile"); + if (!wPlanning || !wSeg || !wPreset) return; + + const planningMode = String(wPlanning.value || "manual_segment_seconds"); + const explicitPresetMode = planningMode === "explicit_preset_seconds" || planningMode === "auto_profile"; + + setWidgetVisibility(wSeg, !explicitPresetMode); + setWidgetVisibility(wPreset, explicitPresetMode); +} + function ensurePlannerNodeSize(node) { try { const width = Math.max(460, Number(node.size?.[0] || 0)); @@ -235,6 +277,8 @@ function updateSegmentPlannerSettingsReport(node) { const wAutoSync = getWidget(node, "auto_sync_overlap"); if (!wSeg || !wPlanning || !wProfile || !wOverlap || !wAutoSync) return; + applySegmentPlannerSettingsVisibility(node); + const segmentDuration = clampNumber(wSeg.value, 0.01, 3600.0); const planningMode = String(wPlanning.value || "manual_segment_seconds"); const segmentPreset = String(wProfile.value || "10sec"); @@ -357,7 +401,8 @@ function updateSegmentPlannerPreview(node) { const currentUnique = Math.max(1, Math.min(uniqueFrames, totalFrames - currentStart)); const currentEnd = Math.min(totalFrames, currentStart + currentUnique); const currentRemaining = Math.max(0, totalFrames - currentEnd); - const currentRaw = clampedIndex === 0 ? firstRaw : nextRaw; + const lastRaw = segments <= 1 ? firstRaw : snapLengthToLtx2Rule(lastUnique + effectiveOverlapFrames, roundMode); + const currentRaw = clampedIndex === 0 ? firstRaw : (clampedIndex >= segments - 1 ? lastRaw : nextRaw); const currentStartS = currentStart / fps; const currentEndS = currentEnd / fps; @@ -394,6 +439,7 @@ function updateSegmentPlannerPreview(node) { `unique_segment_frames = ${uniqueFrames}`, `first_segment_raw_frames = ${firstRaw}`, `continuation_raw_frames = ${nextRaw}`, + `last_segment_raw_frames = ${lastRaw}`, `estimated_segments = ${segments}`, `continuation_loops = ${loops}`, `last_segment_unique_frames = ${lastUnique}`, @@ -470,6 +516,7 @@ function installSegmentPlannerSettingsSync(node) { ].filter(Boolean); if (!widgets.length || node._iamccsSegmentPlannerSettingsSyncInstalled) { + applySegmentPlannerSettingsVisibility(node); updateSegmentPlannerSettingsReport(node); return; } diff --git a/web/iamccs_supernodes_exec_ui.js b/web/iamccs_supernodes_exec_ui.js index d20f1a2..1ccd0de 100644 --- a/web/iamccs_supernodes_exec_ui.js +++ b/web/iamccs_supernodes_exec_ui.js @@ -5,10 +5,10 @@ const PRESET_CONFIGS = { presetWidget: "ui_preset", defaultPreset: "balanced", values: { - low_ram_safe: { modular_decode: "low_ram", steps: 16, image_compression: 40, continuity_anchor_mode: "off", anchor_refresh_interval: 2, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0 }, - balanced: { modular_decode: "normal", steps: 20, image_compression: 33, continuity_anchor_mode: "off", anchor_refresh_interval: 3, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0 }, - high_quality: { modular_decode: "high", steps: 24, image_compression: 28, continuity_anchor_mode: "off", anchor_refresh_interval: 1, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0 }, - fast_preview: { modular_decode: "low_ram", steps: 12, image_compression: 45, continuity_anchor_mode: "off", anchor_refresh_interval: 2, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0 }, + low_ram_safe: { modular_decode: "low_ram", steps: 16, image_compression: 40, continuity_anchor_mode: "off", anchor_refresh_interval: 2, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0, generation_mode: "img2vid", second_stage_mode: "off", stage2_model_policy: "stage2_model_if_connected", second_stage_reinject_strength: 0.0 }, + balanced: { modular_decode: "normal", steps: 20, image_compression: 33, continuity_anchor_mode: "off", anchor_refresh_interval: 3, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0, generation_mode: "img2vid", second_stage_mode: "off", stage2_model_policy: "stage2_model_if_connected", second_stage_reinject_strength: 0.0 }, + high_quality: { modular_decode: "high", steps: 24, image_compression: 28, continuity_anchor_mode: "off", anchor_refresh_interval: 1, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0, generation_mode: "img2vid", second_stage_mode: "off", stage2_model_policy: "stage2_model_if_connected", second_stage_reinject_strength: 0.0 }, + fast_preview: { modular_decode: "low_ram", steps: 12, image_compression: 45, continuity_anchor_mode: "off", anchor_refresh_interval: 2, anti_drift_mode: "off", anti_drift_strength: 0.0, identity_persistence_strength: 0.0, generation_mode: "img2vid", second_stage_mode: "off", stage2_model_policy: "stage2_model_if_connected", second_stage_reinject_strength: 0.0 }, }, visibility: { low_ram_safe: { anchor_image_strength: false, anti_drift_mode: false, anti_drift_strength: false, identity_persistence_strength: false, output_root: true }, @@ -50,6 +50,7 @@ const NODE_GROUPS = { { key: "anchor", label: "Anchor", color: "#b35c5c", widgets: ["continuity_anchor_mode", "anchor_refresh_interval", "anchor_image_strength", "anti_drift_mode", "anti_drift_strength", "identity_persistence_strength"] }, { key: "modular", label: "Modular", color: "#46769a", widgets: ["modular_decode", "downstream_stage_mode", "output_root"] }, { key: "debug", label: "Debug", color: "#8a8a36", widgets: ["segment_overlay_mode", "segment_overlay_text"] }, + { key: "stage2", label: "Mode + Second Stage", color: "#8a5ca0", widgets: ["generation_mode", "second_stage_mode", "stage2_model_policy", "second_stage_upscale_model", "second_stage_reinject_strength", "second_stage_cfg", "second_stage_manual_sigmas"] }, ], "IAMCCS-SuperNodes Second Stage": [ { key: "stage2", label: "Second Stage", color: "#8a5ca0", widgets: ["second_stage_mode", "stage2_model_policy", "second_stage_upscale_model", "second_stage_reinject_strength", "second_stage_cfg", "second_stage_manual_sigmas"] }, @@ -324,9 +325,29 @@ function applyRenderAnchorLabels(node) { if (node.comfyClass !== "IAMCCS-SuperNodes AU+IMG2VID Exec Render") { return; } + setWidgetLabel(node, "generation_mode", "Generation Mode"); setWidgetLabel(node, "continuity_anchor_mode", "Anchor Refresh"); setWidgetLabel(node, "anchor_refresh_interval", "Refresh Interval"); setWidgetLabel(node, "anchor_image_strength", "Anchor Guidance Strength"); + setWidgetLabel(node, "second_stage_mode", "Second Stage"); + setWidgetLabel(node, "stage2_model_policy", "Stage2 Model Policy"); + setWidgetLabel(node, "second_stage_upscale_model", "2x Upscale Model"); + setWidgetLabel(node, "second_stage_reinject_strength", "Anchor Reinject Strength"); +} + +function applyRenderSecondStageVisibility(node) { + if (node.comfyClass !== "IAMCCS-SuperNodes AU+IMG2VID Exec Render") { + return; + } + const mode = String(findWidget(node, "second_stage_mode")?.value || "off"); + const stageExpanded = !!node.properties?.iamccs_section_stage2; + const enabled = stageExpanded && mode !== "off"; + setWidgetVisibility(findWidget(node, "stage2_model_policy"), enabled); + setWidgetVisibility(findWidget(node, "second_stage_reinject_strength"), enabled); + setWidgetVisibility(findWidget(node, "second_stage_cfg"), enabled); + setWidgetVisibility(findWidget(node, "second_stage_manual_sigmas"), enabled); + setWidgetVisibility(findWidget(node, "second_stage_upscale_model"), enabled && mode === "latent_upscale_refine_x2_beta"); + fitNodeToWidgets(node); } function syncDownstreamVaeDecodeModes(renderNode) { @@ -373,6 +394,7 @@ function applyPresetConfig(node, nodeName) { } if (nodeName === "IAMCCS-SuperNodes AU+IMG2VID Exec Render") { applyRenderAnchorLabels(node); + applyRenderSecondStageVisibility(node); } fitNodeToWidgets(node); } @@ -444,6 +466,9 @@ function applyGroupVisibility(node, group, propKey, button) { for (const widgetName of group.widgets) { setWidgetVisibility(findWidget(node, widgetName), isExpanded); } + if (node.comfyClass === "IAMCCS-SuperNodes AU+IMG2VID Exec Render" && group.key === "stage2") { + applyRenderSecondStageVisibility(node); + } fitNodeToWidgets(node); app.graph.setDirtyCanvas(true, true); } @@ -482,6 +507,8 @@ function refreshNodeLayoutState(node, nodeName) { applyVaeDecodeModeVisibility(node); } else if (nodeName === "IAMCCS-SuperNodes AU+IMG2VID Exec Planner") { applyPlannerModeVisibility(node); + } else if (nodeName === "IAMCCS-SuperNodes AU+IMG2VID Exec Render") { + applyRenderSecondStageVisibility(node); } else { fitNodeToWidgets(node); } @@ -685,6 +712,19 @@ app.registerExtension({ }; syncDownstreamVaeDecodeModes(this); } + for (const widgetName of ["generation_mode", "second_stage_mode"]) { + const widget = findWidget(this, widgetName); + if (!widget) { + continue; + } + const originalCallback = widget.callback; + widget.callback = (...args) => { + originalCallback?.apply(widget, args); + applyRenderSecondStageVisibility(this); + app.graph.setDirtyCanvas(true, true); + }; + } + applyRenderSecondStageVisibility(this); } if (nodeName === "IAMCCS-SuperNodes AU+IMG2VID Exec Planner") { installExecPlannerExplicitPresetSync(this);