diff --git a/src/pixelrush.py b/src/pixelrush.py index 4784605..f27ed27 100644 --- a/src/pixelrush.py +++ b/src/pixelrush.py @@ -51,15 +51,12 @@ class PixelRushConfig: Paper default: 24.0. Rule of thumb: sigma ~ patch_size / 5. eps : float Numerical stability epsilon. - operate_in_vae_space : bool - When True (default), the algorithm runs in VAE latent space (std ≈ 1) - and the injected adapters convert to model space internally. This is - required for models whose ``process_latent_in`` scales the latent - (e.g. SDXL ``scale_factor=0.13025``): without it, the fixed-magnitude - noise injection (std ≈ 0.95) would dominate the scaled-down signal - (std ≈ 0.13) and produce a noisy output. When False, the legacy path - is used (``execute`` applies ``process_latent_in`` and the adapters - operate in model space). + + Notes + ----- + The core algorithm runs in VAE latent space (the ComfyUI LATENT + convention); adapters injected by the node own the VAE<->model + conversions internally (plan 2026-09-02). """ patch_h: int @@ -70,7 +67,6 @@ class PixelRushConfig: noise_injection: str = "slerp" gaussian_sigma: float = 24.0 eps: float = 1e-8 - operate_in_vae_space: bool = True # --------------------------------------------------------------------------- diff --git a/src/pixelrush_node.py b/src/pixelrush_node.py index 2e11dcb..28d3c14 100644 --- a/src/pixelrush_node.py +++ b/src/pixelrush_node.py @@ -124,8 +124,7 @@ def _make_eps_to_x0(model_sampling, prediction_type): return eps_to_x0 -def _make_predict_eps(model, positive, negative, cfg_scale, latent_dimensions=2, - operate_in_vae_space=True): +def _make_predict_eps(model, positive, negative, cfg_scale, latent_dimensions=2): """Create a predict_eps adapter that runs the model with CFG. Uses ComfyUI's full conditioning pipeline: @@ -138,19 +137,19 @@ def _make_predict_eps(model, positive, negative, cfg_scale, latent_dimensions=2, type (EPS, CONST/flow, V_PREDICTION, X0). This is critical: FLUX uses CONST (flow matching) where the raw output is velocity, not epsilon. + Space contract (plan 2026-09-02): the adapter accepts a VAE-space latent + (the ComfyUI LATENT convention), converts it to model space via + ``process_latent_in`` for the model call, and returns the epsilon in + MODEL space. The model's true epsilon has std ~ 1 in model space, which + is exactly the space the corrected-theory slerp assumes when mixing + eps_refined with a std-1 random vector. The x-side VAE<->model + conversions are owned by the forward_step/reverse_step adapters. + For 3D latent models (latent_dimensions=3), the core PixelRush algorithm works in 4D spatial [B, C, H, W], but the model expects 5D [B, C, T, H, W]. This adapter unsqueezes 4D patches to 5D before calling diffusion_model, and squeezes the 5D eps output back to 4D. - When ``operate_in_vae_space`` is True, the adapter accepts a VAE-space latent - (std ≈ 1) and returns a VAE-space epsilon. It converts to model space via - ``process_latent_in`` before running the model and converts the epsilon back - via ``process_latent_out``. This keeps the core algorithm in a space where - the fixed-magnitude noise injection (std ≈ 0.95) is comparable to the signal - (std ≈ 1), which is required for models whose ``process_latent_in`` scales - the latent (e.g. SDXL ``scale_factor=0.13025``). - Returns a callable: predict_eps(latent, timestep) -> eps [B, C, H, W] """ import comfy.model_management @@ -161,16 +160,10 @@ def _make_predict_eps(model, positive, negative, cfg_scale, latent_dimensions=2, device = model.load_device if hasattr(model, 'load_device') else torch.device("cpu") is_3d = latent_dimensions == 3 - # Capture latent-space converters. When operating in VAE space, the adapter - # converts VAE latents -> model space before the model call and converts the - # resulting epsilon back to VAE space. When None (no process_latent_in/out - # on the model, or legacy mode), the conversions are no-ops. + # VAE-space latent -> model space for the model call. None (no + # process_latent_in on the model) means the formats coincide and the + # conversion is a no-op. process_latent_in = getattr(model.model, 'process_latent_in', None) - if not operate_in_vae_space: - # Legacy mode: execute() already applied process_latent_in; eps stays in - # model space. Disable the conversion here (process_latent_out is only - # needed in VAE-space mode, so it is not captured at all). - process_latent_in = None # Ensure the model is loaded to GPU and pre_run is called # pre_run sets model.model.current_patcher = model (the ModelPatcher) @@ -314,59 +307,72 @@ def _make_predict_eps(model, positive, negative, cfg_scale, latent_dimensions=2, if is_3d and eps.ndim == 5 and was_4d: eps = eps.squeeze(2) # [B, C, H, W] - # When operating in VAE space, the input latent was converted to model - # space via process_latent_in before the model call, but the returned - # epsilon is intentionally NOT converted back. The model predicts noise - # with std ≈ 1 in model space; the VAE-space latent also has std ≈ 1. - # Keeping the epsilon at std ≈ 1 (numerically comparable to the latent) - # is what makes the fixed-magnitude noise injection (std ≈ 0.95) balanced - # against the signal, instead of being scaled down by scale_factor and - # drowned out. forward_step/reverse_step then operate directly in VAE - # space (no latent conversion), so x (std ≈ 1) and eps (std ≈ 1) stay - # comparable throughout. + # The returned epsilon stays in MODEL space by design: the model + # predicts noise with std ~ 1 there, which is the space the + # corrected-theory slerp (mixing eps_refined with std-1 random + # noise) and the forward/reverse adapters assume. The x-side + # VAE<->model conversions are owned by the adapters. return eps return predict_eps -def _make_forward_step(model, operate_in_vae_space=True): +def _make_forward_step(model, process_latent_in=None, process_latent_out=None): """Create a forward_step adapter using the model's noise_scaling. forward_step(x_0, eps, sigma) -> x_K (noises x_0 to timestep K) Uses the model's own noise schedule, which is correct for all prediction types (EPS, CONST/flow, V_PREDICTION, etc.). - When ``operate_in_vae_space`` is True (default), ``x_0`` and ``eps`` both - arrive in VAE space with std ≈ 1 (the epsilon is NOT scaled by - process_latent_out — see _make_predict_eps). The DDIM forward is applied - directly in VAE space, so the fixed-magnitude noise injection stays - balanced against the signal. No latent<->model conversion happens here. + Space contract (plan 2026-09-02): ``x_0`` arrives in VAE space (the + ComfyUI LATENT convention) and ``eps`` in model space (as returned by + predict_eps). The adapter owns the x-side conversion: it converts x to + model space, applies noise_scaling, and converts the result back to + VAE space. For pure-scaling formats (e.g. SDXL scale_factor 0.13025) + this is exactly equivalent to running the whole algorithm in model + space: process_latent_in(forward(x, eps, s)) == s*x + s*eps_noise, + so the model receives noise at the scale the timestep claims. Before + this fix the noise component arrived scaled by the format factor + (7.7x too small for SDXL) and the input SNR did not match sigma. """ ms = model.model.model_sampling + if process_latent_in is None: + process_latent_in = lambda t: t + if process_latent_out is None: + process_latent_out = lambda t: t def forward_step(x_0, eps, sigma): - return ms.noise_scaling(sigma, eps, x_0) + x_model = process_latent_in(x_0) + x_k_model = ms.noise_scaling(sigma, eps, x_model) + return process_latent_out(x_k_model) return forward_step -def _make_reverse_step(model, operate_in_vae_space=True): +def _make_reverse_step(model, process_latent_in=None, process_latent_out=None): """Create a reverse_step adapter using eps_to_x0. reverse_step(x_K, eps_injected, sigma) -> x_0_hat Converts injected epsilon back to x0 using the inverse of noise_scaling, which is correct for all prediction types. - When ``operate_in_vae_space`` is True (default), ``x_K`` and ``eps_injected`` - both arrive in VAE space with std ≈ 1, so the DDIM reverse is applied - directly in VAE space. No latent<->model conversion happens here. + Space contract (plan 2026-09-02): mirrors _make_forward_step — x in VAE + space, eps in model space; the adapter converts x to model space, + recovers x0, and converts the result back to VAE space. The two + conversions cancel exactly on a forward/reverse round trip. """ ms = model.model.model_sampling prediction_type = _detect_prediction_type(ms) eps_to_x0 = _make_eps_to_x0(ms, prediction_type) + if process_latent_in is None: + process_latent_in = lambda t: t + if process_latent_out is None: + process_latent_out = lambda t: t def reverse_step(x_K, eps_injected, sigma): - return eps_to_x0(x_K, eps_injected, sigma) + x_model = process_latent_in(x_K) + x0_model = eps_to_x0(x_model, eps_injected, sigma) + return process_latent_out(x0_model) return reverse_step @@ -405,7 +411,7 @@ def _make_alpha_bar_at(model): return alpha_bar_at -def _make_vae_adapters(vae, device, model=None, operate_in_vae_space=True): +def _make_vae_adapters(vae, device, model=None): """Create VAE decode/encode adapters. Returns (vae_decode, vae_encode) callables. @@ -413,53 +419,24 @@ def _make_vae_adapters(vae, device, model=None, operate_in_vae_space=True): Handles both 2D VAEs (latent_dim=2, 4D latents [B,C,H,W]) and 3D/video VAEs (latent_dim=3, 5D latents [B,C,T,H,W]). - Uses model.process_latent_out/in to convert between model latent - format and VAE latent format — UNLESS ``operate_in_vae_space`` is True, - in which case the latent already lives in VAE space and the adapters must - NOT apply process_latent_out/in (that would re-scale an already-VAE-space - latent and corrupt it). The adapters then only handle shape (3D unsqueeze) - and the raw vae.decode/encode calls. + Space contract (plan 2026-09-02): PixelRush runs the core algorithm in + VAE latent space (the ComfyUI LATENT convention), so these adapters + must NOT apply ``process_latent_out``/``process_latent_in`` — that + would re-scale an already-VAE-space latent and corrupt it. The + VAE<->model conversions are owned by predict_eps / forward_step / + reverse_step. The adapters only handle shape (3D unsqueeze) and the + raw vae.decode/encode calls. - For 3D latent models (Wan21, Krea2, Qwen, Anima), the model's - ``process_latent_out``/``process_latent_in`` use 5D ``latents_mean``/ - ``latents_std`` with shape ``[1, C, 1, 1, 1]``. Calling these on a 4D - tensor causes a broadcasting misalignment that corrupts the batch - (see plan 2026-08-10-freescale-krea2-5d-latent-fix.md). - - Therefore, for 3D latent models: - - ``vae_decode`` accepts 5D latents and calls ``process_latent_out`` - directly on the 5D tensor. - - ``vae_encode`` returns 5D latents (with singleton temporal dim) so - they can be passed directly to the sampler. + For 3D latent models (Wan21, Krea2, Qwen, Anima), a 4D latent is + unsqueezed to 5D only where the raw VAE needs it (decode input, + encode output), never for latent-space scaling. """ latent_dim = getattr(vae, 'latent_dim', 2) - process_latent_out = None - process_latent_in = None - if model is not None and hasattr(model, 'model'): - if hasattr(model.model, 'process_latent_out'): - process_latent_out = model.model.process_latent_out - if hasattr(model.model, 'process_latent_in'): - process_latent_in = model.model.process_latent_in - # When operating in VAE space, the latent is already in VAE format; the - # adapters must not re-apply the model<->VAE scaling. - if operate_in_vae_space: - process_latent_out = None - process_latent_in = None def vae_decode(latent: torch.Tensor) -> torch.Tensor: if isinstance(latent, dict): latent = latent["samples"] latent = latent.to(device) - # Convert from model latent format to VAE latent format. - # For 3D latent models, process_latent_out expects 5D input. - if process_latent_out is not None: - if latent_dim == 3: - # Ensure 5D for process_latent_out - if latent.ndim == 4: - latent = latent.unsqueeze(2) # [B, C, 1, H, W] - latent = process_latent_out(latent) - else: - latent = process_latent_out(latent) # For 3D VAEs, ensure temporal dimension is present for vae.decode if latent_dim == 3 and latent.ndim == 4: latent = latent.unsqueeze(2) @@ -484,20 +461,9 @@ def _make_vae_adapters(vae, device, model=None, operate_in_vae_space=True): # For 3D VAEs, take first temporal frame to get 4D if latent_dim == 3 and encoded.ndim == 5: encoded = encoded[:, :, 0] - # Convert from VAE latent format to model latent format (legacy path). - # In VAE-space mode (operate_in_vae_space=True) process_latent_in is - # None, so no scaling is applied and the latent stays in VAE space. - if process_latent_in is not None: - if latent_dim == 3: - if encoded.ndim == 4: - encoded = encoded.unsqueeze(2) # [B, C, 1, H, W] - encoded = process_latent_in(encoded) - else: - encoded = process_latent_in(encoded) # For 3D latent models, always return 5D [B, C, 1, H, W] (singleton # temporal dim) so the latent can be passed directly to the sampler / - # core algorithm (which squeezes to 4D). This is independent of whether - # process_latent_in scaling was applied. + # core algorithm (which squeezes to 4D). if latent_dim == 3 and encoded.ndim == 4: encoded = encoded.unsqueeze(2) # [B, C, 1, H, W] return encoded @@ -505,42 +471,17 @@ def _make_vae_adapters(vae, device, model=None, operate_in_vae_space=True): return vae_decode, vae_encode -def _prepare_initial_latent(initial_latent, process_latent_in, latent_dimensions, - operate_in_vae_space): - """Convert the initial latent to model format ONLY when not in VAE space. +def _prepare_initial_latent(initial_latent, latent_dimensions): + """Validate/normalize the initial latent shape (no space conversion). - When ``operate_in_vae_space`` is True (default), the latent stays in VAE - space (std ≈ 1) and the adapters convert to model space internally. This is - required for models whose ``process_latent_in`` scales the latent down (e.g. - SDXL ``scale_factor=0.13025``), otherwise the fixed-magnitude noise - injection would dominate the signal. - - When ``operate_in_vae_space`` is False (legacy path), ``process_latent_in`` - is applied here. For 3D latent models, ``process_latent_in`` expects 5D input - ``[B, C, T, H, W]`` — a 4D latent is unsqueezed first to avoid broadcasting - misalignment with 5D ``latents_mean``/``latents_std``. - - Parameters - ---------- - initial_latent : Tensor - ``[B, C, H, W]`` (or ``[B, C, T, H, W]`` for 3D) latent at native res. - process_latent_in : callable or None - Model's ``process_latent_in`` (scales latent to model space), or None. - latent_dimensions : int - 2 for 2D VAEs, 3 for 3D/video VAEs. - operate_in_vae_space : bool - If True, skip ``process_latent_in`` (latent already in VAE space). - - Returns - ------- - Tensor - The (possibly converted) initial latent. + The core algorithm runs in VAE latent space (the ComfyUI LATENT + convention); ``process_latent_in`` must NOT be applied here. The + VAE<->model conversions happen inside the adapters. For 3D latent + models a 5D latent is kept as-is; the temporal squeeze for the core + algorithm happens in execute(). """ - if process_latent_in is not None and not operate_in_vae_space: - if latent_dimensions == 3: - if initial_latent.ndim == 4: - initial_latent = initial_latent.unsqueeze(2) # [B, C, 1, H, W] - initial_latent = process_latent_in(initial_latent) + if latent_dimensions == 3 and initial_latent.ndim == 4: + initial_latent = initial_latent.unsqueeze(2) # [B, C, 1, H, W] return initial_latent @@ -654,25 +595,14 @@ class PixelRushNode(io.ComfyNode): # Move initial latent to model device for GPU acceleration initial_latent = initial_latent.to(device) - # PixelRush runs in VAE latent space by default (std ≈ 1). The adapters - # convert to model space internally. Set to False only to use the legacy - # model-space path. - operate_in_vae_space = True - - # Convert initial latent to model format (process_latent_in) ONLY when - # NOT operating in VAE space. When operate_in_vae_space is True, the - # latent stays in VAE space (std ≈ 1) and the adapters convert to model - # space internally. This is required for models whose process_latent_in - # scales the latent down (e.g. SDXL scale_factor=0.13025), otherwise the - # fixed-magnitude noise injection would dominate the signal. - # In normal ComfyUI sampling, the guider calls process_latent_in before - # apply_model. Since PixelRush calls apply_model directly via predict_eps, - # we must convert here (legacy path). For 3D latent models, - # process_latent_in expects 5D input [B, C, T, H, W] — passing 4D causes - # broadcasting misalignment with 5D latents_mean/std - # (see plan 2026-08-10-freescale-krea2-5d-latent-fix.md). + # PixelRush runs the core algorithm in VAE latent space (the ComfyUI + # LATENT convention). The predict_eps / forward_step / reverse_step + # adapters own the VAE<->model conversions internally; the initial + # latent is never pre-converted here. For 3D latent models a 4D latent + # is unsqueezed to 5D (the temporal squeeze for the core algorithm + # happens below). initial_latent = _prepare_initial_latent( - initial_latent, process_latent_in, latent_dimensions, operate_in_vae_space + initial_latent, latent_dimensions ) # Auto-detect patch size from native resolution @@ -699,21 +629,26 @@ class PixelRushNode(io.ComfyNode): noise_lambda=noise_lambda, noise_injection=noise_injection, gaussian_sigma=gaussian_sigma, - operate_in_vae_space=operate_in_vae_space, ) # Create adapters — predict_eps needs to know if model is 3D latent predict_eps = _make_predict_eps( model, positive, negative, cfg, latent_dimensions, - operate_in_vae_space=cfg_obj.operate_in_vae_space, ) alpha_bar_at = _make_alpha_bar_at(model) - vae_decode, vae_encode = _make_vae_adapters( - vae, device, model, operate_in_vae_space=cfg_obj.operate_in_vae_space, + vae_decode, vae_encode = _make_vae_adapters(vae, device, model) + # Model-agnostic forward/reverse steps (handle CONST/flow, V_PRED, EPS). + # They own the x-side VAE<->model conversion (process_latent_in/out). + forward_step = _make_forward_step( + model, + process_latent_in=process_latent_in, + process_latent_out=getattr(model.model, 'process_latent_out', None), + ) + reverse_step = _make_reverse_step( + model, + process_latent_in=process_latent_in, + process_latent_out=getattr(model.model, 'process_latent_out', None), ) - # Model-agnostic forward/reverse steps (handle CONST/flow, V_PRED, EPS) - forward_step = _make_forward_step(model, operate_in_vae_space=cfg_obj.operate_in_vae_space) - reverse_step = _make_reverse_step(model, operate_in_vae_space=cfg_obj.operate_in_vae_space) sigma_at = _make_sigma_at(model) # For 3D latent models, squeeze temporal dim for the core algorithm diff --git a/tests/test_pixelrush.py b/tests/test_pixelrush.py index 02da6eb..ec59fad 100644 --- a/tests/test_pixelrush.py +++ b/tests/test_pixelrush.py @@ -728,23 +728,6 @@ class TestPixelRushCascade: assert result.shape[3] == expected -# --------------------------------------------------------------------------- -# PixelRushConfig: operate_in_vae_space flag (plan 2026-08-12) -# --------------------------------------------------------------------------- - -@pytest.mark.unit -class TestPixelRushConfigVAESpace: - def test_default_operate_in_vae_space_true(self): - """Default must be True (algorithm runs in VAE space).""" - cfg = PixelRushConfig(patch_h=32, patch_w=32) - assert cfg.operate_in_vae_space is True - - def test_override_operate_in_vae_space_false(self): - """Can be set to False (legacy model-space path).""" - cfg = PixelRushConfig(patch_h=32, patch_w=32, operate_in_vae_space=False) - assert cfg.operate_in_vae_space is False - - # --------------------------------------------------------------------------- # Regression: SDXL noise-dominance fix (plan 2026-08-12) # --------------------------------------------------------------------------- @@ -818,25 +801,51 @@ class TestPixelRushCascadeVAESpace: def test_vae_space_cascade_signal_dominated(self): """Full cascade on a realistic SDXL mock must be signal-dominated. - Regression guard for the 'totally noisy' bug: out.std / z0.std < 2.0. - (Before the VAE-space fix this ratio was > 6.) + Regression guard for the 'totally noisy' bug: 0.5 < out.std/z0.std < 2.0. + (Before the VAE-space fix this ratio was > 6; the lower bound guards + against an inert/no-op refinement.) """ cfg = PixelRushConfig( patch_h=32, patch_w=32, overlap=0.5, k_timestep=249, - noise_lambda=0.95, operate_in_vae_space=True, + noise_lambda=0.95, ) z0, result = self._run_cascade(cfg) ratio = result.std() / z0.std() assert ratio < 2.0, ( f"Output noise dominates signal (ratio={ratio:.2f}); expected < 2.0" ) + assert ratio > 0.5, ( + f"Output is inert vs input (ratio={ratio:.2f}); refinement is a no-op?" + ) + + def test_refinement_changes_latent(self): + """Refinement must actually change the latent (> 1% relative delta). + + Guards against the refinement collapsing to a no-op under any + future refactor (space-conversion or injection changes). + """ + cfg = PixelRushConfig( + patch_h=32, patch_w=32, overlap=0.5, k_timestep=249, + noise_lambda=0.95, + ) + z0, result = self._run_cascade(cfg) + # The cascade's first-stage input is the bicubic-upscaled z0; the + # refined result must differ from a pure passthrough meaningfully. + import torch.nn.functional as F + z0_up = F.interpolate(z0, size=result.shape[2:], mode="bicubic", + align_corners=False, antialias=True) + rel = (result - z0_up).norm() / z0_up.norm() + assert rel > 0.01, ( + f"Refinement barely changed the latent (rel={rel:.4f}); " + "suspect a no-op pipeline" + ) def test_vae_space_cascade_correlates_with_input(self): """Output should positively correlate with the input (structure kept).""" import torch.nn.functional as F cfg = PixelRushConfig( patch_h=32, patch_w=32, overlap=0.5, k_timestep=249, - noise_lambda=0.95, operate_in_vae_space=True, + noise_lambda=0.95, ) z0, result = self._run_cascade(cfg) res_down = F.interpolate( @@ -930,8 +939,7 @@ class TestPixelRushCompressionDiagnostics: def _make_cfg(self, **overrides): base = dict(patch_h=32, patch_w=32, overlap=0.5, k_timestep=249, - noise_lambda=0.95, noise_injection="additive", - operate_in_vae_space=True) + noise_lambda=0.95, noise_injection="additive") base.update(overrides) return PixelRushConfig(**base) diff --git a/tests/test_pixelrush_node.py b/tests/test_pixelrush_node.py index 096b670..2592d08 100644 --- a/tests/test_pixelrush_node.py +++ b/tests/test_pixelrush_node.py @@ -1213,14 +1213,190 @@ class TestPixelRushKTimestepScaling: @pytest.mark.unit -class TestPrepareInitialLatent: - """Tests for _prepare_initial_latent (regression guard for the SDXL - UnboundLocalError: cfg_obj referenced before assignment in execute). +class TestPipelineSpaceConvention: + """Plan 2026-09-02 Step 7: the operate_in_vae_space flag is removed. - The guard that decides whether to apply process_latent_in to the initial - latent was previously inlined in execute and referenced cfg_obj (defined - later). Extracting it into this helper makes operate_in_vae_space an - explicit parameter, so it can never be undefined. + The pipeline is ALWAYS VAE-space at the interfaces (ComfyUI LATENT + convention): execute never pre-converts the initial latent, the VAE + adapters never apply process_latent_out/in, and predict_eps / + forward_step / reverse_step own the VAE<->model conversions. + """ + + def _read_source(self): + return (pathlib.Path(__file__).parent.parent / "src" / "pixelrush_node.py").read_text(encoding="utf-8") + + @staticmethod + def _strip_docstrings_and_comments(content): + import ast + tree = ast.parse(content) + lines = content.splitlines() + for node in ast.walk(tree): + if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef, ast.Module)): + doc = ast.get_docstring(node, clean=False) + if doc: + # blank out the docstring lines only + for i in range(node.body[0].lineno - 1, + node.body[0].lineno - 1 + doc.count("\n") + 1): + lines[i] = "" + code = "\n".join(lines) + code = "\n".join(ln.split("#")[0] for ln in code.splitlines()) + return code + + def test_no_operate_in_vae_space_flag(self): + """The flag must be gone from code (docstrings may mention removal).""" + for rel in ("src/pixelrush_node.py", "src/pixelrush.py"): + content = (pathlib.Path(__file__).parent.parent / rel).read_text(encoding="utf-8") + code = self._strip_docstrings_and_comments(content) + assert "operate_in_vae_space" not in code, ( + f"{rel} must not reference the removed operate_in_vae_space flag in code" + ) + + def test_execute_never_preconverts_initial_latent(self): + """execute must pass the initial latent through _prepare_initial_latent + without process_latent_in (core runs in VAE space).""" + content = self._read_source() + assert "initial_latent = _prepare_initial_latent(" in content + call = content[content.index("initial_latent = _prepare_initial_latent("):] + call = call[:call.index(")")] + assert "process_latent_in" not in call, ( + "_prepare_initial_latent must not receive process_latent_in" + ) + + def test_vae_adapters_do_not_convert_space(self): + """_make_vae_adapters must not apply process_latent_out/in (VAE space + in, VAE space out); conversions live in predict_eps/forward/reverse.""" + content = self._read_source() + start = content.index("def _make_vae_adapters") + end = content.index("def _prepare_initial_latent") + section = content[start:end] + code = self._strip_docstrings_and_comments(section) + assert "process_latent_out(" not in code, ( + "vae_decode must not call process_latent_out (latent is already in VAE space)" + ) + assert "process_latent_in(" not in code, ( + "vae_encode must not call process_latent_in (latent stays in VAE space)" + ) + + def test_predict_eps_converts_input_via_process_latent_in(self): + """predict_eps must convert the VAE-space latent to model space for + the model call (and return model-space eps).""" + content = self._read_source() + start = content.index("def _make_predict_eps") + end = content.index("def _make_forward_step") + section = content[start:end] + assert "process_latent_in(latent)" in section, ( + "predict_eps must apply process_latent_in to the input latent" + ) + + def test_forward_reverse_adapters_own_conversion(self): + """forward/reverse adapters must convert via process_latent_in/out.""" + content = self._read_source() + start = content.index("def _make_forward_step") + end = content.index("def _make_sigma_at") + section = content[start:end] + assert section.count("process_latent_in(x_0)") >= 1 + assert section.count("process_latent_out(x_k_model)") >= 1 + assert section.count("process_latent_in(x_K)") >= 1 + assert section.count("process_latent_out(x0_model)") >= 1 + + +@pytest.mark.unit +class TestAdapterSpaceConversion: + """The Step 7 exactness tests: forward/reverse must convert spaces such + that the MODEL-SPACE view of the noised latent carries eps at full + model-space scale (SNR matches the timestep sigma). + + Before the fix, VAE-space x was noised with model-space eps directly: + for SDXL (scale_factor 0.13025) the model then saw + s*x + s*sigma*eps — noise 7.7x too small for the claimed timestep. + """ + + def _make_adapters(self, scale=0.13025): + """EPS-model mock with a pure-scaling latent format (SDXL-like).""" + import sys + import types + sys.path.insert(0, str(pathlib.Path(__file__).parent.parent)) + + def noise_scaling(sigma, noise, latent_image): + sigma_r = sigma.reshape(sigma.shape + (1,) * (latent_image.ndim - sigma.ndim)) + return sigma_r * noise + latent_image + + EpsClass = type("EPS", (), {}) + ModelSampling = type("ModelSampling", (EpsClass,), {}) + ms_instance = ModelSampling() + ms_instance.noise_scaling = noise_scaling + ms_instance.timestep = lambda sigma: sigma * 999.0 + + model = types.SimpleNamespace() + model.model = types.SimpleNamespace(model_sampling=ms_instance) + + def process_latent_in(t): + return t * scale + + def process_latent_out(t): + return t / scale + + from src.pixelrush_node import _make_forward_step, _make_reverse_step + forward = _make_forward_step(model, process_latent_in, process_latent_out) + reverse = _make_reverse_step(model, process_latent_in, process_latent_out) + return forward, reverse, process_latent_in, scale + + def test_forward_step_produces_model_space_snr(self): + """process_latent_in(forward(x, eps, sigma)) == s*x + sigma*eps. + + This is the core exactness property: the model-space view of the + noised latent must carry the eps at full model-space scale. + """ + forward, _, process_latent_in, s = self._make_adapters() + x = torch.randn(1, 4, 8, 8) # VAE space + eps = torch.randn(1, 4, 8, 8) # model space + sigma = torch.tensor([0.5]) + x_k_vae = forward(x, eps, sigma) + model_view = process_latent_in(x_k_vae) + expected = s * x + 0.5 * eps # s*x + sigma*eps + assert torch.allclose(model_view, expected, atol=1e-5), ( + "forward_step must produce s*x + sigma*eps in model space " + f"(got max err {(model_view - expected).abs().max():.3e}; the " + "pre-fix bug gave s*x + s*sigma*eps — noise 7.7x too small)" + ) + + def test_reverse_step_round_trip_identity(self): + """reverse(forward(x, e, sigma), e, sigma) must return x exactly.""" + forward, reverse, _, _ = self._make_adapters() + x = torch.randn(2, 4, 8, 8) + eps = torch.randn(2, 4, 8, 8) + for sigma_val in (0.1, 0.6, 0.9): + sigma = torch.tensor([sigma_val]) + x_k = forward(x, eps, sigma) + x_rec = reverse(x_k, eps, sigma) + assert torch.allclose(x_rec, x, atol=1e-4), ( + f"round trip failed at sigma={sigma_val}" + ) + + def test_forward_reverse_preserve_vae_space_magnitude(self): + """With realistic SDXL magnitudes (VAE std ~7.7, eps std ~1), the + model-space noise/signal ratio of forward output must equal sigma.""" + forward, _, process_latent_in, s = self._make_adapters() + x = 7.7 * torch.randn(1, 4, 16, 16) # VAE space + eps = 1.0 * torch.randn(1, 4, 16, 16) # model space + sigma = torch.tensor([0.25]) + x_k_vae = forward(x, eps, sigma) + model_view = process_latent_in(x_k_vae) + noise_part = model_view - s * x + assert torch.allclose(noise_part, 0.25 * eps, atol=1e-4), ( + "noise component in model space must be exactly sigma*eps" + ) + ratio = noise_part.std() / (s * x).std() + assert abs(ratio.item() - 0.25) < 0.05, ( + f"noise/signal ratio must match sigma (0.25), got {ratio.item():.3f}" + ) + + +@pytest.mark.unit +class TestPrepareInitialLatent: + """Tests for _prepare_initial_latent under the always-VAE convention + (plan 2026-09-02 Step 7): no space conversion, 3D shape normalization + only. """ def _import_helper(self): @@ -1229,65 +1405,24 @@ class TestPrepareInitialLatent: from src.pixelrush_node import _prepare_initial_latent return _prepare_initial_latent - def test_vae_space_skips_process_latent_in(self): - """operate_in_vae_space=True must NOT call process_latent_in (SDXL fix).""" + def test_never_applies_process_latent_in(self): + """The helper must never scale the latent (space conversions live in + the adapters).""" helper = self._import_helper() latent = torch.randn(1, 4, 32, 32) - calls = [] - def process_latent_in(x): - calls.append(1) - return x * 0.13025 - out = helper(latent, process_latent_in, latent_dimensions=2, - operate_in_vae_space=True) - assert len(calls) == 0, "process_latent_in must be skipped in VAE space" - assert torch.equal(out, latent), "latent must be unchanged in VAE space" + out = helper(latent, latent_dimensions=2) + assert torch.equal(out, latent) - def test_model_space_applies_process_latent_in(self): - """operate_in_vae_space=False must call process_latent_in (legacy path).""" - helper = self._import_helper() - latent = torch.randn(1, 4, 32, 32) - calls = [] - def process_latent_in(x): - calls.append(1) - return x * 0.13025 - out = helper(latent, process_latent_in, latent_dimensions=2, - operate_in_vae_space=False) - assert len(calls) == 1, "process_latent_in must be called in model space" - assert torch.allclose(out, latent * 0.13025) - - def test_none_process_latent_in_is_noop(self): - """process_latent_in=None must be a no-op in both modes.""" - helper = self._import_helper() - latent = torch.randn(1, 4, 32, 32) - out_vae = helper(latent, None, latent_dimensions=2, operate_in_vae_space=True) - out_model = helper(latent, None, latent_dimensions=2, operate_in_vae_space=False) - assert torch.equal(out_vae, latent) - assert torch.equal(out_model, latent) - - def test_3d_unsqueezes_before_process_latent_in(self): - """3D model-space path must unsqueeze 4D -> 5D before process_latent_in.""" + def test_3d_unsqueezes_4d_to_5d(self): + """3D latent models: 4D input is unsqueezed to 5D [B, C, 1, H, W].""" helper = self._import_helper() latent = torch.randn(1, 4, 32, 32) # 4D - seen_shape = {} - def process_latent_in(x): - seen_shape["shape"] = tuple(x.shape) - return x - out = helper(latent, process_latent_in, latent_dimensions=3, - operate_in_vae_space=False) - assert seen_shape["shape"] == (1, 4, 1, 32, 32), ( - f"3D process_latent_in should receive 5D, got {seen_shape['shape']}" - ) + out = helper(latent, latent_dimensions=3) assert tuple(out.shape) == (1, 4, 1, 32, 32) - def test_3d_vae_space_skips_process_latent_in(self): - """3D VAE-space path must NOT call process_latent_in and keep 4D.""" + def test_3d_5d_passthrough(self): + """5D input stays 5D unchanged.""" helper = self._import_helper() - latent = torch.randn(1, 4, 32, 32) - calls = [] - def process_latent_in(x): - calls.append(1) - return x - out = helper(latent, process_latent_in, latent_dimensions=3, - operate_in_vae_space=True) - assert len(calls) == 0 - assert tuple(out.shape) == (1, 4, 32, 32) + latent = torch.randn(1, 4, 1, 32, 32) + out = helper(latent, latent_dimensions=3) + assert torch.equal(out, latent)