From 29b0a34865f2488529e3d93f12bcc981b65907b9 Mon Sep 17 00:00:00 2001 From: xmarre Date: Thu, 16 Apr 2026 08:11:55 +0200 Subject: [PATCH 1/3] Fix external trim gate for sticky VAE --- patches.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/patches.py b/patches.py index db4d789..d923c67 100644 --- a/patches.py +++ b/patches.py @@ -772,6 +772,7 @@ def _prepare_sticky_vae_batch( sticky_floor_priority=0, allow_partial_unload=True, keep_models=(patcher,), + include_external=True, ) except Exception as exc: _LOG.debug( @@ -780,6 +781,10 @@ def _prepare_sticky_vae_batch( model_load_required, exc, ) + else: + _LOG.info( + "GPU Resident Loader: native sticky VAE trim was insufficient; retried with external candidates enabled" + ) free_memory = _sticky_vae_free_memory(device=device, patcher=patcher) batch_budget = 0 if model_load_target is None else max(0, free_memory - model_load_target) @@ -1502,7 +1507,7 @@ def _wrap_free_memory(func: Callable[..., Any]) -> Callable[..., Any]: EXTERNAL_REGISTRY.refresh_runtime_state() for_dynamic = bool(kwargs.get("for_dynamic", args[0] if args else False)) - if device is not None and not external_trim_enabled() and not for_dynamic: + if device is not None and external_trim_enabled() and not for_dynamic: fallback_target = memory_required if REGISTRY.get_policy() == "sticky_gpu": fallback_target = max(fallback_target, _sticky_protection_target(memory_required, device)) From 84d10add7056b175f2e6a5b0e95464285a1f82fd Mon Sep 17 00:00:00 2001 From: xmarre Date: Thu, 16 Apr 2026 08:17:59 +0200 Subject: [PATCH 2/3] Add second-chance external VAE trim --- patches.py | 29 ++++++++++++++++++++++++----- 1 file changed, 24 insertions(+), 5 deletions(-) diff --git a/patches.py b/patches.py index d923c67..e39e162 100644 --- a/patches.py +++ b/patches.py @@ -772,7 +772,6 @@ def _prepare_sticky_vae_batch( sticky_floor_priority=0, allow_partial_unload=True, keep_models=(patcher,), - include_external=True, ) except Exception as exc: _LOG.debug( @@ -781,12 +780,32 @@ def _prepare_sticky_vae_batch( model_load_required, exc, ) - else: - _LOG.info( - "GPU Resident Loader: native sticky VAE trim was insufficient; retried with external candidates enabled" - ) free_memory = _sticky_vae_free_memory(device=device, patcher=patcher) + if free_memory < target_free: + try: + trim_resident_vram( + device=device, + target_free_vram_bytes=target_free, + respect_sticky=True, + sticky_floor_priority=0, + allow_partial_unload=True, + keep_models=(patcher,), + include_external=True, + ) + except Exception as exc: + _LOG.debug( + "GPU Resident Loader: second-chance external VAE trim failed for batch=%s load=%s bytes: %s", + batch_memory_used, + model_load_required, + exc, + ) + else: + _LOG.info( + "GPU Resident Loader: native sticky VAE trim was insufficient; retried with external candidates enabled" + ) + + free_memory = _sticky_vae_free_memory(device=device, patcher=patcher) batch_budget = 0 if model_load_target is None else max(0, free_memory - model_load_target) batch_number = _sticky_safe_batch_number( batch_count=total_batch_count, From 026d8527f860857da32bafa04008347d08865163 Mon Sep 17 00:00:00 2001 From: xmarre Date: Thu, 16 Apr 2026 08:26:30 +0200 Subject: [PATCH 3/3] Make VAE preflight trim respect external opt-in --- patches.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/patches.py b/patches.py index e39e162..0a6933c 100644 --- a/patches.py +++ b/patches.py @@ -772,6 +772,7 @@ def _prepare_sticky_vae_batch( sticky_floor_priority=0, allow_partial_unload=True, keep_models=(patcher,), + include_external=False, ) except Exception as exc: _LOG.debug( @@ -782,7 +783,7 @@ def _prepare_sticky_vae_batch( ) free_memory = _sticky_vae_free_memory(device=device, patcher=patcher) - if free_memory < target_free: + if free_memory < target_free and external_trim_enabled(): try: trim_resident_vram( device=device,