Merge pull request #117 - refactor: Replace keep_loaded with eject_models Boolean for improved memory management

This commit is contained in:
John Pollock
2025-10-04 18:30:03 -05:00
committed by GitHub
22 changed files with 210 additions and 245 deletions
+4 -4
View File
@@ -21,7 +21,7 @@ from .model_management_mgpu import (
)
WEB_DIRECTORY = "./web"
MGPU_MM_LOG = False
MGPU_MM_LOG = False
DEBUG_LOG = False
logger = logging.getLogger("MultiGPU")
@@ -85,12 +85,12 @@ def text_encoder_device_patched():
else:
devs = set(get_device_list())
device = torch.device(current_text_encoder_device) if str(current_text_encoder_device) in devs else torch.device("cpu")
logger.debug(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
logger.info(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
return device
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device and mm.text_encoder_device")
logger.debug(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
logger.debug(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
logger.info(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
logger.info(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
mm.get_torch_device = get_torch_device_patched
mm.text_encoder_device = text_encoder_device_patched
+132 -34
View File
@@ -58,51 +58,149 @@ def register_patched_safetensor_modelpatcher():
# Patch ComfyUI's ModelPatcher
if not hasattr(comfy.model_patcher.ModelPatcher, '_distorch_patched'):
# Patch LoadedModel.model_memory_required to drive behavior purely by Phase 2 = unload_distorch_model flag
from comfy.model_management import current_loaded_models
original_loaded_model_memory_required = None
for cls in current_loaded_models.__class__.__mro__:
if hasattr(cls, 'model_memory_required'):
original_loaded_model_memory_required = cls.model_memory_required
break
# PATCH load_models_gpu with correct memory calculations per model flags
original_load_models_gpu = mm.load_models_gpu
if original_loaded_model_memory_required is None:
# Global patch of LoadedModel class if available
import comfy.model_management as mm
def patched_load_models_gpu(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False):
from comfy.model_management import cleanup_models_gc, get_free_memory, free_memory, current_loaded_models
from comfy.model_management import VRAMState, vram_state, lowvram_available, MIN_WEIGHT_MEMORY_RATIO
from comfy.model_management import minimum_inference_memory, extra_reserved_memory, is_device_cpu
multigpu_memory_log("load_models_gpu_top_level", "start")
original_loaded_model_memory_required = mm.LoadedModel.model_memory_required
cleanup_models_gc()
def patched_loaded_model_memory_required(self, device):
"""Drive unload behavior purely by unload_distorch_model flag"""
multigpu_memory_log("unload_distorch_model_memory_check", "start")
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Memory assessment requested for model on device: {device}")
inference_memory = minimum_inference_memory()
extra_reserved_mem = extra_reserved_memory()
memory_required_total = memory_required + extra_reserved_mem
extra_mem = max(inference_memory, memory_required_total)
if minimum_memory_required is None:
minimum_memory_required = extra_mem
else:
minimum_memory_required = max(inference_memory, minimum_memory_required + extra_reserved_mem)
# Check if this is a DisTorch model with unload_distorch_model flag
is_distorch_model = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_unload_distorch_model')
models_temp = set()
for m in models:
models_temp.add(m)
for mm_patch in m.model_patches_models():
models_temp.add(mm_patch)
model_name = type(getattr(getattr(self, 'model', None), 'model', None)).__name__ if getattr(getattr(self, 'model', None), 'model', None) else "Unknown"
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] DisTorch model: {model_name}, is_distorch_model={is_distorch_model}")
models = models_temp
if is_distorch_model:
if self.model.model._mgpu_unload_distorch_model:
models_to_load = []
for x in models:
loaded_model = mm.LoadedModel(x)
try:
loaded_model_index = current_loaded_models.index(loaded_model)
except:
loaded_model_index = None
if loaded_model_index is not None:
loaded = current_loaded_models[loaded_model_index]
loaded.currently_used = True
models_to_load.append(loaded)
else:
if hasattr(x, "model"):
logging.info(f"Requested to load {x.model.__class__.__name__}")
models_to_load.append(loaded_model)
for loaded_model in models_to_load:
to_unload = []
for i in range(len(current_loaded_models)):
if loaded_model.model.is_clone(current_loaded_models[i].model):
to_unload = [i] + to_unload
for i in to_unload:
model_to_unload = current_loaded_models.pop(i)
model_to_unload.model.detach(unpatch_all=False)
model_to_unload.model_finalizer.detach()
# DisTorch Processing
total_memory_required = {}
eject_device = None
for loaded_model in models_to_load:
device = loaded_model.device
base_memory = loaded_model.model_memory_required(device)
# Check DisTorch flags
is_distorch = hasattr(loaded_model.model.model, '_mgpu_virtual_vram_gb')
has_eject = hasattr(loaded_model.model.model, '_mgpu_eject_models')
if has_eject:
eject_device = device
logger.mgpu_mm_log("DisTorch eject_models=True, is_distorch=True - MAX memory eviction")
if is_distorch:
# is_distorch=True: use compute device allocation size
virtual_vram_gb = loaded_model.model.model._mgpu_virtual_vram_gb
virtual_vram_bytes = virtual_vram_gb * (1024**3)
adjusted_memory = max(0, base_memory - virtual_vram_bytes)
total_memory_required[device] = total_memory_required.get(device, 0) + adjusted_memory
logger.mgpu_mm_log(f"DisTorch is_distorch=True, model adjusted {(base_memory - virtual_vram_bytes)/(1024**3):.2f}GB for device {device}")
else:
# is_distorch=False: use full model size
total_memory_required[device] = total_memory_required.get(device, 0) + base_memory
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] Standard model {(base_memory)/(1024**3):.2f}GB for device {device}")
for device in total_memory_required:
if device != torch.device("cpu"):
requested_mem = total_memory_required[device] * 1.1 + extra_mem
logger.mgpu_mm_log(f"[FREE_MEMORY_CALL] Device {device}: requesting {requested_mem/(1024**3):.2f}GB = {total_memory_required[device]/(1024**3):.2f}GB * 1.1 + {extra_mem/(1024**3):.2f}GB inference")
multigpu_memory_log("free_memory", "pre")
for device in total_memory_required:
if device != torch.device("cpu"):
if device == eject_device:
total_device_memory = mm.get_total_memory(device)
memory_gb = total_device_memory / (1024**3)
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=True - Reporting MAX memory ({memory_gb:.2f}GB) to force complete eviction")
return total_device_memory
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=1, is_distorch=1 → using MAX memory ({total_device_memory/(1024**3):.2f}GB) for eviction")
free_memory(total_device_memory,device)
else:
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=False - Reporting 0 bytes (prevents eviction)")
return 0
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=0, using Comfy Core Computed memory ({(total_memory_required[device] * 1.1 + extra_mem)/(1024**3):.2f}GB) for eviction")
free_memory(total_memory_required[device] * 1.1 + extra_mem, device)
multigpu_memory_log("free_memory/minimum_memory_required", "post/pre")
# Not a DisTorch model - use original behavior
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] Non-DisTorch model - Using original Comfy memory calculation")
original_result = original_loaded_model_memory_required(self, device)
original_gb = original_result / (1024**3) if original_result else 0
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Original calculation returned: {original_gb:.2f}GB")
multigpu_memory_log("keep_loaded_memory_check", "end")
return original_result
for device in total_memory_required:
if device != torch.device("cpu"):
free_mem = get_free_memory(device)
free_mem_gb = free_mem / (1024**3)
min_required_gb = minimum_memory_required / (1024**3)
logger.mgpu_mm_log(f"[MIN_MEMORY_CHECK] Device {device}: free={free_mem_gb:.2f}GB, required={min_required_gb:.2f}GB, will_evict={free_mem < minimum_memory_required}")
mm.LoadedModel.model_memory_required = patched_loaded_model_memory_required
if free_mem < minimum_memory_required:
models_l = free_memory(minimum_memory_required, device)
logger.mgpu_mm_log(f"[EVICTION] Device {device}: unloaded {len(models_l)} models due to insufficient memory")
logging.info("{} models unloaded.".format(len(models_l)))
multigpu_memory_log("minimum_memory_required", "post")
for loaded_model in models_to_load:
model = loaded_model.model
torch_dev = model.load_device
if is_device_cpu(torch_dev):
vram_set_state = VRAMState.DISABLED
else:
vram_set_state = vram_state
lowvram_model_memory = 0
if lowvram_available and (vram_set_state == VRAMState.LOW_VRAM or vram_set_state == VRAMState.NORMAL_VRAM) and not force_full_load:
loaded_memory = loaded_model.model_loaded_memory()
current_free_mem = get_free_memory(torch_dev) + loaded_memory
lowvram_model_memory = max(128 * 1024 * 1024, (current_free_mem - minimum_memory_required), min(current_free_mem * MIN_WEIGHT_MEMORY_RATIO, current_free_mem - minimum_inference_memory()))
lowvram_model_memory = max(0.1, lowvram_model_memory - loaded_memory)
if vram_set_state == VRAMState.NO_VRAM:
lowvram_model_memory = 0.1
loaded_model.model_load(lowvram_model_memory, force_patch_weights=force_patch_weights)
current_loaded_models.insert(0, loaded_model)
# Replace the module function
mm.load_models_gpu = patched_load_models_gpu
original_partially_load = comfy.model_patcher.ModelPatcher.partially_load
-153
View File
@@ -11,35 +11,12 @@ import comfy.model_management as mm
import gc
from datetime import datetime, timezone
import server
import weakref
import platform
import ctypes
import comfy.model_patcher
from collections import defaultdict
logger = logging.getLogger("MultiGPU")
# ==========================================================================================
# GC Anchor System for Model Retention
# ==========================================================================================
# Global anchor set to prevent GC of models during selective unload
_MGPU_RETENTION_ANCHORS = set()
def add_retention_anchor(model_patcher, reason="keep_loaded"):
"""Add a model patcher to the GC anchor set to prevent premature garbage collection"""
if model_patcher is not None:
_MGPU_RETENTION_ANCHORS.add(model_patcher)
model_name = type(getattr(model_patcher, 'model', model_patcher)).__name__
logger.mgpu_mm_log(f"[GC_ANCHOR] Added retention anchor for {model_name}, reason: {reason}, total anchors: {len(_MGPU_RETENTION_ANCHORS)}")
def clear_all_retention_anchors(reason="manual_clear"):
"""Clear all retention anchors"""
count = len(_MGPU_RETENTION_ANCHORS)
_MGPU_RETENTION_ANCHORS.clear()
logger.mgpu_mm_log(f"[GC_ANCHOR] Cleared all {count} retention anchors, reason: {reason}")
# ==========================================================================================
# Model Analysis and Store Management (DisTorch V1 & V2)
@@ -232,133 +209,3 @@ def force_full_system_cleanup(reason="manual", force=True):
summary = f"[ManagerMatch] Cleanup requested (reason={reason}) | models {pre_models}->{post_models}, cpu_delta_mb={delta_cpu_mb:.2f}"
logger.mgpu_mm_log(summary)
return summary
# ==========================================================================================
# Core Patching: unload_all_models
# ==========================================================================================
if not hasattr(mm.unload_all_models, '_mgpu_eject_distorch_patched'):
logger.info("[MultiGPU Core Patching] Patching mm.unload_all_models for DisTorch2 ejection support")
_mgpu_original_unload_all_models = mm.unload_all_models
def _mgpu_patched_unload_all_models():
"""Patched mm.unload_all_models with selective ejection support and comprehensive diagnostics."""
logger.mgpu_mm_log(f"[UNLOAD_START] Patched unload_all_models called - initial model count: {len(mm.current_loaded_models)}")
# Check if there are any DisTorch models that want to be unloaded
has_distorch_to_unload = any(
(hasattr(lm.model, '_mgpu_unload_distorch_model') and lm.model._mgpu_unload_distorch_model) or
(hasattr(getattr(lm.model, 'model', None), '_mgpu_unload_distorch_model') and lm.model.model._mgpu_unload_distorch_model)
for lm in mm.current_loaded_models
if lm.model is not None
)
if not has_distorch_to_unload:
logger.mgpu_mm_log("No DisTorch models requesting unload - clearing anchors and delegating to original unload_all_models")
clear_all_retention_anchors(reason="no_selective_unload_needed")
_mgpu_original_unload_all_models()
return
# Direct approach: iterate through loaded models and selectively unload
models_to_unload = []
kept_models = []
for i, lm in enumerate(mm.current_loaded_models):
mp = lm.model # weakref call to ModelPatcher
# DIAGNOSTIC: Log full object chain
lm_id = id(lm)
mp_id = id(mp)
inner_model = getattr(mp, 'model', None)
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
# Format inner_model_id properly for f-string
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_READ] Model {i}: lm_id=0x{lm_id:x}, mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
# FIX: Check flag on ModelPatcher (where it was set), not on inner model
# OLD BUG: unload_distorch_model = getattr(mp.model, '_mgpu_unload_distorch_model', False)
# NEW FIX: Check both locations to see which one has the flag
flag_on_mp = getattr(mp, '_mgpu_unload_distorch_model', None)
flag_on_inner = getattr(mp.model, '_mgpu_unload_distorch_model', None) if inner_model else None
logger.mgpu_mm_log(f"[FLAG_CHECK] Model {i} ({inner_model_name}): flag_on_mp={flag_on_mp}, flag_on_inner={flag_on_inner}")
# Use whichever location has the flag (for backwards compatibility during transition)
if flag_on_mp is not None:
unload_distorch_model = flag_on_mp
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from ModelPatcher (mp_id=0x{mp_id:x})")
elif flag_on_inner is not None:
unload_distorch_model = flag_on_inner
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from inner model (inner_model_id={inner_id_str})")
else:
unload_distorch_model = False
logger.mgpu_mm_log(f"[FLAG_SOURCE] No flag found - defaulting to False (keep loaded)")
logger.mgpu_mm_log(f"[DECISION] Model {i} ({inner_model_name}): unload_distorch_model={unload_distorch_model}")
if unload_distorch_model:
models_to_unload.append(lm)
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → models_to_unload")
else:
kept_models.append(lm)
add_retention_anchor(mp, "keep_loaded_protection")
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → kept_models")
# After the kept_models/models_to_unload evaluation
logger.mgpu_mm_log(f"[CATEGORIZE_SUMMARY] kept_models: {len(kept_models)}, models_to_unload: {len(models_to_unload)}, total: {len(mm.current_loaded_models)}")
if len(kept_models) == len(mm.current_loaded_models):
# All models are meant to be kept - no DisTorch selective unloading needed
logger.mgpu_mm_log("[DELEGATION] All models flagged to be kept - delegating to standard unload_all_models")
_mgpu_original_unload_all_models()
return
if kept_models:
logger.mgpu_mm_log(f"[SELECTIVE_UNLOAD] Proceeding with selective unload: retaining {len(kept_models)}, unloading {len(models_to_unload)}")
# Unload models flagged for unload
for lm in models_to_unload:
try:
model_name = type(lm.model.model).__name__ if lm.model and hasattr(lm.model, 'model') else 'Unknown'
logger.mgpu_mm_log(f"[UNLOAD_EXECUTE] Unloading model: {model_name} (lm_id=0x{id(lm):x})")
lm.model_unload(unpatch_weights=True)
except Exception as e:
logger.warning(f"[UNLOAD_ERROR] Error unloading model: {e}")
# WEAKREF TRACKING: Attach weakref callbacks to prove if kept models are GC'd
def model_deleted_callback(ref, model_name, model_id):
logger.mgpu_mm_log(f"[WEAKREF_DELETED] Kept model GARBAGE COLLECTED: {model_name} (id=0x{model_id:x})")
for i, lm in enumerate(kept_models):
mp = lm.model
inner_model = getattr(mp, 'model', None)
model_name = type(inner_model).__name__ if inner_model else 'Unknown'
model_id = id(lm)
weakref.ref(lm, lambda ref, name=model_name, mid=model_id: model_deleted_callback(ref, name, mid))
logger.mgpu_mm_log(f"[WEAKREF_ATTACHED] Tracking kept model {i}: {model_name} (lm_id=0x{model_id:x}, mp_id=0x{id(mp):x})")
# Remove unloaded models from current_loaded_models
mm.current_loaded_models = kept_models
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] Updated mm.current_loaded_models, new count: {len(mm.current_loaded_models)}")
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] mm.current_loaded_models id: 0x{id(mm.current_loaded_models):x}")
# DIAGNOSTIC: Log what's remaining
for i, lm in enumerate(mm.current_loaded_models):
mp = lm.model
inner_model = getattr(mp, 'model', None)
model_name = type(inner_model).__name__ if inner_model else "None"
logger.mgpu_mm_log(f"[REMAINING_MODEL] {i}: {model_name} (lm_id=0x{id(lm):x}, mp_id=0x{id(mp):x})")
else:
logger.mgpu_mm_log("[DELEGATION] No models with keep_loaded=True found - delegating to original unload_all_models")
_mgpu_original_unload_all_models()
mm.unload_all_models = _mgpu_patched_unload_all_models
mm.unload_all_models._mgpu_eject_distorch_patched = True
logger.info("[MultiGPU Core Patching] mm.unload_all_models patched successfully")
else:
logger.debug("[MultiGPU Core Patching] mm.unload_all_models already patched - skipping")
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "comfyui-multigpu"
description = "Provides a suite of custom nodes to manage multiple GPUs for ComfyUI, including advanced model offloading for both GGUF and Safetensor formats with DisTorch, and bespoke MultiGPU support for WanVideoWrapper and other custom nodes."
version = "2.5.0"
version = "2.5.1"
license = {file = "LICENSE"}
[project.urls]
+1 -1
View File
@@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
+1 -1
View File
@@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/clip_visio
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/checkpoint
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/controlnet
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
@@ -13,7 +13,7 @@ This node loads ControlNet models directly from HuggingFace model repositories b
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
+1 -1
View File
@@ -13,7 +13,7 @@ This node loads models directly from HuggingFace model repositories by specifyin
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
+1 -1
View File
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
## Outputs
+1 -1
View File
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet` fold
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf`
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
+1 -1
View File
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf`
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
+1 -1
View File
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/vae` folde
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
## Outputs
+56 -36
View File
@@ -15,7 +15,7 @@ logger = logging.getLogger("MultiGPU")
# DISTORCH V2 SAFETENSOR WRAPPERS (DisTorch2 for .safetensors and .gguf)
# ============================================================================
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround):
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround, eject_models_default=True):
"""Internal factory function creating DisTorch2 override class with parameterized device selection behavior."""
from .distorch_2 import (
register_patched_safetensor_modelpatcher,
@@ -37,7 +37,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
inputs["optional"]["keep_loaded"] = ("BOOLEAN", {"default": True})
inputs["optional"]["eject_models"] = ("BOOLEAN", {"default": eject_models_default})
return inputs
CATEGORY = "multigpu/distorch_2"
@@ -45,10 +45,10 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
@classmethod
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
expert_mode_allocations="", keep_loaded=True, **kwargs):
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
expert_mode_allocations="", eject_models=eject_models_default, **kwargs):
device_value = kwargs.get(device_param_name)
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{keep_loaded}"
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{eject_models}"
current_hash = hashlib.sha256(settings_str.encode()).hexdigest()
if not hasattr(cls, '_last_hash'):
@@ -60,19 +60,40 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
return current_hash
def override(self, *args, virtual_vram_gb=4.0, donor_device="cpu",
expert_mode_allocations="", keep_loaded=True, **kwargs):
expert_mode_allocations="", eject_models=eject_models_default, **kwargs):
device_value = kwargs.get(device_param_name)
unload_distorch_model = not keep_loaded
import comfy.model_management as mm
if eject_models:
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP] eject_models=True - marking all loaded models for eviction, device target: {device_value}")
ejection_count = 0
for i, lm in enumerate(mm.current_loaded_models):
# Set _mgpu_unload_distorch_model=True on all models to force Comfy Core eviction
model_name = type(getattr(lm.model, 'model', lm.model)).__name__ if lm.model else 'Unknown'
if hasattr(lm.model, 'model') and lm.model.model is not None:
lm.model.model._mgpu_unload_distorch_model = True
logger.mgpu_mm_log(f"[EJECT_MARKED] Model {i}: {model_name} (id=0x{id(lm):x}) → marked for eviction")
ejection_count += 1
elif lm.model is not None:
lm.model._mgpu_unload_distorch_model = True
logger.mgpu_mm_log(f"[EJECT_MARKED] Model {i}: {model_name} (direct patcher) → marked for eviction")
ejection_count += 1
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP_COMPLETE] Marked {ejection_count} models for Comfy Core eviction during load_models_gpu")
else:
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP] eject_models=False - loading without eviction")
if device_value is not None:
device_setter_func(device_value)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in [device_param_name, 'virtual_vram_gb',
'donor_device', 'expert_mode_allocations',
'keep_loaded']}
# Strip MultiGPU-specific parameters before calling original function (REMOVE eject_models, eject_models and virtual_vram_gb since we handle them above)
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in [device_param_name, 'virtual_vram_gb',
'donor_device', 'expert_mode_allocations',
'eject_models']}
if apply_device_kwarg_workaround:
clean_kwargs['device'] = 'default'
@@ -106,7 +127,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
logger.debug(f"[MultiGPU DisTorch V2] Stored allocation for model {model_hash[:8]}: {full_allocation}")
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
logger.mgpu_mm_log(f"[FLAG_SET_START] Setting '_mgpu_unload_distorch_model' to: {unload_distorch_model} (keep_loaded={keep_loaded})")
logger.mgpu_mm_log(f"[MODEL_SETUP] Setting DisTorch model properties: virtual_vram_gb={virtual_vram_gb}")
if hasattr(out[0], 'model'):
mp = out[0]
@@ -115,16 +136,19 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
mp._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
# SET VIRTUAL VRAM PROPERTY FOR MEMORY CALCULATION
if inner_model:
inner_model._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
inner_model._mgpu_virtual_vram_gb = virtual_vram_gb
logger.mgpu_mm_log(f"[VIRTUAL_VRAM_SET] Set _mgpu_virtual_vram_gb={virtual_vram_gb}GB on inner model (id=0x{inner_model_id:x}) for memory assessment")
# SET EJECT MODELS PROPERTY IF ENABLED
if eject_models and inner_model:
inner_model._mgpu_eject_models = True
logger.mgpu_mm_log(f"[EJECT_FLAG_SET] Set _mgpu_eject_models=True on inner model (id=0x{inner_model_id:x}) - will trigger ejection during load_models_gpu")
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
mp = out[0].patcher
mp_id = id(mp)
@@ -132,19 +156,13 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
mp._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
if inner_model:
inner_model._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
if unload_distorch_model:
logger.mgpu_mm_log("[FLAG_TRIGGER] unload_distorch_model=True, triggering full system cleanup")
force_full_system_cleanup(reason="policy_every_load", force=True)
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
# SET VIRTUAL VRAM PROPERTY FOR MEMORY CALCULATION
if inner_model:
inner_model._mgpu_virtual_vram_gb = virtual_vram_gb
logger.mgpu_mm_log(f"[VIRTUAL_VRAM_SET] Set _mgpu_virtual_vram_gb={virtual_vram_gb}GB on inner model (id=0x{inner_model_id:x}) for memory assessment")
return out
@@ -169,7 +187,8 @@ def override_class_with_distorch_safetensor_v2_clip(cls):
cls,
device_param_name="device",
device_setter_func=set_current_text_encoder_device,
apply_device_kwarg_workaround=True
apply_device_kwarg_workaround=True,
eject_models_default=False # CLIP defaults to False
)
@@ -180,7 +199,8 @@ def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
cls,
device_param_name="device",
device_setter_func=set_current_text_encoder_device,
apply_device_kwarg_workaround=False
apply_device_kwarg_workaround=False,
eject_models_default=False # CLIP defaults to False
)