Merge pull request #117 - refactor: Replace keep_loaded with eject_models Boolean for improved memory management
This commit is contained in:
+4
-4
@@ -21,7 +21,7 @@ from .model_management_mgpu import (
|
||||
)
|
||||
|
||||
WEB_DIRECTORY = "./web"
|
||||
MGPU_MM_LOG = False
|
||||
MGPU_MM_LOG = False
|
||||
DEBUG_LOG = False
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
@@ -85,12 +85,12 @@ def text_encoder_device_patched():
|
||||
else:
|
||||
devs = set(get_device_list())
|
||||
device = torch.device(current_text_encoder_device) if str(current_text_encoder_device) in devs else torch.device("cpu")
|
||||
logger.debug(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
|
||||
logger.info(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
|
||||
return device
|
||||
|
||||
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device and mm.text_encoder_device")
|
||||
logger.debug(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
|
||||
logger.debug(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
|
||||
logger.info(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
|
||||
logger.info(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
|
||||
mm.get_torch_device = get_torch_device_patched
|
||||
mm.text_encoder_device = text_encoder_device_patched
|
||||
|
||||
|
||||
+132
-34
@@ -58,51 +58,149 @@ def register_patched_safetensor_modelpatcher():
|
||||
# Patch ComfyUI's ModelPatcher
|
||||
if not hasattr(comfy.model_patcher.ModelPatcher, '_distorch_patched'):
|
||||
|
||||
# Patch LoadedModel.model_memory_required to drive behavior purely by Phase 2 = unload_distorch_model flag
|
||||
from comfy.model_management import current_loaded_models
|
||||
|
||||
original_loaded_model_memory_required = None
|
||||
for cls in current_loaded_models.__class__.__mro__:
|
||||
if hasattr(cls, 'model_memory_required'):
|
||||
original_loaded_model_memory_required = cls.model_memory_required
|
||||
break
|
||||
# PATCH load_models_gpu with correct memory calculations per model flags
|
||||
original_load_models_gpu = mm.load_models_gpu
|
||||
|
||||
if original_loaded_model_memory_required is None:
|
||||
# Global patch of LoadedModel class if available
|
||||
import comfy.model_management as mm
|
||||
def patched_load_models_gpu(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False):
|
||||
from comfy.model_management import cleanup_models_gc, get_free_memory, free_memory, current_loaded_models
|
||||
from comfy.model_management import VRAMState, vram_state, lowvram_available, MIN_WEIGHT_MEMORY_RATIO
|
||||
from comfy.model_management import minimum_inference_memory, extra_reserved_memory, is_device_cpu
|
||||
|
||||
multigpu_memory_log("load_models_gpu_top_level", "start")
|
||||
|
||||
original_loaded_model_memory_required = mm.LoadedModel.model_memory_required
|
||||
cleanup_models_gc()
|
||||
|
||||
def patched_loaded_model_memory_required(self, device):
|
||||
"""Drive unload behavior purely by unload_distorch_model flag"""
|
||||
multigpu_memory_log("unload_distorch_model_memory_check", "start")
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Memory assessment requested for model on device: {device}")
|
||||
inference_memory = minimum_inference_memory()
|
||||
extra_reserved_mem = extra_reserved_memory()
|
||||
memory_required_total = memory_required + extra_reserved_mem
|
||||
extra_mem = max(inference_memory, memory_required_total)
|
||||
if minimum_memory_required is None:
|
||||
minimum_memory_required = extra_mem
|
||||
else:
|
||||
minimum_memory_required = max(inference_memory, minimum_memory_required + extra_reserved_mem)
|
||||
|
||||
# Check if this is a DisTorch model with unload_distorch_model flag
|
||||
is_distorch_model = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_unload_distorch_model')
|
||||
models_temp = set()
|
||||
for m in models:
|
||||
models_temp.add(m)
|
||||
for mm_patch in m.model_patches_models():
|
||||
models_temp.add(mm_patch)
|
||||
|
||||
model_name = type(getattr(getattr(self, 'model', None), 'model', None)).__name__ if getattr(getattr(self, 'model', None), 'model', None) else "Unknown"
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] DisTorch model: {model_name}, is_distorch_model={is_distorch_model}")
|
||||
models = models_temp
|
||||
|
||||
if is_distorch_model:
|
||||
if self.model.model._mgpu_unload_distorch_model:
|
||||
models_to_load = []
|
||||
|
||||
for x in models:
|
||||
loaded_model = mm.LoadedModel(x)
|
||||
try:
|
||||
loaded_model_index = current_loaded_models.index(loaded_model)
|
||||
except:
|
||||
loaded_model_index = None
|
||||
|
||||
if loaded_model_index is not None:
|
||||
loaded = current_loaded_models[loaded_model_index]
|
||||
loaded.currently_used = True
|
||||
models_to_load.append(loaded)
|
||||
else:
|
||||
if hasattr(x, "model"):
|
||||
logging.info(f"Requested to load {x.model.__class__.__name__}")
|
||||
models_to_load.append(loaded_model)
|
||||
|
||||
for loaded_model in models_to_load:
|
||||
to_unload = []
|
||||
for i in range(len(current_loaded_models)):
|
||||
if loaded_model.model.is_clone(current_loaded_models[i].model):
|
||||
to_unload = [i] + to_unload
|
||||
for i in to_unload:
|
||||
model_to_unload = current_loaded_models.pop(i)
|
||||
model_to_unload.model.detach(unpatch_all=False)
|
||||
model_to_unload.model_finalizer.detach()
|
||||
|
||||
# DisTorch Processing
|
||||
total_memory_required = {}
|
||||
eject_device = None
|
||||
|
||||
for loaded_model in models_to_load:
|
||||
device = loaded_model.device
|
||||
base_memory = loaded_model.model_memory_required(device)
|
||||
|
||||
# Check DisTorch flags
|
||||
is_distorch = hasattr(loaded_model.model.model, '_mgpu_virtual_vram_gb')
|
||||
has_eject = hasattr(loaded_model.model.model, '_mgpu_eject_models')
|
||||
|
||||
if has_eject:
|
||||
eject_device = device
|
||||
logger.mgpu_mm_log("DisTorch eject_models=True, is_distorch=True - MAX memory eviction")
|
||||
|
||||
if is_distorch:
|
||||
# is_distorch=True: use compute device allocation size
|
||||
virtual_vram_gb = loaded_model.model.model._mgpu_virtual_vram_gb
|
||||
virtual_vram_bytes = virtual_vram_gb * (1024**3)
|
||||
adjusted_memory = max(0, base_memory - virtual_vram_bytes)
|
||||
total_memory_required[device] = total_memory_required.get(device, 0) + adjusted_memory
|
||||
logger.mgpu_mm_log(f"DisTorch is_distorch=True, model adjusted {(base_memory - virtual_vram_bytes)/(1024**3):.2f}GB for device {device}")
|
||||
else:
|
||||
# is_distorch=False: use full model size
|
||||
total_memory_required[device] = total_memory_required.get(device, 0) + base_memory
|
||||
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] Standard model {(base_memory)/(1024**3):.2f}GB for device {device}")
|
||||
|
||||
for device in total_memory_required:
|
||||
if device != torch.device("cpu"):
|
||||
requested_mem = total_memory_required[device] * 1.1 + extra_mem
|
||||
logger.mgpu_mm_log(f"[FREE_MEMORY_CALL] Device {device}: requesting {requested_mem/(1024**3):.2f}GB = {total_memory_required[device]/(1024**3):.2f}GB * 1.1 + {extra_mem/(1024**3):.2f}GB inference")
|
||||
|
||||
|
||||
multigpu_memory_log("free_memory", "pre")
|
||||
|
||||
for device in total_memory_required:
|
||||
if device != torch.device("cpu"):
|
||||
if device == eject_device:
|
||||
total_device_memory = mm.get_total_memory(device)
|
||||
memory_gb = total_device_memory / (1024**3)
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=True - Reporting MAX memory ({memory_gb:.2f}GB) to force complete eviction")
|
||||
return total_device_memory
|
||||
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=1, is_distorch=1 → using MAX memory ({total_device_memory/(1024**3):.2f}GB) for eviction")
|
||||
free_memory(total_device_memory,device)
|
||||
else:
|
||||
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=False - Reporting 0 bytes (prevents eviction)")
|
||||
return 0
|
||||
logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=0, using Comfy Core Computed memory ({(total_memory_required[device] * 1.1 + extra_mem)/(1024**3):.2f}GB) for eviction")
|
||||
free_memory(total_memory_required[device] * 1.1 + extra_mem, device)
|
||||
|
||||
multigpu_memory_log("free_memory/minimum_memory_required", "post/pre")
|
||||
|
||||
# Not a DisTorch model - use original behavior
|
||||
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] Non-DisTorch model - Using original Comfy memory calculation")
|
||||
original_result = original_loaded_model_memory_required(self, device)
|
||||
original_gb = original_result / (1024**3) if original_result else 0
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Original calculation returned: {original_gb:.2f}GB")
|
||||
multigpu_memory_log("keep_loaded_memory_check", "end")
|
||||
return original_result
|
||||
for device in total_memory_required:
|
||||
if device != torch.device("cpu"):
|
||||
free_mem = get_free_memory(device)
|
||||
free_mem_gb = free_mem / (1024**3)
|
||||
min_required_gb = minimum_memory_required / (1024**3)
|
||||
logger.mgpu_mm_log(f"[MIN_MEMORY_CHECK] Device {device}: free={free_mem_gb:.2f}GB, required={min_required_gb:.2f}GB, will_evict={free_mem < minimum_memory_required}")
|
||||
|
||||
mm.LoadedModel.model_memory_required = patched_loaded_model_memory_required
|
||||
if free_mem < minimum_memory_required:
|
||||
models_l = free_memory(minimum_memory_required, device)
|
||||
logger.mgpu_mm_log(f"[EVICTION] Device {device}: unloaded {len(models_l)} models due to insufficient memory")
|
||||
logging.info("{} models unloaded.".format(len(models_l)))
|
||||
|
||||
multigpu_memory_log("minimum_memory_required", "post")
|
||||
|
||||
for loaded_model in models_to_load:
|
||||
model = loaded_model.model
|
||||
torch_dev = model.load_device
|
||||
if is_device_cpu(torch_dev):
|
||||
vram_set_state = VRAMState.DISABLED
|
||||
else:
|
||||
vram_set_state = vram_state
|
||||
lowvram_model_memory = 0
|
||||
if lowvram_available and (vram_set_state == VRAMState.LOW_VRAM or vram_set_state == VRAMState.NORMAL_VRAM) and not force_full_load:
|
||||
loaded_memory = loaded_model.model_loaded_memory()
|
||||
current_free_mem = get_free_memory(torch_dev) + loaded_memory
|
||||
|
||||
lowvram_model_memory = max(128 * 1024 * 1024, (current_free_mem - minimum_memory_required), min(current_free_mem * MIN_WEIGHT_MEMORY_RATIO, current_free_mem - minimum_inference_memory()))
|
||||
lowvram_model_memory = max(0.1, lowvram_model_memory - loaded_memory)
|
||||
|
||||
if vram_set_state == VRAMState.NO_VRAM:
|
||||
lowvram_model_memory = 0.1
|
||||
|
||||
loaded_model.model_load(lowvram_model_memory, force_patch_weights=force_patch_weights)
|
||||
current_loaded_models.insert(0, loaded_model)
|
||||
|
||||
# Replace the module function
|
||||
mm.load_models_gpu = patched_load_models_gpu
|
||||
|
||||
original_partially_load = comfy.model_patcher.ModelPatcher.partially_load
|
||||
|
||||
|
||||
@@ -11,35 +11,12 @@ import comfy.model_management as mm
|
||||
import gc
|
||||
from datetime import datetime, timezone
|
||||
import server
|
||||
import weakref
|
||||
import platform
|
||||
import ctypes
|
||||
import comfy.model_patcher
|
||||
from collections import defaultdict
|
||||
|
||||
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
|
||||
# ==========================================================================================
|
||||
# GC Anchor System for Model Retention
|
||||
# ==========================================================================================
|
||||
|
||||
# Global anchor set to prevent GC of models during selective unload
|
||||
_MGPU_RETENTION_ANCHORS = set()
|
||||
|
||||
def add_retention_anchor(model_patcher, reason="keep_loaded"):
|
||||
"""Add a model patcher to the GC anchor set to prevent premature garbage collection"""
|
||||
if model_patcher is not None:
|
||||
_MGPU_RETENTION_ANCHORS.add(model_patcher)
|
||||
model_name = type(getattr(model_patcher, 'model', model_patcher)).__name__
|
||||
logger.mgpu_mm_log(f"[GC_ANCHOR] Added retention anchor for {model_name}, reason: {reason}, total anchors: {len(_MGPU_RETENTION_ANCHORS)}")
|
||||
|
||||
def clear_all_retention_anchors(reason="manual_clear"):
|
||||
"""Clear all retention anchors"""
|
||||
count = len(_MGPU_RETENTION_ANCHORS)
|
||||
_MGPU_RETENTION_ANCHORS.clear()
|
||||
logger.mgpu_mm_log(f"[GC_ANCHOR] Cleared all {count} retention anchors, reason: {reason}")
|
||||
|
||||
# ==========================================================================================
|
||||
# Model Analysis and Store Management (DisTorch V1 & V2)
|
||||
@@ -232,133 +209,3 @@ def force_full_system_cleanup(reason="manual", force=True):
|
||||
summary = f"[ManagerMatch] Cleanup requested (reason={reason}) | models {pre_models}->{post_models}, cpu_delta_mb={delta_cpu_mb:.2f}"
|
||||
logger.mgpu_mm_log(summary)
|
||||
return summary
|
||||
|
||||
# ==========================================================================================
|
||||
# Core Patching: unload_all_models
|
||||
# ==========================================================================================
|
||||
|
||||
if not hasattr(mm.unload_all_models, '_mgpu_eject_distorch_patched'):
|
||||
logger.info("[MultiGPU Core Patching] Patching mm.unload_all_models for DisTorch2 ejection support")
|
||||
|
||||
_mgpu_original_unload_all_models = mm.unload_all_models
|
||||
|
||||
def _mgpu_patched_unload_all_models():
|
||||
"""Patched mm.unload_all_models with selective ejection support and comprehensive diagnostics."""
|
||||
|
||||
logger.mgpu_mm_log(f"[UNLOAD_START] Patched unload_all_models called - initial model count: {len(mm.current_loaded_models)}")
|
||||
|
||||
# Check if there are any DisTorch models that want to be unloaded
|
||||
has_distorch_to_unload = any(
|
||||
(hasattr(lm.model, '_mgpu_unload_distorch_model') and lm.model._mgpu_unload_distorch_model) or
|
||||
(hasattr(getattr(lm.model, 'model', None), '_mgpu_unload_distorch_model') and lm.model.model._mgpu_unload_distorch_model)
|
||||
for lm in mm.current_loaded_models
|
||||
if lm.model is not None
|
||||
)
|
||||
|
||||
if not has_distorch_to_unload:
|
||||
logger.mgpu_mm_log("No DisTorch models requesting unload - clearing anchors and delegating to original unload_all_models")
|
||||
clear_all_retention_anchors(reason="no_selective_unload_needed")
|
||||
_mgpu_original_unload_all_models()
|
||||
return
|
||||
|
||||
# Direct approach: iterate through loaded models and selectively unload
|
||||
models_to_unload = []
|
||||
kept_models = []
|
||||
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
mp = lm.model # weakref call to ModelPatcher
|
||||
|
||||
# DIAGNOSTIC: Log full object chain
|
||||
lm_id = id(lm)
|
||||
mp_id = id(mp)
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
|
||||
# Format inner_model_id properly for f-string
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_READ] Model {i}: lm_id=0x{lm_id:x}, mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
# FIX: Check flag on ModelPatcher (where it was set), not on inner model
|
||||
# OLD BUG: unload_distorch_model = getattr(mp.model, '_mgpu_unload_distorch_model', False)
|
||||
# NEW FIX: Check both locations to see which one has the flag
|
||||
flag_on_mp = getattr(mp, '_mgpu_unload_distorch_model', None)
|
||||
flag_on_inner = getattr(mp.model, '_mgpu_unload_distorch_model', None) if inner_model else None
|
||||
|
||||
logger.mgpu_mm_log(f"[FLAG_CHECK] Model {i} ({inner_model_name}): flag_on_mp={flag_on_mp}, flag_on_inner={flag_on_inner}")
|
||||
|
||||
# Use whichever location has the flag (for backwards compatibility during transition)
|
||||
if flag_on_mp is not None:
|
||||
unload_distorch_model = flag_on_mp
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from ModelPatcher (mp_id=0x{mp_id:x})")
|
||||
elif flag_on_inner is not None:
|
||||
unload_distorch_model = flag_on_inner
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from inner model (inner_model_id={inner_id_str})")
|
||||
else:
|
||||
unload_distorch_model = False
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] No flag found - defaulting to False (keep loaded)")
|
||||
|
||||
logger.mgpu_mm_log(f"[DECISION] Model {i} ({inner_model_name}): unload_distorch_model={unload_distorch_model}")
|
||||
|
||||
if unload_distorch_model:
|
||||
models_to_unload.append(lm)
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → models_to_unload")
|
||||
else:
|
||||
kept_models.append(lm)
|
||||
add_retention_anchor(mp, "keep_loaded_protection")
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → kept_models")
|
||||
|
||||
# After the kept_models/models_to_unload evaluation
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE_SUMMARY] kept_models: {len(kept_models)}, models_to_unload: {len(models_to_unload)}, total: {len(mm.current_loaded_models)}")
|
||||
|
||||
if len(kept_models) == len(mm.current_loaded_models):
|
||||
# All models are meant to be kept - no DisTorch selective unloading needed
|
||||
logger.mgpu_mm_log("[DELEGATION] All models flagged to be kept - delegating to standard unload_all_models")
|
||||
_mgpu_original_unload_all_models()
|
||||
return
|
||||
|
||||
if kept_models:
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_UNLOAD] Proceeding with selective unload: retaining {len(kept_models)}, unloading {len(models_to_unload)}")
|
||||
|
||||
# Unload models flagged for unload
|
||||
for lm in models_to_unload:
|
||||
try:
|
||||
model_name = type(lm.model.model).__name__ if lm.model and hasattr(lm.model, 'model') else 'Unknown'
|
||||
logger.mgpu_mm_log(f"[UNLOAD_EXECUTE] Unloading model: {model_name} (lm_id=0x{id(lm):x})")
|
||||
lm.model_unload(unpatch_weights=True)
|
||||
except Exception as e:
|
||||
logger.warning(f"[UNLOAD_ERROR] Error unloading model: {e}")
|
||||
|
||||
# WEAKREF TRACKING: Attach weakref callbacks to prove if kept models are GC'd
|
||||
def model_deleted_callback(ref, model_name, model_id):
|
||||
logger.mgpu_mm_log(f"[WEAKREF_DELETED] Kept model GARBAGE COLLECTED: {model_name} (id=0x{model_id:x})")
|
||||
|
||||
for i, lm in enumerate(kept_models):
|
||||
mp = lm.model
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
model_name = type(inner_model).__name__ if inner_model else 'Unknown'
|
||||
model_id = id(lm)
|
||||
weakref.ref(lm, lambda ref, name=model_name, mid=model_id: model_deleted_callback(ref, name, mid))
|
||||
logger.mgpu_mm_log(f"[WEAKREF_ATTACHED] Tracking kept model {i}: {model_name} (lm_id=0x{model_id:x}, mp_id=0x{id(mp):x})")
|
||||
|
||||
# Remove unloaded models from current_loaded_models
|
||||
mm.current_loaded_models = kept_models
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] Updated mm.current_loaded_models, new count: {len(mm.current_loaded_models)}")
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] mm.current_loaded_models id: 0x{id(mm.current_loaded_models):x}")
|
||||
|
||||
# DIAGNOSTIC: Log what's remaining
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
mp = lm.model
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
logger.mgpu_mm_log(f"[REMAINING_MODEL] {i}: {model_name} (lm_id=0x{id(lm):x}, mp_id=0x{id(mp):x})")
|
||||
else:
|
||||
logger.mgpu_mm_log("[DELEGATION] No models with keep_loaded=True found - delegating to original unload_all_models")
|
||||
_mgpu_original_unload_all_models()
|
||||
|
||||
mm.unload_all_models = _mgpu_patched_unload_all_models
|
||||
mm.unload_all_models._mgpu_eject_distorch_patched = True
|
||||
logger.info("[MultiGPU Core Patching] mm.unload_all_models patched successfully")
|
||||
else:
|
||||
logger.debug("[MultiGPU Core Patching] mm.unload_all_models already patched - skipping")
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "comfyui-multigpu"
|
||||
description = "Provides a suite of custom nodes to manage multiple GPUs for ComfyUI, including advanced model offloading for both GGUF and Safetensor formats with DisTorch, and bespoke MultiGPU support for WanVideoWrapper and other custom nodes."
|
||||
version = "2.5.0"
|
||||
version = "2.5.1"
|
||||
license = {file = "LICENSE"}
|
||||
|
||||
[project.urls]
|
||||
|
||||
@@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/clip_visio
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/checkpoint
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/controlnet
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node loads ControlNet models directly from HuggingFace model repositories b
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node loads models directly from HuggingFace model repositories by specifyin
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet` fold
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf`
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf`
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
@@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/vae` folde
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
|
||||
+56
-36
@@ -15,7 +15,7 @@ logger = logging.getLogger("MultiGPU")
|
||||
# DISTORCH V2 SAFETENSOR WRAPPERS (DisTorch2 for .safetensors and .gguf)
|
||||
# ============================================================================
|
||||
|
||||
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround):
|
||||
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround, eject_models_default=True):
|
||||
"""Internal factory function creating DisTorch2 override class with parameterized device selection behavior."""
|
||||
from .distorch_2 import (
|
||||
register_patched_safetensor_modelpatcher,
|
||||
@@ -37,7 +37,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
inputs["optional"]["keep_loaded"] = ("BOOLEAN", {"default": True})
|
||||
inputs["optional"]["eject_models"] = ("BOOLEAN", {"default": eject_models_default})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
@@ -45,10 +45,10 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
|
||||
expert_mode_allocations="", keep_loaded=True, **kwargs):
|
||||
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
|
||||
expert_mode_allocations="", eject_models=eject_models_default, **kwargs):
|
||||
device_value = kwargs.get(device_param_name)
|
||||
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{keep_loaded}"
|
||||
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{eject_models}"
|
||||
current_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
if not hasattr(cls, '_last_hash'):
|
||||
@@ -60,19 +60,40 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
return current_hash
|
||||
|
||||
def override(self, *args, virtual_vram_gb=4.0, donor_device="cpu",
|
||||
expert_mode_allocations="", keep_loaded=True, **kwargs):
|
||||
|
||||
expert_mode_allocations="", eject_models=eject_models_default, **kwargs):
|
||||
|
||||
device_value = kwargs.get(device_param_name)
|
||||
unload_distorch_model = not keep_loaded
|
||||
|
||||
import comfy.model_management as mm
|
||||
|
||||
if eject_models:
|
||||
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP] eject_models=True - marking all loaded models for eviction, device target: {device_value}")
|
||||
ejection_count = 0
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
# Set _mgpu_unload_distorch_model=True on all models to force Comfy Core eviction
|
||||
model_name = type(getattr(lm.model, 'model', lm.model)).__name__ if lm.model else 'Unknown'
|
||||
|
||||
if hasattr(lm.model, 'model') and lm.model.model is not None:
|
||||
lm.model.model._mgpu_unload_distorch_model = True
|
||||
logger.mgpu_mm_log(f"[EJECT_MARKED] Model {i}: {model_name} (id=0x{id(lm):x}) → marked for eviction")
|
||||
ejection_count += 1
|
||||
elif lm.model is not None:
|
||||
lm.model._mgpu_unload_distorch_model = True
|
||||
logger.mgpu_mm_log(f"[EJECT_MARKED] Model {i}: {model_name} (direct patcher) → marked for eviction")
|
||||
ejection_count += 1
|
||||
|
||||
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP_COMPLETE] Marked {ejection_count} models for Comfy Core eviction during load_models_gpu")
|
||||
else:
|
||||
logger.mgpu_mm_log(f"[EJECT_MODELS_SETUP] eject_models=False - loading without eviction")
|
||||
|
||||
if device_value is not None:
|
||||
device_setter_func(device_value)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in [device_param_name, 'virtual_vram_gb',
|
||||
'donor_device', 'expert_mode_allocations',
|
||||
'keep_loaded']}
|
||||
# Strip MultiGPU-specific parameters before calling original function (REMOVE eject_models, eject_models and virtual_vram_gb since we handle them above)
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in [device_param_name, 'virtual_vram_gb',
|
||||
'donor_device', 'expert_mode_allocations',
|
||||
'eject_models']}
|
||||
|
||||
if apply_device_kwarg_workaround:
|
||||
clean_kwargs['device'] = 'default'
|
||||
@@ -106,7 +127,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Stored allocation for model {model_hash[:8]}: {full_allocation}")
|
||||
|
||||
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_START] Setting '_mgpu_unload_distorch_model' to: {unload_distorch_model} (keep_loaded={keep_loaded})")
|
||||
logger.mgpu_mm_log(f"[MODEL_SETUP] Setting DisTorch model properties: virtual_vram_gb={virtual_vram_gb}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
mp = out[0]
|
||||
@@ -115,16 +136,19 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
mp._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
|
||||
|
||||
|
||||
# SET VIRTUAL VRAM PROPERTY FOR MEMORY CALCULATION
|
||||
if inner_model:
|
||||
inner_model._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
|
||||
|
||||
inner_model._mgpu_virtual_vram_gb = virtual_vram_gb
|
||||
logger.mgpu_mm_log(f"[VIRTUAL_VRAM_SET] Set _mgpu_virtual_vram_gb={virtual_vram_gb}GB on inner model (id=0x{inner_model_id:x}) for memory assessment")
|
||||
|
||||
# SET EJECT MODELS PROPERTY IF ENABLED
|
||||
if eject_models and inner_model:
|
||||
inner_model._mgpu_eject_models = True
|
||||
logger.mgpu_mm_log(f"[EJECT_FLAG_SET] Set _mgpu_eject_models=True on inner model (id=0x{inner_model_id:x}) - will trigger ejection during load_models_gpu")
|
||||
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
mp = out[0].patcher
|
||||
mp_id = id(mp)
|
||||
@@ -132,19 +156,13 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
mp._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
|
||||
|
||||
if inner_model:
|
||||
inner_model._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
|
||||
|
||||
if unload_distorch_model:
|
||||
logger.mgpu_mm_log("[FLAG_TRIGGER] unload_distorch_model=True, triggering full system cleanup")
|
||||
force_full_system_cleanup(reason="policy_every_load", force=True)
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
# SET VIRTUAL VRAM PROPERTY FOR MEMORY CALCULATION
|
||||
if inner_model:
|
||||
inner_model._mgpu_virtual_vram_gb = virtual_vram_gb
|
||||
logger.mgpu_mm_log(f"[VIRTUAL_VRAM_SET] Set _mgpu_virtual_vram_gb={virtual_vram_gb}GB on inner model (id=0x{inner_model_id:x}) for memory assessment")
|
||||
|
||||
return out
|
||||
|
||||
@@ -169,7 +187,8 @@ def override_class_with_distorch_safetensor_v2_clip(cls):
|
||||
cls,
|
||||
device_param_name="device",
|
||||
device_setter_func=set_current_text_encoder_device,
|
||||
apply_device_kwarg_workaround=True
|
||||
apply_device_kwarg_workaround=True,
|
||||
eject_models_default=False # CLIP defaults to False
|
||||
)
|
||||
|
||||
|
||||
@@ -180,7 +199,8 @@ def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
|
||||
cls,
|
||||
device_param_name="device",
|
||||
device_setter_func=set_current_text_encoder_device,
|
||||
apply_device_kwarg_workaround=False
|
||||
apply_device_kwarg_workaround=False,
|
||||
eject_models_default=False # CLIP defaults to False
|
||||
)
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user