diff --git a/distorch_2.py b/distorch_2.py index 733d677..9b1e159 100644 --- a/distorch_2.py +++ b/distorch_2.py @@ -58,96 +58,149 @@ def register_patched_safetensor_modelpatcher(): # Patch ComfyUI's ModelPatcher if not hasattr(comfy.model_patcher.ModelPatcher, '_distorch_patched'): - # Patch LoadedModel.model_memory_required to drive behavior purely by Phase 2 = unload_distorch_model flag - from comfy.model_management import current_loaded_models - original_loaded_model_memory_required = None - for cls in current_loaded_models.__class__.__mro__: - if hasattr(cls, 'model_memory_required'): - original_loaded_model_memory_required = cls.model_memory_required - break + # PATCH load_models_gpu with correct memory calculations per model flags + original_load_models_gpu = mm.load_models_gpu - if original_loaded_model_memory_required is None: - # Global patch of LoadedModel class if available - import comfy.model_management as mm + def patched_load_models_gpu(models, memory_required=0, force_patch_weights=False, minimum_memory_required=None, force_full_load=False): + from comfy.model_management import cleanup_models_gc, get_free_memory, free_memory, current_loaded_models + from comfy.model_management import VRAMState, vram_state, lowvram_available, MIN_WEIGHT_MEMORY_RATIO + from comfy.model_management import minimum_inference_memory, extra_reserved_memory, is_device_cpu + + multigpu_memory_log("load_models_gpu_top_level", "start") - original_loaded_model_memory_required = mm.LoadedModel.model_memory_required + cleanup_models_gc() - def patched_loaded_model_memory_required(self, device): - """Truth table for memory reporting: - eject_models=0, is_distorch=0: return original - eject_models=0, is_distorch=1: return original - virtual_vram_gb_bytes - eject_models=1, is_distorch=0: mutually exclusive (shouldn't occur) - eject_models=1, is_distorch=1: return MAX memory to force eviction""" - multigpu_memory_log("unload_distorch_model_memory_check", "start") - model_name = type(getattr(getattr(self, 'model', None), 'model', None)).__name__ if getattr(getattr(self, 'model', None), 'model', None) else "Unknown" - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] Memory assessment requested for model on device: {device}") + inference_memory = minimum_inference_memory() + extra_reserved_mem = extra_reserved_memory() + memory_required_total = memory_required + extra_reserved_mem + extra_mem = max(inference_memory, memory_required_total) + if minimum_memory_required is None: + minimum_memory_required = extra_mem + else: + minimum_memory_required = max(inference_memory, minimum_memory_required + extra_reserved_mem) - # GET ORIGINAL MEMORY REQUIREMENT - original_result = original_loaded_model_memory_required(self, device) - original_gb = original_result / (1024**3) if original_result else 0 + models_temp = set() + for m in models: + models_temp.add(m) + for mm_patch in m.model_patches_models(): + models_temp.add(mm_patch) - # CHECK FOR EJECT_MODELS PROPERTY - has_eject_models = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_eject_models') - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] Original needs: {original_gb:.2f}GB, has_eject_models={has_eject_models}") + models = models_temp - # CHECK IF DISTORCH MODEL WITH VIRTUAL VRAM PROPERTY - is_distorch_model = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_virtual_vram_gb') + models_to_load = [] - # TRUTH TABLE APPLICATION - if has_eject_models: - if not is_distorch_model: - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] ERROR: eject_models=1 but not DisTorch (mutually exclusive)") - # eject_models=1, is_distorch=1: RETURN MAX MEMORY TO FORCE EVICTION - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] eject_models=1, is_distorch={is_distorch_model} → FORCING EVICTION WITH MAX MEMORY") - - # DISABLED: Manual ejection should happen automatically when MAX memory is returned - DISABLE_MANUAL_EJECTION = True # TODO: Remove this once auto-eviction confirmed - if not DISABLE_MANUAL_EJECTION: - logger.mgpu_mm_log(f"======= DIRECT MODEL EJECTION START[{model_name}] =======") - logger.mgpu_mm_log(f"[DIRECT_EJECTION][{model_name}] Current loaded models count: {len(mm.current_loaded_models)}") - - # DIRECTLY UNLOAD ALL MODELS - models_unloaded = [] - for i, lm in enumerate(mm.current_loaded_models): - model_name_to_eject = type(getattr(lm.model, 'model', lm.model)).__name__ if lm.model else 'Unknown' - logger.mgpu_mm_log(f"[DIRECT_EJECTION][{model_name}] UNLOADING MODEL {i+1}/{len(mm.current_loaded_models)}: {model_name_to_eject}") - try: - lm.model_unload(unpatch_weights=True) - models_unloaded.append(model_name_to_eject) - logger.mgpu_mm_log(f"[DIRECT_EJECTION][{model_name}] SUCCESSFULLY UNLOADED: {model_name_to_eject}") - except Exception as e: - logger.mgpu_mm_log(f"[DIRECT_EJECTION][{model_name}] ERROR unloading {model_name_to_eject}: {e}") - - mm.current_loaded_models = [] - logger.mgpu_mm_log(f"[DIRECT_EJECTION][{model_name}] Models unloaded: {models_unloaded}") - logger.mgpu_mm_log(f"======= DIRECT MODEL EJECTION COMPLETE[{model_name}] =======") - multigpu_memory_log("eject_models_post", "complete") - - # RETURN MAX MEMORY - Should trigger auto-eviction by Comfy Core - total_device_memory = mm.get_total_memory(device) - max_gb = total_device_memory / (1024**3) - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] Returning MAX memory ({max_gb:.2f}GB) for auto-eviction by Comfy Core") - return total_device_memory - - elif is_distorch_model: - # eject_models=0, is_distorch=1: SUBTRACT VIRTUAL VRAM FROM ORIGINAL - virtual_vram_gb = getattr(getattr(self, 'model', None), 'model', None)._mgpu_virtual_vram_gb - virtual_vram_bytes = virtual_vram_gb * (1024**3) - adjusted_result = max(0, original_result - virtual_vram_bytes) - adjusted_gb = adjusted_result / (1024**3) if adjusted_result else 0 - - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] eject_models=0, is_distorch=1 → adjusted {original_gb:.2f}GB - {virtual_vram_gb:.2f}GB = {adjusted_gb:.2f}GB (DisTorch allocation)") - multigpu_memory_log("distorch_allocation", "reported") - return adjusted_result + for x in models: + loaded_model = mm.LoadedModel(x) + try: + loaded_model_index = current_loaded_models.index(loaded_model) + except: + loaded_model_index = None + if loaded_model_index is not None: + loaded = current_loaded_models[loaded_model_index] + loaded.currently_used = True + models_to_load.append(loaded) else: - # eject_models=0, is_distorch=0: RETURN ORIGINAL - logger.mgpu_mm_log(f"[MEM_REPORT][{model_name}] eject_models=0, is_distorch=0 → returning original {original_gb:.2f}GB") - multigpu_memory_log("keep_loaded_memory_check", "end") - return original_result + if hasattr(x, "model"): + logging.info(f"Requested to load {x.model.__class__.__name__}") + models_to_load.append(loaded_model) - mm.LoadedModel.model_memory_required = patched_loaded_model_memory_required + for loaded_model in models_to_load: + to_unload = [] + for i in range(len(current_loaded_models)): + if loaded_model.model.is_clone(current_loaded_models[i].model): + to_unload = [i] + to_unload + for i in to_unload: + model_to_unload = current_loaded_models.pop(i) + model_to_unload.model.detach(unpatch_all=False) + model_to_unload.model_finalizer.detach() + + # DisTorch Processing + total_memory_required = {} + eject_device = None + + for loaded_model in models_to_load: + device = loaded_model.device + base_memory = loaded_model.model_memory_required(device) + + # Check DisTorch flags + is_distorch = hasattr(loaded_model.model.model, '_mgpu_virtual_vram_gb') + has_eject = hasattr(loaded_model.model.model, '_mgpu_eject_models') + + if has_eject: + eject_device = device + logger.mgpu_mm_log("DisTorch eject_models=True, is_distorch=True - MAX memory eviction") + + if is_distorch: + # is_distorch=True: use compute device allocation size + virtual_vram_gb = loaded_model.model.model._mgpu_virtual_vram_gb + virtual_vram_bytes = virtual_vram_gb * (1024**3) + adjusted_memory = max(0, base_memory - virtual_vram_bytes) + total_memory_required[device] = total_memory_required.get(device, 0) + adjusted_memory + logger.mgpu_mm_log(f"DisTorch is_distorch=True, model adjusted {(base_memory - virtual_vram_bytes)/(1024**3):.2f}GB for device {device}") + else: + # is_distorch=False: use full model size + total_memory_required[device] = total_memory_required.get(device, 0) + base_memory + logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] Standard model {(base_memory)/(1024**3):.2f}GB for device {device}") + + for device in total_memory_required: + if device != torch.device("cpu"): + requested_mem = total_memory_required[device] * 1.1 + extra_mem + logger.mgpu_mm_log(f"[FREE_MEMORY_CALL] Device {device}: requesting {requested_mem/(1024**3):.2f}GB = {total_memory_required[device]/(1024**3):.2f}GB * 1.1 + {extra_mem/(1024**3):.2f}GB inference") + + + multigpu_memory_log("free_memory", "pre") + + for device in total_memory_required: + if device != torch.device("cpu"): + if device == eject_device: + total_device_memory = mm.get_total_memory(device) + logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=1, is_distorch=1 → using MAX memory ({total_device_memory/(1024**3):.2f}GB) for eviction") + free_memory(total_device_memory,device) + else: + logger.mgpu_mm_log(f"[LOAD_MODELS_GPU] eject_models=0, using Comfy Core Computed memory ({(total_memory_required[device] * 1.1 + extra_mem)/(1024**3):.2f}GB) for eviction") + free_memory(total_memory_required[device] * 1.1 + extra_mem, device) + + multigpu_memory_log("free_memory/minimum_memory_required", "post/pre") + + for device in total_memory_required: + if device != torch.device("cpu"): + free_mem = get_free_memory(device) + free_mem_gb = free_mem / (1024**3) + min_required_gb = minimum_memory_required / (1024**3) + logger.mgpu_mm_log(f"[MIN_MEMORY_CHECK] Device {device}: free={free_mem_gb:.2f}GB, required={min_required_gb:.2f}GB, will_evict={free_mem < minimum_memory_required}") + + if free_mem < minimum_memory_required: + models_l = free_memory(minimum_memory_required, device) + logger.mgpu_mm_log(f"[EVICTION] Device {device}: unloaded {len(models_l)} models due to insufficient memory") + logging.info("{} models unloaded.".format(len(models_l))) + + multigpu_memory_log("minimum_memory_required", "post") + + for loaded_model in models_to_load: + model = loaded_model.model + torch_dev = model.load_device + if is_device_cpu(torch_dev): + vram_set_state = VRAMState.DISABLED + else: + vram_set_state = vram_state + lowvram_model_memory = 0 + if lowvram_available and (vram_set_state == VRAMState.LOW_VRAM or vram_set_state == VRAMState.NORMAL_VRAM) and not force_full_load: + loaded_memory = loaded_model.model_loaded_memory() + current_free_mem = get_free_memory(torch_dev) + loaded_memory + + lowvram_model_memory = max(128 * 1024 * 1024, (current_free_mem - minimum_memory_required), min(current_free_mem * MIN_WEIGHT_MEMORY_RATIO, current_free_mem - minimum_inference_memory())) + lowvram_model_memory = max(0.1, lowvram_model_memory - loaded_memory) + + if vram_set_state == VRAMState.NO_VRAM: + lowvram_model_memory = 0.1 + + loaded_model.model_load(lowvram_model_memory, force_patch_weights=force_patch_weights) + current_loaded_models.insert(0, loaded_model) + + # Replace the module function + mm.load_models_gpu = patched_load_models_gpu original_partially_load = comfy.model_patcher.ModelPatcher.partially_load diff --git a/web/docs/CLIPLoaderDisTorch2MultiGPU.md b/web/docs/CLIPLoaderDisTorch2MultiGPU.md index 85ba923..ca0c197 100644 --- a/web/docs/CLIPLoaderDisTorch2MultiGPU.md +++ b/web/docs/CLIPLoaderDisTorch2MultiGPU.md @@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/CLIPLoaderGGUFDisTorch2MultiGPU.md b/web/docs/CLIPLoaderGGUFDisTorch2MultiGPU.md index 3fe43e8..fe383c4 100644 --- a/web/docs/CLIPLoaderGGUFDisTorch2MultiGPU.md +++ b/web/docs/CLIPLoaderGGUFDisTorch2MultiGPU.md @@ -14,7 +14,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/CLIPVisionLoaderDisTorch2MultiGPU.md b/web/docs/CLIPVisionLoaderDisTorch2MultiGPU.md index af24618..f01fbbe 100644 --- a/web/docs/CLIPVisionLoaderDisTorch2MultiGPU.md +++ b/web/docs/CLIPVisionLoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/clip_visio | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/CheckpointLoaderSimpleDisTorch2MultiGPU.md b/web/docs/CheckpointLoaderSimpleDisTorch2MultiGPU.md index dcbd5fa..2e65802 100644 --- a/web/docs/CheckpointLoaderSimpleDisTorch2MultiGPU.md +++ b/web/docs/CheckpointLoaderSimpleDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/checkpoint | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/ControlNetLoaderDisTorch2MultiGPU.md b/web/docs/ControlNetLoaderDisTorch2MultiGPU.md index c5a70fa..857d4e5 100644 --- a/web/docs/ControlNetLoaderDisTorch2MultiGPU.md +++ b/web/docs/ControlNetLoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/controlnet | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/DiffControlNetLoaderDisTorch2MultiGPU.md b/web/docs/DiffControlNetLoaderDisTorch2MultiGPU.md index 9f1f091..d41e37d 100644 --- a/web/docs/DiffControlNetLoaderDisTorch2MultiGPU.md +++ b/web/docs/DiffControlNetLoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node loads ControlNet models directly from HuggingFace model repositories b | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/DiffusersLoaderDisTorch2MultiGPU.md b/web/docs/DiffusersLoaderDisTorch2MultiGPU.md index 0654562..f666024 100644 --- a/web/docs/DiffusersLoaderDisTorch2MultiGPU.md +++ b/web/docs/DiffusersLoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node loads models directly from HuggingFace model repositories by specifyin | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/DualCLIPLoaderDisTorch2MultiGPU.md b/web/docs/DualCLIPLoaderDisTorch2MultiGPU.md index 38ff506..3e5053c 100644 --- a/web/docs/DualCLIPLoaderDisTorch2MultiGPU.md +++ b/web/docs/DualCLIPLoaderDisTorch2MultiGPU.md @@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/DualCLIPLoaderGGUFDisTorch2MultiGPU.md b/web/docs/DualCLIPLoaderGGUFDisTorch2MultiGPU.md index eb6387e..657af23 100644 --- a/web/docs/DualCLIPLoaderGGUFDisTorch2MultiGPU.md +++ b/web/docs/DualCLIPLoaderGGUFDisTorch2MultiGPU.md @@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/QuadrupleCLIPLoaderDisTorch2MultiGPU.md b/web/docs/QuadrupleCLIPLoaderDisTorch2MultiGPU.md index dd4ba62..623cc34 100644 --- a/web/docs/QuadrupleCLIPLoaderDisTorch2MultiGPU.md +++ b/web/docs/QuadrupleCLIPLoaderDisTorch2MultiGPU.md @@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU.md b/web/docs/QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU.md index 3623767..95af4e8 100644 --- a/web/docs/QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU.md +++ b/web/docs/QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU.md @@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/TripleCLIPLoaderDisTorch2MultiGPU.md b/web/docs/TripleCLIPLoaderDisTorch2MultiGPU.md index b714fb6..88aac1d 100644 --- a/web/docs/TripleCLIPLoaderDisTorch2MultiGPU.md +++ b/web/docs/TripleCLIPLoaderDisTorch2MultiGPU.md @@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` fold | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/TripleCLIPLoaderGGUFDisTorch2MultiGPU.md b/web/docs/TripleCLIPLoaderGGUFDisTorch2MultiGPU.md index e80a2fa..9c11735 100644 --- a/web/docs/TripleCLIPLoaderGGUFDisTorch2MultiGPU.md +++ b/web/docs/TripleCLIPLoaderGGUFDisTorch2MultiGPU.md @@ -15,7 +15,7 @@ This node automatically detects models located in the `ComfyUI/models/clip` and | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: false for CLIP loaders). | ## Outputs diff --git a/web/docs/UNETLoaderDisTorch2MultiGPU.md b/web/docs/UNETLoaderDisTorch2MultiGPU.md index bf79e54..dd3246f 100644 --- a/web/docs/UNETLoaderDisTorch2MultiGPU.md +++ b/web/docs/UNETLoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet` fold | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/UnetLoaderGGUFAdvancedDisTorch2MultiGPU.md b/web/docs/UnetLoaderGGUFAdvancedDisTorch2MultiGPU.md index 601f585..1c1785d 100644 --- a/web/docs/UnetLoaderGGUFAdvancedDisTorch2MultiGPU.md +++ b/web/docs/UnetLoaderGGUFAdvancedDisTorch2MultiGPU.md @@ -16,7 +16,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf` | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/UnetLoaderGGUFDisTorch2MultiGPU.md b/web/docs/UnetLoaderGGUFDisTorch2MultiGPU.md index 8d37e3f..de1f3e0 100644 --- a/web/docs/UnetLoaderGGUFDisTorch2MultiGPU.md +++ b/web/docs/UnetLoaderGGUFDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/unet_gguf` | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/web/docs/VAELoaderDisTorch2MultiGPU.md b/web/docs/VAELoaderDisTorch2MultiGPU.md index fb35916..9df192d 100644 --- a/web/docs/VAELoaderDisTorch2MultiGPU.md +++ b/web/docs/VAELoaderDisTorch2MultiGPU.md @@ -13,7 +13,7 @@ This node automatically detects models located in the `ComfyUI/models/vae` folde | `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). | | `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). | | `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). | -| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). | +| `eject_models` | `BOOLEAN` | Whether to unload ALL models from the target device before loading this model, enabling deterministic model eviction for testing and memory management (default: true). | ## Outputs diff --git a/wrappers.py b/wrappers.py index d8d70ed..9abf9bf 100644 --- a/wrappers.py +++ b/wrappers.py @@ -15,7 +15,7 @@ logger = logging.getLogger("MultiGPU") # DISTORCH V2 SAFETENSOR WRAPPERS (DisTorch2 for .safetensors and .gguf) # ============================================================================ -def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround): +def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround, eject_models_default=True): """Internal factory function creating DisTorch2 override class with parameterized device selection behavior.""" from .distorch_2 import ( register_patched_safetensor_modelpatcher, @@ -37,7 +37,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1}) inputs["optional"]["donor_device"] = (devices, {"default": "cpu"}) inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""}) - inputs["optional"]["eject_models"] = ("BOOLEAN", {"default": True}) + inputs["optional"]["eject_models"] = ("BOOLEAN", {"default": eject_models_default}) return inputs CATEGORY = "multigpu/distorch_2" @@ -46,7 +46,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette @classmethod def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu", - expert_mode_allocations="", eject_models=True, **kwargs): + expert_mode_allocations="", eject_models=eject_models_default, **kwargs): device_value = kwargs.get(device_param_name) settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{eject_models}" current_hash = hashlib.sha256(settings_str.encode()).hexdigest() @@ -60,7 +60,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette return current_hash def override(self, *args, virtual_vram_gb=4.0, donor_device="cpu", - expert_mode_allocations="", eject_models=True, **kwargs): + expert_mode_allocations="", eject_models=eject_models_default, **kwargs): device_value = kwargs.get(device_param_name) @@ -89,7 +89,7 @@ def _create_distorch_safetensor_v2_override(cls, device_param_name, device_sette if device_value is not None: device_setter_func(device_value) - # Strip MultiGPU-specific parameters before calling original function (REMOVE eject_models, keep_loaded and virtual_vram_gb since we handle them above) + # Strip MultiGPU-specific parameters before calling original function (REMOVE eject_models, eject_models and virtual_vram_gb since we handle them above) clean_kwargs = {k: v for k, v in kwargs.items() if k not in [device_param_name, 'virtual_vram_gb', 'donor_device', 'expert_mode_allocations', @@ -187,7 +187,8 @@ def override_class_with_distorch_safetensor_v2_clip(cls): cls, device_param_name="device", device_setter_func=set_current_text_encoder_device, - apply_device_kwarg_workaround=True + apply_device_kwarg_workaround=True, + eject_models_default=False # CLIP defaults to False ) @@ -198,7 +199,8 @@ def override_class_with_distorch_safetensor_v2_clip_no_device(cls): cls, device_param_name="device", device_setter_func=set_current_text_encoder_device, - apply_device_kwarg_workaround=False + apply_device_kwarg_workaround=False, + eject_models_default=False # CLIP defaults to False )