Merge pull request #114 from pollockjj/d2_clip

Major Refactor
This commit is contained in:
John Pollock
2025-09-30 20:38:39 -05:00
committed by GitHub
49 changed files with 2657 additions and 2563 deletions
+3 -1
View File
@@ -1,3 +1,5 @@
# Python and IDE
__pycache__/
.vscode/settings.json
.clinerules
.vscode
memory-bank/
+9
View File
@@ -112,6 +112,15 @@ Currently supported nodes (automatically detected if available):
All MultiGPU nodes available for your install can be found in the "multigpu" category in the node menu.
## Node Documentation
Detailed technical documentation is available for all **automatically-detected core MultiGPU and DisTorch2 nodes**, covering 36+ documented nodes with comprehensive parameter details, output specifications, and DisTorch2 allocation guidance where applicable.
- **To access documentation**: Click on any core MultiGPU or DisTorch2 node in ComfyUI and select "Help" (question mark inside a circle) from the resultant menu
- **Coverage**: All standard ComfyUI loader nodes (UNet, VAE, Checkpoints, CLIP, ControlNet, Diffusers) plus popular GGUF loader variants
- **Contents**: Input parameters with data types and descriptions, output specifications, usage examples, and DisTorch2 distributed loading explanations with allocation modes and strategies
- **Note**: Documentation covers core ComfyUI-MultiGPU functionality only. Third-party custom node integrations (WanVideoWrapper, Florence2, etc.) have their own separate documentation.
## Example workflows
All workflows have been tested on a 2x 3090 + 1060ti linux setup, a 4070 win 11 setup, and a 3090/1070ti linux setup.
+59 -145
View File
@@ -1,121 +1,73 @@
import torch
import logging
import weakref
import os
import copy
from pathlib import Path
import folder_paths
import comfy.model_management as mm
import comfy.model_patcher
from nodes import NODE_CLASS_MAPPINGS as GLOBAL_NODE_CLASS_MAPPINGS
from .device_utils import get_device_list, is_accelerator_available
from .device_utils import (
get_device_list,
is_accelerator_available,
soft_empty_cache_multigpu,
)
from .model_management_mgpu import (
trigger_executor_cache_reset,
check_cpu_memory_threshold,
multigpu_memory_log,
force_full_system_cleanup,
)
# --- DisTorch V2 Logging Configuration ---
# Set to "E" for Engineering (DEBUG) or "P" for Production (INFO)
LOG_LEVEL = "P"
WEB_DIRECTORY = "./web"
MGPU_MM_LOG = False
DEBUG_LOG = False
# Configure logger
logger = logging.getLogger("MultiGPU")
logger.propagate = False
if not logger.handlers:
log_level = logging.DEBUG if LOG_LEVEL == "E" else logging.INFO
log_level = logging.DEBUG if DEBUG_LOG else logging.INFO
handler = logging.StreamHandler()
formatter = logging.Formatter('%(message)s')
handler.setFormatter(formatter)
logger.addHandler(handler)
logger.setLevel(log_level)
logger.info(f"[MultiGPU Initialization] Logger initialized with level: {logging.getLevelName(log_level)}")
def mgpu_mm_log_method(self, msg):
"""Add MultiGPU model management logging method to logger instance."""
if MGPU_MM_LOG:
self.info(f"[MultiGPU Model Management] {msg}")
logger.mgpu_mm_log = mgpu_mm_log_method.__get__(logger, type(logger))
def check_module_exists(module_path):
"""Check if a custom node module exists in ComfyUI custom_nodes directory."""
full_path = os.path.join(folder_paths.get_folder_paths("custom_nodes")[0], module_path)
logger.debug(f"[MultiGPU] Checking for module at {full_path}")
if not os.path.exists(full_path):
logger.debug(f"[MultiGPU] Module {module_path} not found - skipping")
return False
logger.debug(f"[MultiGPU] Found {module_path}, creating compatible MultiGPU nodes")
return True
# Global device state management
current_device = mm.get_torch_device()
current_text_encoder_device = mm.text_encoder_device()
def set_current_device(device):
"""Set the current device context for MultiGPU operations."""
global current_device
current_device = device
logger.info(f"[MultiGPU Initialization] current_device set to: {device}")
logger.debug(f"[MultiGPU Initialization] current_device set to: {device}")
def set_current_text_encoder_device(device):
"""Set the current text encoder device context for CLIP models."""
global current_text_encoder_device
current_text_encoder_device = device
logger.info(f"[MultiGPU Initialization] current_text_encoder_device set to: {device}")
def override_class(cls):
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_device(device)
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride
def override_class_clip(cls):
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
kwargs['device'] = 'default'
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride
def override_class_clip_no_device(cls):
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride
logger.debug(f"[MultiGPU Initialization] current_text_encoder_device set to: {device}")
def get_torch_device_patched():
"""Return MultiGPU-aware device selection for patched mm.get_torch_device."""
device = None
if (not is_accelerator_available() or mm.cpu_state == mm.CPUState.CPU or "cpu" in str(current_device).lower()):
device = torch.device("cpu")
@@ -126,6 +78,7 @@ def get_torch_device_patched():
return device
def text_encoder_device_patched():
"""Return MultiGPU-aware text encoder device for patched mm.text_encoder_device."""
device = None
if (not is_accelerator_available() or mm.cpu_state == mm.CPUState.CPU or "cpu" in str(current_text_encoder_device).lower()):
device = torch.device("cpu")
@@ -135,23 +88,12 @@ def text_encoder_device_patched():
logger.debug(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
return device
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device, mm.text_encoder_device, and mm.text_encoder_initial_device")
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device and mm.text_encoder_device")
logger.debug(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
logger.debug(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
mm.get_torch_device = get_torch_device_patched
mm.text_encoder_device = text_encoder_device_patched
def check_module_exists(module_path):
full_path = os.path.join(folder_paths.get_folder_paths("custom_nodes")[0], module_path)
logger.debug(f"[MultiGPU] Checking for module at {full_path}")
if not os.path.exists(full_path):
logger.debug(f"[MultiGPU] Module {module_path} not found - skipping")
return False
logger.debug(f"[MultiGPU] Found {module_path}, creating compatible MultiGPU nodes")
return True
# Import from nodes.py
from .nodes import (
DeviceSelectorMultiGPU,
HunyuanVideoEmbeddingsAdapter,
@@ -175,9 +117,9 @@ from .nodes import (
HyVideoModelLoader,
HyVideoVAELoader,
DownloadAndLoadHyVideoTextEncoder,
UNetLoaderLP,
)
# Import from wanvideo.py
from .wanvideo import (
WanVideoModelLoader,
WanVideoModelLoader_2,
@@ -189,81 +131,63 @@ from .wanvideo import (
WanVideoSampler
)
# Import from distorch.py
from .distorch import (
model_allocation_store,
create_model_hash,
register_patched_ggufmodelpatcher,
analyze_ggml_loading,
calculate_vvram_allocation_string,
from .wrappers import (
override_class,
override_class_clip,
override_class_clip_no_device,
override_class_with_distorch_gguf,
override_class_with_distorch_gguf_v2,
override_class_with_distorch_clip,
override_class_with_distorch_clip_no_device,
override_class_with_distorch
override_class_with_distorch,
override_class_with_distorch_safetensor_v2,
override_class_with_distorch_safetensor_v2_clip,
override_class_with_distorch_safetensor_v2_clip_no_device,
)
# Import from distorch_2.py for DisTorch v2 SafeTensor support
from .distorch_2 import (
safetensor_allocation_store,
create_safetensor_model_hash,
register_patched_safetensor_modelpatcher,
analyze_safetensor_loading,
calculate_safetensor_vvram_allocation,
override_class_with_distorch_safetensor_v2,
override_class_with_distorch_safetensor_v2_clip,
override_class_with_distorch_safetensor_v2_clip_no_device
)
# Import advanced checkpoint loaders
from .checkpoint_multigpu import (
CheckpointLoaderAdvancedMultiGPU,
CheckpointLoaderAdvancedDisTorch2MultiGPU
)
# Initialize NODE_CLASS_MAPPINGS
NODE_CLASS_MAPPINGS = {
"DeviceSelectorMultiGPU": DeviceSelectorMultiGPU,
"HunyuanVideoEmbeddingsAdapter": HunyuanVideoEmbeddingsAdapter,
"CheckpointLoaderAdvancedMultiGPU": CheckpointLoaderAdvancedMultiGPU,
"CheckpointLoaderAdvancedDisTorch2MultiGPU": CheckpointLoaderAdvancedDisTorch2MultiGPU,
"UNetLoaderLP": UNetLoaderLP,
}
# Standard MultiGPU nodes
NODE_CLASS_MAPPINGS["UNETLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["UNETLoader"])
NODE_CLASS_MAPPINGS["VAELoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["VAELoader"])
NODE_CLASS_MAPPINGS["CLIPLoaderMultiGPU"] = override_class_clip(GLOBAL_NODE_CLASS_MAPPINGS["CLIPLoader"])
NODE_CLASS_MAPPINGS["DualCLIPLoaderMultiGPU"] = override_class_clip(GLOBAL_NODE_CLASS_MAPPINGS["DualCLIPLoader"])
if "TripleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["TripleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
if "QuadrupleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
NODE_CLASS_MAPPINGS["TripleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
NODE_CLASS_MAPPINGS["CLIPVisionLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["CLIPVisionLoader"])
NODE_CLASS_MAPPINGS["CheckpointLoaderSimpleMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["CheckpointLoaderSimple"])
NODE_CLASS_MAPPINGS["ControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["ControlNetLoader"])
if "DiffusersLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["DiffusersLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
if "DiffControlNetLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["DiffControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
# DisTorch 2 SafeTensor nodes for FLUX and other safetensor models
NODE_CLASS_MAPPINGS["DiffusersLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
NODE_CLASS_MAPPINGS["DiffControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
NODE_CLASS_MAPPINGS["UNETLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["UNETLoader"])
NODE_CLASS_MAPPINGS["VAELoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["VAELoader"])
NODE_CLASS_MAPPINGS["CLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip(GLOBAL_NODE_CLASS_MAPPINGS["CLIPLoader"])
NODE_CLASS_MAPPINGS["DualCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip(GLOBAL_NODE_CLASS_MAPPINGS["DualCLIPLoader"])
if "TripleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["TripleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
if "QuadrupleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
NODE_CLASS_MAPPINGS["TripleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
NODE_CLASS_MAPPINGS["CLIPVisionLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["CLIPVisionLoader"])
NODE_CLASS_MAPPINGS["CheckpointLoaderSimpleDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["CheckpointLoaderSimple"])
NODE_CLASS_MAPPINGS["ControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["ControlNetLoader"])
if "DiffusersLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["DiffusersLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
if "DiffControlNetLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
NODE_CLASS_MAPPINGS["DiffControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
NODE_CLASS_MAPPINGS["DiffusersLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
NODE_CLASS_MAPPINGS["DiffControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
# --- Registration Table ---
logger.info("[MultiGPU] Initiating custom_node Registration. . .")
dash_line = "-" * 47
fmt_reg = "{:<30}{:>5}{:>10}"
@@ -274,6 +198,7 @@ logger.info(dash_line)
registration_data = []
def register_and_count(module_names, node_map):
"""Register MultiGPU node wrappers for detected custom node modules."""
found = False
for name in module_names:
if check_module_exists(name):
@@ -290,26 +215,21 @@ def register_and_count(module_names, node_map):
registration_data.append({"name": module_names[0], "found": "Y" if found else "N", "count": count})
return found
# ComfyUI-LTXVideo
ltx_nodes = {"LTXVLoaderMultiGPU": override_class(LTXVLoader)}
register_and_count(["ComfyUI-LTXVideo", "comfyui-ltxvideo"], ltx_nodes)
# ComfyUI-Florence2
florence_nodes = {
"Florence2ModelLoaderMultiGPU": override_class(Florence2ModelLoader),
"DownloadAndLoadFlorence2ModelMultiGPU": override_class(DownloadAndLoadFlorence2Model)
}
register_and_count(["ComfyUI-Florence2", "comfyui-florence2"], florence_nodes)
# ComfyUI_bitsandbytes_NF4
nf4_nodes = {"CheckpointLoaderNF4MultiGPU": override_class(CheckpointLoaderNF4)}
register_and_count(["ComfyUI_bitsandbytes_NF4", "comfyui_bitsandbytes_nf4"], nf4_nodes)
# x-flux-comfyui
flux_controlnet_nodes = {"LoadFluxControlNetMultiGPU": override_class(LoadFluxControlNet)}
register_and_count(["x-flux-comfyui"], flux_controlnet_nodes)
# ComfyUI-MMAudio
mmaudio_nodes = {
"MMAudioModelLoaderMultiGPU": override_class(MMAudioModelLoader),
"MMAudioFeatureUtilsLoaderMultiGPU": override_class(MMAudioFeatureUtilsLoader),
@@ -317,7 +237,6 @@ mmaudio_nodes = {
}
register_and_count(["ComfyUI-MMAudio", "comfyui-mmaudio"], mmaudio_nodes)
# ComfyUI-GGUF
gguf_nodes = {
"UnetLoaderGGUFDisTorchMultiGPU": override_class_with_distorch_gguf(UnetLoaderGGUF),
"UnetLoaderGGUFAdvancedDisTorchMultiGPU": override_class_with_distorch_gguf(UnetLoaderGGUFAdvanced),
@@ -340,7 +259,6 @@ gguf_nodes = {
}
register_and_count(["ComfyUI-GGUF", "comfyui-gguf"], gguf_nodes)
# PuLID_ComfyUI
pulid_nodes = {
"PulidModelLoaderMultiGPU": override_class(PulidModelLoader),
"PulidInsightFaceLoaderMultiGPU": override_class(PulidInsightFaceLoader),
@@ -348,7 +266,6 @@ pulid_nodes = {
}
register_and_count(["PuLID_ComfyUI", "pulid_comfyui"], pulid_nodes)
# ComfyUI-HunyuanVideoWrapper
hunyuan_nodes = {
"HyVideoModelLoaderMultiGPU": override_class(HyVideoModelLoader),
"HyVideoVAELoaderMultiGPU": override_class(HyVideoVAELoader),
@@ -356,7 +273,6 @@ hunyuan_nodes = {
}
register_and_count(["ComfyUI-HunyuanVideoWrapper", "comfyui-hunyuanvideowrapper"], hunyuan_nodes)
# ComfyUI-WanVideoWrapper
wanvideo_nodes = {
"WanVideoModelLoaderMultiGPU": WanVideoModelLoader,
"WanVideoModelLoaderMultiGPU_2": WanVideoModelLoader_2,
@@ -369,10 +285,8 @@ wanvideo_nodes = {
}
register_and_count(["ComfyUI-WanVideoWrapper", "comfyui-wanvideowrapper"], wanvideo_nodes)
# Print the registration table
for item in registration_data:
logger.info(fmt_reg.format(item['name'], item['found'], str(item['count'])))
logger.info(dash_line)
logger.info(f"[MultiGPU] Registration complete. Final mappings: {', '.join(NODE_CLASS_MAPPINGS.keys())}")
logger.info(f"[MultiGPU] Registration complete. Final mappings: {', '.join(NODE_CLASS_MAPPINGS.keys())}")
+22 -24
View File
@@ -1,8 +1,3 @@
"""
Advanced Checkpoint Loaders for MultiGPU
Provides device-specific and DisTorch2 sharding for checkpoint components
"""
import torch
import logging
import hashlib
@@ -13,6 +8,7 @@ import comfy.model_detection
import comfy.clip_vision
from comfy.sd import VAE, CLIP
from .device_utils import get_device_list, soft_empty_cache_multigpu
from .model_management_mgpu import multigpu_memory_log
from .distorch_2 import safetensor_allocation_store, safetensor_settings_store, create_safetensor_model_hash, register_patched_safetensor_modelpatcher
logger = logging.getLogger("MultiGPU")
@@ -23,24 +19,21 @@ checkpoint_distorch_config = {}
original_load_state_dict_guess_config = None
def patch_load_state_dict_guess_config():
"""
Monkey patch the load_state_dict_guess_config function to replace its logic
with a MultiGPU-aware implementation.
"""
"""Monkey patch comfy.sd.load_state_dict_guess_config with MultiGPU-aware checkpoint loading."""
global original_load_state_dict_guess_config
if original_load_state_dict_guess_config is not None:
logger.info("[MultiGPU] load_state_dict_guess_config is already patched.")
logger.debug("[MultiGPU Checkpoint] load_state_dict_guess_config is already patched.")
return
logger.info("[MultiGPU] Patching comfy.sd.load_state_dict_guess_config for advanced MultiGPU loading.")
logger.info("[MultiGPU Core Patching] Patching comfy.sd.load_state_dict_guess_config for advanced MultiGPU loading.")
original_load_state_dict_guess_config = comfy.sd.load_state_dict_guess_config
comfy.sd.load_state_dict_guess_config = patched_load_state_dict_guess_config
def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True, output_clipvision=False,
embedding_directory=None, output_model=True, model_options={},
te_model_options={}, metadata=None):
"""Patched checkpoint loader with MultiGPU and DisTorch2 device placement support."""
from . import set_current_device, set_current_text_encoder_device, current_device, current_text_encoder_device
sd_size = sum(p.numel() for p in sd.values() if hasattr(p, 'numel'))
@@ -51,9 +44,9 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
if not device_config and not distorch_config:
return original_load_state_dict_guess_config(sd, output_vae, output_clip, output_clipvision, embedding_directory, output_model, model_options, te_model_options, metadata)
logger.info("--- [MultiGPU] ENTERING Patched Checkpoint Loader ---")
logger.info(f"Received Device Config: {device_config}")
logger.info(f"Received DisTorch2 Config: {distorch_config}")
logger.debug("[MultiGPU Checkpoint] ENTERING Patched Checkpoint Loader")
logger.debug(f"[MultiGPU Checkpoint] Received Device Config: {device_config}")
logger.debug(f"[MultiGPU Checkpoint] Received DisTorch2 Config: {distorch_config}")
clip = None
clipvision = None
@@ -63,7 +56,6 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
original_main_device = current_device
original_clip_device = current_text_encoder_device
logger.info(f"Saved original device contexts: UNet/VAE='{original_main_device}', CLIP='{original_clip_device}'")
try:
diffusion_model_prefix = comfy.model_detection.unet_prefix_from_state_dict(sd)
@@ -80,7 +72,7 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
return None
return (diffusion_model, None, VAE(sd={}), None)
logger.info(f"[MultiGPU] Detected Model Config: {type(model_config).__name__}, Parameters: {parameters/10**9:.2f}B")
logger.debug(f"[MultiGPU] Detected Model Config: {type(model_config).__name__}, Parameters: {parameters/10**9:.2f}B")
unet_weight_dtype = list(model_config.supported_inference_dtypes)
if model_config.scaled_fp8 is not None:
@@ -105,10 +97,14 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
set_current_device(unet_compute_device)
inital_load_device = mm.unet_inital_load_device(parameters, unet_dtype)
multigpu_memory_log(f"unet:{config_hash[:8]}", "pre-load")
model = model_config.get_model(sd, diffusion_model_prefix, device=inital_load_device)
soft_empty_cache_multigpu(logger)
logger.mgpu_mm_log("Invoking soft_empty_cache_multigpu before UNet ModelPatcher setup")
soft_empty_cache_multigpu()
model_patcher = comfy.model_patcher.ModelPatcher(model, load_device=unet_compute_device, offload_device=mm.unet_offload_device())
multigpu_memory_log(f"unet:{config_hash[:8]}", "post-model")
if distorch_config and 'unet_allocation' in distorch_config:
register_patched_safetensor_modelpatcher()
@@ -117,17 +113,20 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
safetensor_settings_store[model_hash] = distorch_config.get('unet_settings','')
model.is_distorch = True
model._distorch_high_precision_loras = distorch_config.get('high_precision_loras', True)
logger.info(f"Stored DisTorch2 config for UNet (hash {model_hash[:8]}): {distorch_config['unet_allocation']}")
logger.mgpu_mm_log(f"Stored DisTorch2 config for UNet (hash {model_hash[:8]}): {distorch_config['unet_allocation']}")
model.load_model_weights(sd, diffusion_model_prefix)
multigpu_memory_log(f"unet:{config_hash[:8]}", "post-weights")
if output_vae:
vae_target_device = torch.device(device_config.get('vae_device', original_main_device))
set_current_device(vae_target_device) # Use main device context for VAE
multigpu_memory_log(f"vae:{config_hash[:8]}", "pre-load")
vae_sd = comfy.utils.state_dict_prefix_replace(sd, {k: "" for k in model_config.vae_key_prefix}, filter_keys=True)
vae_sd = model_config.process_vae_state_dict(vae_sd)
vae = VAE(sd=vae_sd, metadata=metadata)
multigpu_memory_log(f"vae:{config_hash[:8]}", "post-load")
if output_clip:
clip_target_device = device_config.get('clip_device', original_clip_device)
@@ -137,7 +136,9 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
if clip_target is not None:
clip_sd = model_config.process_clip_state_dict(sd)
if len(clip_sd) > 0:
soft_empty_cache_multigpu(logger)
logger.debug("[MultiGPU Checkpoint] Invoking soft_empty_cache_multigpu before CLIP construction")
multigpu_memory_log(f"clip:{config_hash[:8]}", "pre-load")
soft_empty_cache_multigpu()
clip_params = comfy.utils.calculate_parameters(clip_sd)
clip = CLIP(clip_target, embedding_directory=embedding_directory, tokenizer_data=clip_sd, parameters=clip_params, model_options=te_model_options)
@@ -155,22 +156,19 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
if len(m) > 0: logger.warning(f"CLIP missing keys: {m}")
if len(u) > 0: logger.debug(f"CLIP unexpected keys: {u}")
logger.info("CLIP Loaded.")
multigpu_memory_log(f"clip:{config_hash[:8]}", "post-load")
else:
logger.warning("No CLIP/text encoder weights in checkpoint.")
else:
logger.warning("CLIP target not found in model config.")
finally:
# --- Restore original device contexts and clean up ---
set_current_device(original_main_device)
set_current_text_encoder_device(original_clip_device)
if config_hash in checkpoint_device_config:
del checkpoint_device_config[config_hash]
if config_hash in checkpoint_distorch_config:
del checkpoint_distorch_config[config_hash]
logger.info(f"Restored original device contexts. UNet/VAE='{original_main_device}', CLIP='{original_clip_device}'")
logger.info("--- [MultiGPU] EXITING Patched Checkpoint Loader ---")
return (model_patcher, clip, vae, clipvision)
class CheckpointLoaderAdvancedMultiGPU:
+175 -400
View File
@@ -1,18 +1,12 @@
"""
Device detection, management, and inspection utilities for ComfyUI-MultiGPU.
Single source of truth for all device enumeration, compatibility checks, and state inspection.
Handles all device types supported by ComfyUI core.
"""
import torch
import logging
import hashlib
import psutil
import comfy.model_management as mm
import gc
logger = logging.getLogger("MultiGPU")
# Module-level cache for device list (populated once on first call)
_DEVICE_LIST_CACHE = None
def get_device_list():
@@ -33,81 +27,61 @@ def get_device_list():
"""
global _DEVICE_LIST_CACHE
# Return cached result if already populated
if _DEVICE_LIST_CACHE is not None:
return _DEVICE_LIST_CACHE
# First time - do the actual detection
devs = []
# CPU is always physically present and can store tensors
devs.append("cpu")
# CUDA devices (NVIDIA GPUs)
try:
if hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available():
device_count = torch.cuda.device_count()
devs += [f"cuda:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} CUDA device(s)")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] CUDA detection failed: {e}")
if hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available():
device_count = torch.cuda.device_count()
devs += [f"cuda:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} CUDA device(s)")
# XPU devices (Intel GPUs)
try:
# Try to import intel extension first (may be required for XPU support)
import intel_extension_for_pytorch as ipex
except ImportError:
pass
try:
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
device_count = torch.xpu.device_count()
devs += [f"xpu:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} XPU device(s)")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] XPU detection failed: {e}")
# NPU devices (Ascend NPUs from Huawei)
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
device_count = torch.xpu.device_count()
devs += [f"xpu:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} XPU device(s)")
try:
import torch_npu
if hasattr(torch, "npu") and hasattr(torch.npu, "is_available") and torch.npu.is_available():
device_count = torch.npu.device_count()
devs += [f"npu:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} NPU device(s)")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] NPU detection failed: {e}")
except ImportError:
pass
# MLU devices (Cambricon MLUs)
try:
import torch_mlu
if hasattr(torch, "mlu") and hasattr(torch.mlu, "is_available") and torch.mlu.is_available():
device_count = torch.mlu.device_count()
devs += [f"mlu:{i}" for i in range(device_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} MLU device(s)")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] MLU detection failed: {e}")
except ImportError:
pass
# MPS device (Apple Metal - single device only)
try:
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
devs.append("mps")
logger.debug("[MultiGPU_Device_Utils] Found MPS device")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] MPS detection failed: {e}")
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
devs.append("mps")
logger.debug("[MultiGPU_Device_Utils] Found MPS device")
# DirectML devices (Windows DirectML for AMD/Intel/NVIDIA)
try:
import torch_directml
adapter_count = torch_directml.device_count()
if adapter_count > 0:
devs += [f"directml:{i}" for i in range(adapter_count)]
logger.debug(f"[MultiGPU_Device_Utils] Found {adapter_count} DirectML adapter(s)")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] DirectML detection failed: {e}")
except ImportError:
pass
# IXUCA/CoreX devices (special accelerator)
try:
if hasattr(torch, "corex"):
# CoreX typically exposes single device, but check if there's a count method
if hasattr(torch.corex, "device_count"):
device_count = torch.corex.device_count()
devs += [f"corex:{i}" for i in range(device_count)]
@@ -115,431 +89,232 @@ def get_device_list():
else:
devs.append("corex:0")
logger.debug("[MultiGPU_Device_Utils] Found CoreX device")
except Exception as e:
logger.debug(f"[MultiGPU_Device_Utils] CoreX detection failed: {e}")
except ImportError:
pass
# Cache the result for future calls
_DEVICE_LIST_CACHE = devs
# Log only once when initially populated
logger.info(f"[MultiGPU_Device_Utils] Device list initialized: {devs}")
logger.debug(f"[MultiGPU_Device_Utils] Device list initialized: {devs}")
return devs
def is_accelerator_available():
"""
Check if any accelerator device is available.
Used by patched functions to determine CPU fallback.
"""Check if any GPU or accelerator device is available including CUDA, XPU, NPU, MLU, MPS, DirectML, or CoreX."""
if hasattr(torch, "cuda") and torch.cuda.is_available():
return True
Returns True if any GPU/accelerator is available, False otherwise.
"""
# Check CUDA
try:
if torch.cuda.is_available():
return True
except:
pass
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
return True
# Check XPU (Intel GPU)
try:
if hasattr(torch, "xpu") and torch.xpu.is_available():
return True
except:
pass
# Check NPU (Ascend)
try:
import torch_npu
if hasattr(torch, "npu") and torch.npu.is_available():
if hasattr(torch, "npu") and hasattr(torch.npu, "is_available") and torch.npu.is_available():
return True
except:
except ImportError:
pass
# Check MLU (Cambricon)
try:
import torch_mlu
if hasattr(torch, "mlu") and torch.mlu.is_available():
if hasattr(torch, "mlu") and hasattr(torch.mlu, "is_available") and torch.mlu.is_available():
return True
except:
except ImportError:
pass
# Check MPS (Apple Metal)
try:
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
return True
except:
pass
# Check DirectML
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
return True
try:
import torch_directml
if torch_directml.device_count() > 0:
return True
except:
pass
# Check CoreX/IXUCA
try:
if hasattr(torch, "corex"):
return True
except:
except ImportError:
pass
if hasattr(torch, "corex"):
return True
return False
def is_device_compatible(device_string):
"""
Check if a device string represents a valid, available device.
Args:
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
Returns:
True if the device is available, False otherwise.
"""
"""Check if a device string represents a valid available device."""
available_devices = get_device_list()
return device_string in available_devices
def get_device_type(device_string):
"""
Extract the device type from a device string.
Args:
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
Returns:
Device type string (e.g., "cuda", "cpu", "xpu", "npu", "mlu", "mps", "directml", "corex")
"""
"""Extract device type from device string (e.g. 'cuda' from 'cuda:0')."""
if ":" in device_string:
return device_string.split(":")[0]
return device_string
def parse_device_string(device_string):
"""
Parse a device string into type and index.
Args:
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
Returns:
Tuple of (device_type, device_index) where index is None for non-indexed devices
"""
"""Parse device string into (device_type, device_index) tuple."""
if ":" in device_string:
parts = device_string.split(":")
return parts[0], int(parts[1])
return device_string, None
def soft_empty_cache_multigpu():
"""Clear allocator caches across all devices using context managers to preserve calling thread device context."""
from .model_management_mgpu import multigpu_memory_log
logger.mgpu_mm_log("soft_empty_cache_multigpu: starting GC and multi-device cache clear")
def soft_empty_cache_multigpu(logger):
"""
Replicate ComfyUI's cache clearing but for ALL devices in MultiGPU.
MultiGPU adaptation of ComfyUI's soft_empty_cache() functionality.
"""
import gc
logger.info("[MultiGPU_Device_Utils] Preparing devices for optimized safetensor loading")
# Python GC (same as all implementations)
gc.collect()
logger.debug("[MultiGPU_Device_Utils] Performed garbage collection before safetensor loading")
# Clear cache for ALL devices (not just ComfyUI's single device)
all_devices = get_device_list()
logger.mgpu_mm_log(f"soft_empty_cache_multigpu: devices to clear = {all_devices}")
# Check global availability first to avoid unnecessary iteration if backend is missing
is_cuda_available = hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available()
for device_str in all_devices:
if device_str.startswith("cuda:"):
device_idx = int(device_str.split(":")[1])
torch.cuda.set_device(device_idx)
torch.cuda.empty_cache()
torch.cuda.ipc_collect() # ComfyUI's CUDA optimization
logger.debug(f"[MultiGPU_Device_Utils] Cleared cache + IPC for {device_str}")
if is_cuda_available:
device_idx = int(device_str.split(":")[1])
logger.mgpu_mm_log(f"Clearing CUDA cache on {device_str} (idx={device_idx})")
multigpu_memory_log("general", f"pre-empty:{device_str}")
with torch.cuda.device(device_idx):
torch.cuda.empty_cache()
if hasattr(torch.cuda, "ipc_collect"):
torch.cuda.ipc_collect()
logger.mgpu_mm_log(f"Cleared CUDA cache (and IPC if available) on {device_str}")
multigpu_memory_log("general", f"post-empty:{device_str}")
elif device_str == "mps":
torch.mps.empty_cache()
logger.debug("[MultiGPU_Device_Utils] Cleared cache for MPS")
if hasattr(torch, "mps") and hasattr(torch.mps, "empty_cache"):
logger.mgpu_mm_log("Clearing MPS cache")
multigpu_memory_log("general", f"pre-empty:{device_str}")
torch.mps.empty_cache()
logger.mgpu_mm_log("Cleared MPS cache")
multigpu_memory_log("general", f"post-empty:{device_str}")
elif device_str.startswith("xpu:"):
torch.xpu.empty_cache()
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Intel XPU")
if hasattr(torch, "xpu") and hasattr(torch.xpu, "empty_cache"):
logger.mgpu_mm_log(f"Clearing XPU cache on {device_str}")
multigpu_memory_log("general", f"pre-empty:{device_str}")
torch.xpu.empty_cache()
logger.mgpu_mm_log(f"Cleared XPU cache on {device_str}")
multigpu_memory_log("general", f"post-empty:{device_str}")
elif device_str.startswith("npu:"):
torch.npu.empty_cache()
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Ascend NPU")
if hasattr(torch, "npu") and hasattr(torch.npu, "empty_cache"):
logger.mgpu_mm_log(f"Clearing NPU cache on {device_str}")
multigpu_memory_log("general", f"pre-empty:{device_str}")
torch.npu.empty_cache()
logger.mgpu_mm_log(f"Cleared NPU cache on {device_str}")
multigpu_memory_log("general", f"post-empty:{device_str}")
elif device_str.startswith("mlu:"):
torch.mlu.empty_cache()
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Cambricon MLU")
if hasattr(torch, "mlu") and hasattr(torch.mlu, "empty_cache"):
logger.mgpu_mm_log(f"Clearing MLU cache on {device_str}")
multigpu_memory_log("general", f"pre-empty:{device_str}")
torch.mlu.empty_cache()
logger.mgpu_mm_log(f"Cleared MLU cache on {device_str}")
multigpu_memory_log("general", f"post-empty:{device_str}")
elif device_str.startswith("corex:"):
torch.corex.empty_cache() # Hypothetical based on ComfyUI's ixuca support
logger.debug("[MultiGPU_Device_Utils] Cleared cache for CoreX")
if hasattr(torch, "corex") and hasattr(torch.corex, "empty_cache"):
logger.mgpu_mm_log(f"Clearing CoreX cache on {device_str}")
multigpu_memory_log("general", f"pre-empty:{device_str}")
torch.corex.empty_cache()
logger.mgpu_mm_log(f"Cleared CoreX cache on {device_str}")
multigpu_memory_log("general", f"post-empty:{device_str}")
multigpu_memory_log("general", "post-soft-empty")
# ==========================================================================================
# Model Management Inspection Utilities (End-to-End Tracking)
# Comprehensive Memory Management (VRAM + CPU + Store Pruning)
# ==========================================================================================
def create_model_identifier(model_patcher):
"""Creates a concise, unique identifier for a model patcher based on type and size."""
if not model_patcher or not model_patcher.model:
return "N/A (Detached)"
logger.info("[MultiGPU Core Patching] Patching mm.soft_empty_cache for Comprehensive Memory Management (VRAM + CPU + Store Pruning)")
model = model_patcher.model
model_type = type(model).__name__
original_soft_empty_cache = mm.soft_empty_cache
# Try the fast path first (using size calculated by ModelPatcher)
try:
model_size = model_patcher.model_size()
except Exception:
model_size = 0
def soft_empty_cache_distorch2_patched(force=False):
"""Patched mm.soft_empty_cache managing VRAM across all devices, CPU RAM with adaptive thresholding, and DisTorch store pruning."""
from .model_management_mgpu import multigpu_memory_log, check_cpu_memory_threshold, trigger_executor_cache_reset
from .distorch_2 import safetensor_allocation_store, create_safetensor_model_hash
multigpu_memory_log("patched_soft_empty", f"start:force={force}")
is_distorch_active = False
# If the fast path fails or returns 0, perform a safe deep inspection
if model_size == 0:
try:
# Safely inspect parameters without triggering hooks/loads
with model_patcher.use_ejected(skip_and_inject_on_exit_only=True):
# We must iterate parameters() AND buffers() as both consume memory
params = list(model.parameters()) + list(model.buffers())
# Use data_ptr to handle potential weight tying/shared tensors correctly
seen_tensors = set()
for p in params:
if p.data_ptr() not in seen_tensors:
model_size += p.numel() * p.element_size()
seen_tensors.add(p.data_ptr())
except Exception as e:
logger.debug(f"[MultiGPU_Inspection] Error during safe size calculation for identifier: {e}")
return f"{model_type} (ID_Err)"
# Detect DisTorch2-managed models
logger.mgpu_mm_log(f"[DETECT_DEBUG] Checking DisTorch2 active status - loaded models: {len(mm.current_loaded_models)}, store entries: {len(safetensor_allocation_store)}")
for i, lm in enumerate(mm.current_loaded_models):
mp = lm.model # weakref call to ModelPatcher
if mp is not None:
try:
model_hash = create_safetensor_model_hash(mp, "cache_patch_check")
in_store = model_hash in safetensor_allocation_store
alloc_value = safetensor_allocation_store.get(model_hash, "")
model_name = type(getattr(mp, 'model', mp)).__name__
unload_distorch_model = getattr(getattr(mp, 'model', None), '_mgpu_unload_distorch_model', False)
logger.mgpu_mm_log(f"[DETECT_DEBUG] Model {i}: {model_name}, hash={model_hash[:8]}, in_store={in_store}, alloc_value='{alloc_value}', unload_distorch_model={unload_distorch_model}")
if in_store and alloc_value:
is_distorch_active = True
logger.mgpu_mm_log(f"[DETECT_DEBUG] DisTorch2 ACTIVE detected on model: {model_name}")
break
except Exception as e:
logger.mgpu_mm_log(f"[DETECT_DEBUG] Model {i}: Error during detection - {e}")
logger.mgpu_mm_log(f"[DETECT_DEBUG] Final DisTorch2 active status: {is_distorch_active}")
# Create a hash based on type and calculated size
identifier = f"{model_type}_{model_size}"
model_hash = hashlib.sha256(identifier.encode()).hexdigest()
return f"{model_type} ({model_hash[:8]})"
# Phase 2: adaptive CPU memory management
check_cpu_memory_threshold()
# VRAM allocator management
if is_distorch_active:
logger.mgpu_mm_log("DisTorch2 active: clearing allocator caches on all devices (VRAM)")
soft_empty_cache_multigpu()
else:
logger.mgpu_mm_log("DisTorch2 not active: delegating allocator cache clear (VRAM) to original mm.soft_empty_cache")
original_soft_empty_cache(force)
# Optional: return CPU heap to OS (not part of Comfy Core)
def analyze_tensor_locations(model_patcher):
"""
Analyzes the physical device placement of model tensors (parameters and buffers).
This provides the Ground Truth location of the data, handling shared weights correctly.
"""
device_summary = {}
seen_tensors = set()
total_memory = 0
# Phase 1/3: forced executor reset mirrors ComfyUI 'Free memory' semantics
if force:
logger.mgpu_mm_log("Force flag active: triggering executor cache reset (CPU)")
trigger_executor_cache_reset(reason="forced_soft_empty", force=True)
multigpu_memory_log("patched_soft_empty", "end")
if not model_patcher or not model_patcher.model:
return {"error": "Model not available"}, 0
mm.soft_empty_cache = soft_empty_cache_distorch2_patched
model = model_patcher.model
# ==========================================================================================
# Memory Inspection Utilities
# ==========================================================================================
# Crucial: Use the ejector to ensure we can access the model weights safely
# without interfering with injections, hooks, or triggering unintended loads (like in standard LowVRAM mode).
try:
with model_patcher.use_ejected(skip_and_inject_on_exit_only=True):
# Helper to process tensors (parameters or buffers)
def process_tensor(tensor):
nonlocal total_memory
# Use data_ptr() for unique identification of the underlying memory
if tensor.data_ptr() in seen_tensors:
return
seen_tensors.add(tensor.data_ptr())
def comfyui_memory_load(tag):
"""Return single-line pipe-delimited snapshot of system and device memory usage in GiB."""
# CPU RAM
vm = psutil.virtual_memory()
cpu_used_gib = vm.used / (1024.0 ** 3)
cpu_total_gib = vm.total / (1024.0 ** 3)
if tensor.numel() > 0:
tensor_mem = tensor.numel() * tensor.element_size()
total_memory += tensor_mem
segments = [f"tag={tag}", f"cpu={cpu_used_gib:.2f}/{cpu_total_gib:.2f}"]
if hasattr(tensor, 'device'):
device = str(tensor.device)
else:
# Handle cases like NF4 quantization or other custom tensors
device = "Unknown/Managed"
# Enumerate non-CPU devices
devices = [d for d in get_device_list() if d != "cpu"]
if device not in device_summary:
device_summary[device] = {'tensors': 0, 'memory': 0}
device_summary[device]['tensors'] += 1
device_summary[device]['memory'] += tensor_mem
# Iterate over all parameters (weights, biases)
for param in model.parameters():
process_tensor(param)
# Iterate over all buffers (like batch norm running stats)
for buffer in model.buffers():
process_tensor(buffer)
except Exception as e:
logger.error(f"[MultiGPU_Inspection] Error during tensor location analysis: {e}")
return {"error": str(e)}, 0
return device_summary, total_memory
def inspect_model_management_state(context_description=""):
"""
Provides a detailed, structured overview of the current state of ComfyUI's model management,
including memory usage across all devices and the status, location, and patching of all loaded models.
Call this function anywhere in the code to get an immediate snapshot of the system state.
"""
# Ensure logger configuration (handles calls before full MultiGPU init if needed)
if not logger.handlers:
handler = logging.StreamHandler()
formatter = logging.Formatter('%(message)s')
handler.setFormatter(formatter)
logger.addHandler(handler)
# Default to INFO if log level isn't set by main __init__.py
if logger.level == logging.NOTSET:
logger.setLevel(logging.INFO)
# We inspect the state without forcing GC or cache clearing, which might alter the state we want to observe.
logger.info("\n" + "=" * 100)
logger.info(f" INSPECTION: ComfyUI Model Management State [Context: {context_description}]")
logger.info("=" * 100)
# 1. Device Memory Overview
# Provides context on available resources across the system.
logger.info("--- [1] System Device Memory Overview (GB) ---")
# Sys Free: Memory available to the OS. Torch Alloc: Memory reserved by PyTorch (Active + Cache).
fmt_mem = "{:<12} | {:>10} | {:>10} | {:>10} | {:>15}"
logger.info(fmt_mem.format("Device", "Total", "Sys Free", "Used", "Torch Alloc"))
logger.info("-" * 70)
all_devices = get_device_list()
# Sort devices for consistent display (CPU last)
sorted_devices = sorted(all_devices, key=lambda d: (d == 'cpu', d))
for dev_str in sorted_devices:
try:
device = torch.device(dev_str)
if dev_str == "cpu":
vm = psutil.virtual_memory()
mem_total, mem_free_sys, mem_used = vm.total, vm.available, vm.used
torch_alloc = 0 # Difficult to track accurately for CPU globally
else:
# Use ComfyUI's management functions which account for different backends (CUDA, XPU, etc.)
mem_total = mm.get_total_memory(device)
# get_free_memory returns (system_free, torch_cache_free)
free_info = mm.get_free_memory(device, torch_free_too=True)
if isinstance(free_info, tuple):
mem_free_sys = free_info[0]
else:
mem_free_sys = free_info # Fallback for backends that return single value (like MPS)
mem_used = mem_total - mem_free_sys
# Determine Torch Allocation (Reserved memory) - Specific checks for known backends
torch_alloc = 0
if device.type == 'cuda' and hasattr(torch.cuda, 'memory_stats'):
stats = torch.cuda.memory_stats(device)
torch_alloc = stats.get('reserved_bytes.all.current', 0)
elif device.type == 'xpu' and hasattr(torch, 'xpu') and hasattr(torch.xpu, 'memory_stats'):
stats = torch.xpu.memory_stats(device)
torch_alloc = stats.get('reserved_bytes.all.current', 0)
elif device.type == 'npu' and hasattr(torch, 'npu') and hasattr(torch.npu, 'memory_stats'):
stats = torch.npu.memory_stats(device)
torch_alloc = stats.get('reserved_bytes.all.current', 0)
elif device.type == 'mlu' and hasattr(torch, 'mlu') and hasattr(torch.mlu, 'memory_stats'):
stats = torch.mlu.memory_stats(device)
torch_alloc = stats.get('reserved_bytes.all.current', 0)
# MPS, DirectML, CoreX do not always expose detailed reserved memory stats easily.
logger.info(fmt_mem.format(
dev_str,
f"{mem_total / (1024**3):.2f}",
f"{mem_free_sys / (1024**3):.2f}",
f"{mem_used / (1024**3):.2f}",
f"{torch_alloc / (1024**3):.2f}"
))
except Exception as e:
logger.debug(f"Could not retrieve memory stats for {dev_str}: {e}")
logger.info("-" * 70)
# 2. Loaded Models Inspection (Logical and Physical View)
# mm.current_loaded_models holds the list of models ComfyUI is managing.
loaded_models = mm.current_loaded_models
logger.info(f"\n--- [2] Loaded Models Inspection (Count: {len(loaded_models)}) ---")
if not loaded_models:
logger.info("No models currently managed by comfy.model_management.")
logger.info("=" * 100)
return
for i, lm in enumerate(loaded_models):
logger.info(f"\nModel {i+1}/{len(loaded_models)}:")
# Check lifecycle status
mp = lm.model # weakref call to ModelPatcher
if mp is None:
# ModelPatcher is gone. Check if the underlying model is still alive (potential leak)
if lm.is_dead() and lm.real_model() is not None:
logger.warning(f" [!] Status: LEAK DETECTED (Patcher GC'd, but underlying model {lm.real_model().__class__.__name__} persists)")
else:
logger.info(f" Status: Cleaned Up (Patcher and Model GC'd)")
continue
model_id = create_model_identifier(mp)
logger.info(f" Identifier: {model_id}")
logger.info(f" Status: {'Active (In Use)' if lm.currently_used else 'Idle (Cache)'}")
# A. Logical View (What ComfyUI intends/tracks)
logger.info(" [A] Logical View (ComfyUI Tracking):")
# Devices: Target (Compute) vs Offload (Storage)
logger.info(f" Devices: Target={lm.device} | Offload={mp.offload_device} | Current (Model.device)={mp.current_loaded_device()}")
# Memory Footprint
mem_total = lm.model_memory()
mem_loaded = lm.model_loaded_memory()
mem_offloaded = lm.model_offloaded_memory()
logger.info(f" Memory (MB): Total={mem_total/(1024**2):.2f} | Loaded (on Target)={mem_loaded/(1024**2):.2f} | Offloaded={mem_offloaded/(1024**2):.2f}")
# Management Mode (LowVRAM/DisTorch)
# model_lowvram indicates if ComfyUI is managing this model partially
is_lowvram = getattr(mp.model, 'model_lowvram', False)
lowvram_patches_pending = mp.lowvram_patch_counter()
logger.info(f" Mode: {'Partial Load (LowVRAM/DisTorch)' if is_lowvram else 'Full Load'}")
if is_lowvram:
# This indicates how many weights are being managed by the partial loading system
logger.info(f" Weights Managed by LowVRAM/DisTorch System: {lowvram_patches_pending}")
# Patching (LoRAs, etc.) - Tracking Attach/Detach
num_weight_patches = len(mp.patches)
# Check the UUID applied to the actual weights vs the UUID defined in the patcher
current_weight_uuid = getattr(mp.model, 'current_weight_patches_uuid', None)
weights_synced = (mp.patches_uuid == current_weight_uuid) and (current_weight_uuid is not None)
if num_weight_patches > 0:
status = 'Applied & Synced' if weights_synced else 'Pending/Mismatch (Re-patch needed)'
logger.info(f" Patches: {num_weight_patches} weight patches defined | Status: {status}")
logger.info(f" UUIDs: Defined={str(mp.patches_uuid)[:8]}... | Applied={str(current_weight_uuid)[:8] if current_weight_uuid else 'None'}...")
# B. Physical View (Ground Truth Tensor Locations)
logger.info(" [B] Physical View (Ground Truth Tensor Locations):")
device_summary, calculated_total_mem = analyze_tensor_locations(mp)
if "error" in device_summary:
logger.error(f" Analysis Error: {device_summary['error']}")
continue
if not device_summary:
logger.info(" No tensors found (e.g., fully offloaded CLIP or utility object).")
# Append per-device VRAM used/total
for dev_str in devices:
device = torch.device(dev_str)
total = mm.get_total_memory(device)
free_info = mm.get_free_memory(device, torch_free_too=True)
# free_info may be a tuple (system_free, torch_cache_free) or a single value
if isinstance(free_info, tuple):
system_free = free_info[0]
else:
# Sort devices (CPU last)
sorted_devices = sorted(device_summary.keys(), key=lambda d: (d.startswith("cpu"), d))
fmt_loc = " {:<15} | Tensors: {:>6} | Memory (MB): {:>10.2f} | Percent: {:>6.1f}%"
for device in sorted_devices:
data = device_summary[device]
percent = (data['memory'] / calculated_total_mem) * 100 if calculated_total_mem > 0 else 0
logger.info(fmt_loc.format(device, data['tensors'], data['memory']/(1024**2), percent))
system_free = free_info
used = max(0, (total or 0) - (system_free or 0))
# Verification Check
if abs(calculated_total_mem - mem_total) > (1024*1024): # Allow 1MB difference
logger.warning(f" [!] Verification WARNING: Physical memory ({calculated_total_mem/(1024**2):.2f}MB) differs from logical memory ({mem_total/(1024**2):.2f}MB).")
used_gib = used / (1024.0 ** 3)
total_gib = (total or 0) / (1024.0 ** 3)
if total_gib > 0:
segments.append(f"{dev_str}={used_gib:.2f}/{total_gib:.2f}")
logger.info("-" * 100)
logger.info("End of Inspection")
logger.info("=" * 100)
return "|".join(segments)
-525
View File
@@ -1,525 +0,0 @@
"""
DisTorch GGUF/GGML Memory Management Module
Contains all GGUF/GGML related code for distributed memory management
"""
import sys
import torch
import logging
import hashlib
logger = logging.getLogger("MultiGPU")
import copy
from collections import defaultdict
import comfy.model_management as mm
from .device_utils import get_device_list, soft_empty_cache_multigpu
# Global store for model allocations
model_allocation_store = {}
def create_model_hash(model, caller):
"""Create a unique hash for a model to track allocations"""
model_type = type(model.model).__name__
model_size = model.model_size()
first_layers = str(list(model.model_state_dict().keys())[:3])
identifier = f"{model_type}_{model_size}_{first_layers}"
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
logger.debug(f"[MultiGPU_DisTorch_HASH] Created hash for {caller}: {final_hash[:8]}...")
return final_hash
def register_patched_ggufmodelpatcher():
"""Register and patch the GGUFModelPatcher for distributed loading"""
from nodes import NODE_CLASS_MAPPINGS
original_loader = NODE_CLASS_MAPPINGS["UnetLoaderGGUF"]
module = sys.modules[original_loader.__module__]
if not hasattr(module.GGUFModelPatcher, '_patched'):
original_load = module.GGUFModelPatcher.load
def new_load(self, *args, force_patch_weights=False, **kwargs):
global model_allocation_store
super(module.GGUFModelPatcher, self).load(*args, force_patch_weights=True, **kwargs)
debug_hash = create_model_hash(self, "patcher")
linked = []
module_count = 0
for n, m in self.model.named_modules():
module_count += 1
if hasattr(m, "weight"):
device = getattr(m.weight, "device", None)
if device is not None:
linked.append((n, m))
continue
if hasattr(m, "bias"):
device = getattr(m.bias, "device", None)
if device is not None:
linked.append((n, m))
continue
if linked:
if hasattr(self, 'model'):
debug_hash = create_model_hash(self, "patcher")
debug_allocations = model_allocation_store.get(debug_hash)
if debug_allocations:
soft_empty_cache_multigpu(logger)
device_assignments = analyze_ggml_loading(self.model, debug_allocations)['device_assignments']
for device, layers in device_assignments.items():
target_device = torch.device(device)
for n, m, _ in layers:
m.to(self.load_device).to(target_device)
self.mmap_released = True
module.GGUFModelPatcher.load = new_load
module.GGUFModelPatcher._patched = True
def analyze_ggml_loading(model, allocations_str):
"""Analyze and distribute GGML model layers across devices"""
DEVICE_RATIOS_DISTORCH = {}
device_table = {}
distorch_alloc = allocations_str
virtual_vram_gb = 0.0
if '#' in allocations_str:
distorch_alloc, virtual_vram_str = allocations_str.split('#')
if not distorch_alloc:
distorch_alloc = calculate_vvram_allocation_string(model, virtual_vram_str)
eq_line = "=" * 47
dash_line = "-" * 47
fmt_assign = "{:<12}{:>10}{:>14}{:>10}"
for allocation in distorch_alloc.split(';'):
dev_name, fraction = allocation.split(',')
fraction = float(fraction)
total_mem_bytes = mm.get_total_memory(torch.device(dev_name))
alloc_gb = (total_mem_bytes * fraction) / (1024**3)
DEVICE_RATIOS_DISTORCH[dev_name] = alloc_gb
device_table[dev_name] = {
"fraction": fraction,
"total_gb": total_mem_bytes / (1024**3),
"alloc_gb": alloc_gb
}
logger.info(eq_line)
logger.info(" DisTorch Model Device Allocations")
logger.info(eq_line)
logger.info(fmt_assign.format("Device", "Alloc %", "Total (GB)", " Alloc (GB)"))
logger.info(dash_line)
sorted_devices = sorted(device_table.keys(), key=lambda d: (d == "cpu", d))
for dev in sorted_devices:
frac = device_table[dev]["fraction"]
tot_gb = device_table[dev]["total_gb"]
alloc_gb = device_table[dev]["alloc_gb"]
logger.info(fmt_assign.format(dev,f"{int(frac * 100)}%",f"{tot_gb:.2f}",f"{alloc_gb:.2f}"))
logger.info(dash_line)
layer_summary = {}
layer_list = []
memory_by_type = defaultdict(int)
total_memory = 0
for name, module in model.named_modules():
if hasattr(module, "weight"):
layer_type = type(module).__name__
layer_summary[layer_type] = layer_summary.get(layer_type, 0) + 1
layer_list.append((name, module, layer_type))
layer_memory = 0
if module.weight is not None:
layer_memory += module.weight.numel() * module.weight.element_size()
if hasattr(module, "bias") and module.bias is not None:
layer_memory += module.bias.numel() * module.bias.element_size()
memory_by_type[layer_type] += layer_memory
total_memory += layer_memory
logger.info(" DisTorch Model Layer Distribution")
logger.info(dash_line)
fmt_layer = "{:<12}{:>10}{:>14}{:>10}"
logger.info(fmt_layer.format("Layer Type", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
for layer_type, count in layer_summary.items():
mem_mb = memory_by_type[layer_type] / (1024 * 1024)
mem_percent = (memory_by_type[layer_type] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_layer.format(layer_type,str(count),f"{mem_mb:.2f}",f"{mem_percent:.1f}%"))
logger.info(dash_line)
nonzero_devices = [d for d, r in DEVICE_RATIOS_DISTORCH.items() if r > 0]
nonzero_total_ratio = sum(DEVICE_RATIOS_DISTORCH[d] for d in nonzero_devices)
device_assignments = {device: [] for device in DEVICE_RATIOS_DISTORCH.keys()}
total_layers = len(layer_list)
current_layer = 0
for idx, device in enumerate(nonzero_devices):
ratio = DEVICE_RATIOS_DISTORCH[device]
if idx == len(nonzero_devices) - 1:
device_layer_count = total_layers - current_layer
else:
device_layer_count = int((ratio / nonzero_total_ratio) * total_layers)
start_idx = current_layer
end_idx = current_layer + device_layer_count
device_assignments[device] = layer_list[start_idx:end_idx]
current_layer += device_layer_count
logger.info("DisTorch Model Final Device/Layer Assignments")
logger.info(dash_line)
fmt_assign = "{:<12}{:>10}{:>14}{:>10}"
logger.info(fmt_assign.format("Device", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
total_assigned_memory = 0
device_memories = {}
for device, layers in device_assignments.items():
device_memory = 0
for layer_type in layer_summary:
type_layers = sum(1 for _, _, lt in layers if lt == layer_type)
if layer_summary[layer_type] > 0:
mem_per_layer = memory_by_type[layer_type] / layer_summary[layer_type]
device_memory += mem_per_layer * type_layers
device_memories[device] = device_memory
total_assigned_memory += device_memory
sorted_assignments = sorted(device_assignments.keys(), key=lambda d: (d == "cpu", d))
for dev in sorted_assignments:
layers = device_assignments[dev]
mem_mb = device_memories[dev] / (1024 * 1024)
mem_percent = (device_memories[dev] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_assign.format(dev,str(len(layers)),f"{mem_mb:.2f}",f"{mem_percent:.1f}%"))
logger.info(dash_line)
return {"device_assignments": device_assignments}
def calculate_vvram_allocation_string(model, virtual_vram_str):
"""Calculate virtual VRAM allocation string for distributed loading"""
recipient_device, vram_amount, donors = virtual_vram_str.split(';')
virtual_vram_gb = float(vram_amount)
eq_line = "=" * 47
dash_line = "-" * 47
fmt_assign = "{:<8} {:<6} {:>11} {:>9} {:>9}"
logger.info(eq_line)
logger.info(" DisTorch Model Virtual VRAM Analysis")
logger.info(eq_line)
logger.info(fmt_assign.format("Object", "Role", "Original(GB)", "Total(GB)", "Virt(GB)"))
logger.info(dash_line)
recipient_vram = mm.get_total_memory(torch.device(recipient_device)) / (1024**3)
recipient_virtual = recipient_vram + virtual_vram_gb
logger.info(fmt_assign.format(recipient_device, 'recip', f"{recipient_vram:.2f}GB",f"{recipient_virtual:.2f}GB", f"+{virtual_vram_gb:.2f}GB"))
ram_donors = [d for d in donors.split(',') if d != 'cpu']
remaining_vram_needed = virtual_vram_gb
donor_device_info = {}
donor_allocations = {}
for donor in ram_donors:
donor_vram = mm.get_total_memory(torch.device(donor)) / (1024**3)
max_donor_capacity = donor_vram * 0.9
donation = min(remaining_vram_needed, max_donor_capacity)
donor_virtual = donor_vram - donation
remaining_vram_needed -= donation
donor_allocations[donor] = donation
donor_device_info[donor] = (donor_vram, donor_virtual)
logger.info(fmt_assign.format(donor, 'donor', f"{donor_vram:.2f}GB", f"{donor_virtual:.2f}GB", f"-{donation:.2f}GB"))
system_dram_gb = mm.get_total_memory(torch.device('cpu')) / (1024**3)
cpu_donation = remaining_vram_needed
cpu_virtual = system_dram_gb - cpu_donation
donor_allocations['cpu'] = cpu_donation
logger.info(fmt_assign.format('cpu', 'donor', f"{system_dram_gb:.2f}GB", f"{cpu_virtual:.2f}GB", f"-{cpu_donation:.2f}GB"))
logger.info(dash_line)
layer_summary = {}
layer_list = []
memory_by_type = defaultdict(int)
total_memory = 0
for name, module in model.named_modules():
if hasattr(module, "weight"):
layer_type = type(module).__name__
layer_summary[layer_type] = layer_summary.get(layer_type, 0) + 1
layer_list.append((name, module, layer_type))
layer_memory = 0
if module.weight is not None:
layer_memory += module.weight.numel() * module.weight.element_size()
if hasattr(module, "bias") and module.bias is not None:
layer_memory += module.bias.numel() * module.bias.element_size()
memory_by_type[layer_type] += layer_memory
total_memory += layer_memory
model_size_gb = total_memory / (1024**3)
new_model_size_gb = max(0, model_size_gb - virtual_vram_gb)
logger.info(fmt_assign.format('model', 'model', f"{model_size_gb:.2f}GB",f"{new_model_size_gb:.2f}GB", f"-{virtual_vram_gb:.2f}GB"))
if model_size_gb > (recipient_vram * 0.9):
on_recipient = recipient_vram * 0.9
on_virtuals = model_size_gb - on_recipient
logger.info(f"\nWarning: Model size is greater than 90% of recipient VRAM. {on_virtuals:.2f} GB of GGML Layers Offloaded Automatically to Virtual VRAM.\n")
else:
on_recipient = model_size_gb
on_virtuals = 0
new_on_recipient = max(0, on_recipient - virtual_vram_gb)
allocation_parts = []
recipient_percent = new_on_recipient / recipient_vram
allocation_parts.append(f"{recipient_device},{recipient_percent:.4f}")
for donor in ram_donors:
donor_vram = donor_device_info[donor][0]
donor_percent = donor_allocations[donor] / donor_vram
allocation_parts.append(f"{donor},{donor_percent:.4f}")
cpu_percent = donor_allocations['cpu'] / system_dram_gb
allocation_parts.append(f"cpu,{cpu_percent:.4f}")
allocation_string = ";".join(allocation_parts)
fmt_mem = "{:<20}{:>20}"
logger.info(fmt_mem.format("\n v1 Expert String", allocation_string))
return allocation_string
def override_class_with_distorch_gguf(cls):
"""Legacy DisTorch wrapper for GGUF models for backward compatibility."""
from . import current_device
class NodeOverrideDisTorchGGUFLegacy(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
"multiline": False,
"default": "",
})
return inputs
CATEGORY = "multigpu/legacy"
FUNCTION = "override"
if hasattr(cls, 'TITLE'):
TITLE = f"{cls.TITLE} (Legacy)"
else:
TITLE = "Legacy DisTorch Node"
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
from . import set_current_device
if device is not None:
set_current_device(device)
register_patched_ggufmodelpatcher()
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
if hasattr(out[0], 'model'):
model_hash = create_model_hash(out[0], "override")
model_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_model_hash(out[0].patcher, "override")
model_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchGGUFLegacy
def override_class_with_distorch_gguf_v2(cls):
"""DisTorch 2.0 wrapper for GGUF models."""
from . import current_device
class NodeOverrideDisTorchGGUFv2(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
compute_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
def override(self, *args, compute_device=None, virtual_vram_gb=4.0,
donor_device="cpu", expert_mode_allocations="", **kwargs):
from . import set_current_device
if compute_device is not None:
set_current_device(compute_device)
register_patched_ggufmodelpatcher()
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logger.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_model_hash(out[0], "override")
model_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_model_hash(out[0].patcher, "override")
model_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchGGUFv2
def override_class_with_distorch_clip(cls):
"""DisTorch wrapper for CLIP models with GGUF support"""
from . import current_text_encoder_device
class NodeOverrideDisTorch(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
"multiline": False,
"default": "",
"tooltip": "Expert use only: Manual VRAM allocation string. Incorrect values can cause crashes. Do not modify unless you fully understand DisTorch memory management."
})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
from . import set_current_text_encoder_device
if device is not None:
set_current_text_encoder_device(device)
register_patched_ggufmodelpatcher()
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logging.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_model_hash(out[0], "override")
model_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_model_hash(out[0].patcher, "override")
model_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorch
def override_class_with_distorch_clip_no_device(cls):
"""DisTorch wrapper for CLIP models with GGUF support"""
from . import current_text_encoder_device
class NodeOverrideDisTorchClipNoDevice(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
"multiline": False,
"default": "",
"tooltip": "Expert use only: Manual VRAM allocation string. Incorrect values can cause crashes. Do not modify unless you fully understand DisTorch memory management."
})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
from . import set_current_text_encoder_device
if device is not None:
set_current_text_encoder_device(device)
register_patched_ggufmodelpatcher()
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logging.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_model_hash(out[0], "override")
model_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_model_hash(out[0].patcher, "override")
model_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchClipNoDevice
# Alias for backward compatibility
override_class_with_distorch = override_class_with_distorch_gguf
+127 -546
View File
@@ -16,8 +16,9 @@ import inspect
from collections import defaultdict
import comfy.model_management as mm
import comfy.model_patcher
from . import current_device
from .device_utils import get_device_list, soft_empty_cache_multigpu
from .model_management_mgpu import multigpu_memory_log, force_full_system_cleanup
safetensor_allocation_store = {}
safetensor_settings_store = {}
@@ -48,15 +49,61 @@ def create_safetensor_model_hash(model, caller):
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
# DEBUG STATEMENT - ALWAYS LOG THE HASH
logger.debug(f"[MultiGPU_DisTorch2] Created hash for {caller}: {final_hash[:8]}...")
logger.debug(f"[MultiGPU DisTorch V2] Created hash for {caller}: {final_hash[:8]}...")
return final_hash
def register_patched_safetensor_modelpatcher():
"""Register and patch the ModelPatcher for distributed safetensor loading"""
from comfy.model_patcher import wipe_lowvram_weight, move_weight_functions
# Patch ComfyUI's ModelPatcher
if not hasattr(comfy.model_patcher.ModelPatcher, '_distorch_patched'):
# Patch LoadedModel.model_memory_required to drive behavior purely by Phase 2 = unload_distorch_model flag
from comfy.model_management import current_loaded_models
original_loaded_model_memory_required = None
for cls in current_loaded_models.__class__.__mro__:
if hasattr(cls, 'model_memory_required'):
original_loaded_model_memory_required = cls.model_memory_required
break
if original_loaded_model_memory_required is None:
# Global patch of LoadedModel class if available
import comfy.model_management as mm
original_loaded_model_memory_required = mm.LoadedModel.model_memory_required
def patched_loaded_model_memory_required(self, device):
"""Drive unload behavior purely by unload_distorch_model flag"""
multigpu_memory_log("unload_distorch_model_memory_check", "start")
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Memory assessment requested for model on device: {device}")
# Check if this is a DisTorch model with unload_distorch_model flag
is_distorch_model = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_unload_distorch_model')
model_name = type(getattr(getattr(self, 'model', None), 'model', None)).__name__ if getattr(getattr(self, 'model', None), 'model', None) else "Unknown"
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] DisTorch model: {model_name}, is_distorch_model={is_distorch_model}")
if is_distorch_model:
if self.model.model._mgpu_unload_distorch_model:
total_device_memory = mm.get_total_memory(device)
memory_gb = total_device_memory / (1024**3)
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=True - Reporting MAX memory ({memory_gb:.2f}GB) to force complete eviction")
return total_device_memory
else:
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=False - Reporting 0 bytes (prevents eviction)")
return 0
# Not a DisTorch model - use original behavior
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] Non-DisTorch model - Using original Comfy memory calculation")
original_result = original_loaded_model_memory_required(self, device)
original_gb = original_result / (1024**3) if original_result else 0
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Original calculation returned: {original_gb:.2f}GB")
multigpu_memory_log("keep_loaded_memory_check", "end")
return original_result
mm.LoadedModel.model_memory_required = patched_loaded_model_memory_required
original_partially_load = comfy.model_patcher.ModelPatcher.partially_load
def new_partially_load(self, device_to, extra_memory=0, full_load=False, force_patch_weights=False, **kwargs):
@@ -64,11 +111,16 @@ def register_patched_safetensor_modelpatcher():
global safetensor_allocation_store
debug_hash = create_safetensor_model_hash(self, "partial_load")
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "pre-load")
allocations = safetensor_allocation_store.get(debug_hash)
if not hasattr(self.model, '_distorch_high_precision_loras') or not allocations:
# Set default precision flag before checking
if not hasattr(self.model, '_distorch_high_precision_loras'):
self.model._distorch_high_precision_loras = True
if not allocations:
result = original_partially_load(self, device_to, extra_memory, force_patch_weights)
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "post-load")
if hasattr(self, '_distorch_block_assignments'):
del self._distorch_block_assignments
return result
@@ -79,7 +131,7 @@ def register_patched_safetensor_modelpatcher():
unpatch_weights = self.model.current_weight_patches_uuid is not None and (self.model.current_weight_patches_uuid != self.patches_uuid or force_patch_weights)
if unpatch_weights:
logger.info(f"[MultiGPU_DisTorch2] Patches changed or forced. Unpatching model.")
logger.debug(f"[MultiGPU DisTorch V2] Patches changed or forced. Unpatching model.")
self.unpatch_model(self.offload_device, unpatch_weights=True)
self.patch_model(load_weights=False)
@@ -87,15 +139,10 @@ def register_patched_safetensor_modelpatcher():
mem_counter = 0
is_clip_model = getattr(self, 'is_clip', False)
if is_clip_model:
logger.info(f"[MultiGPU_DisTorch2] Using CLIP-specific allocation for model {debug_hash[:8]} (HEAD PRESERVATION ENABLED)")
device_assignments = analyze_safetensor_loading_clip(self, allocations)
else:
logger.debug(f"[MultiGPU_DisTorch2] Using standard allocation for model {debug_hash[:8]} (UNET/VAE - UNTOUCHED)")
device_assignments = analyze_safetensor_loading(self, allocations)
device_assignments = analyze_safetensor_loading(self, allocations, is_clip=is_clip_model)
model_original_dtype = comfy.utils.weight_dtype(self.model.state_dict())
high_precision_loras = self.model._distorch_high_precision_loras
high_precision_loras = getattr(self.model, "_distorch_high_precision_loras", True)
loading = self._load_list()
loading.sort(reverse=True)
for module_size, module_name, module_object, params in loading:
@@ -109,7 +156,7 @@ def register_patched_safetensor_modelpatcher():
pass
if current_module_device is not None and str(current_module_device) != str(block_target_device):
logger.debug(f"[MultiGPU_DisTorch2] Moving already patched {module_name} to {block_target_device}")
logger.debug(f"[MultiGPU DisTorch V2] Moving already patched {module_name} to {block_target_device}")
module_object.to(block_target_device)
mem_counter += module_size
@@ -143,11 +190,11 @@ def register_patched_safetensor_modelpatcher():
new_param = torch.nn.Parameter(cast_data.to(torch.float8_e4m3fn))
new_param.requires_grad = param.requires_grad
setattr(module_object, param_name, new_param)
logger.debug(f"[MultiGPU_DisTorch2] Cast {module_name}.{param_name} to FP8 for CPU storage")
logger.debug(f"[MultiGPU DisTorch V2] Cast {module_name}.{param_name} to FP8 for CPU storage")
# Step 4: Move to ultimate destination based on DisTorch assignment
if block_target_device != device_to:
logger.debug(f"[MultiGPU_DisTorch2] Moving {module_name} from {device_to} to {block_target_device}")
logger.debug(f"[MultiGPU DisTorch V2] Moving {module_name} from {device_to} to {block_target_device}")
module_object.to(block_target_device)
module_object.comfy_cast_weights = True
@@ -157,20 +204,39 @@ def register_patched_safetensor_modelpatcher():
self.model.current_weight_patches_uuid = self.patches_uuid
logger.info(f"[MultiGPU_DisTorch2] DisTorch loading completed. Total memory: {mem_counter / (1024 * 1024):.2f}MB")
logger.info("[MultiGPU DisTorch V2] DisTorch loading completed.")
logger.info(f"[MultiGPU DisTorch V2] Total memory: {mem_counter / (1024 * 1024):.2f}MB")
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "post-load")
return 0
comfy.model_patcher.ModelPatcher.partially_load = new_partially_load
comfy.model_patcher.ModelPatcher._distorch_patched = True
logger.info("[MultiGPU_DisTorch2] Successfully patched ModelPatcher.partially_load")
logger.info("[MultiGPU Core Patching] Successfully patched ModelPatcher.partially_load")
def _extract_clip_head_blocks(raw_block_list, compute_device):
"""Identify and pre-assign CLIP head blocks to compute device returning head_blocks, distributable_blocks, block_assignments, and head_memory."""
head_keywords = ['embed', 'wte', 'wpe', 'token_embedding', 'position_embedding']
head_blocks = []
distributable_blocks = []
head_memory = 0
block_assignments = {}
for module_size, module_name, module_object, params in raw_block_list:
if any(kw in module_name.lower() for kw in head_keywords):
head_blocks.append((module_size, module_name, module_object, params))
block_assignments[module_name] = compute_device
head_memory += module_size
else:
distributable_blocks.append((module_size, module_name, module_object, params))
return head_blocks, distributable_blocks, block_assignments, head_memory
def analyze_safetensor_loading(model_patcher, allocations_string):
def analyze_safetensor_loading(model_patcher, allocations_string, is_clip=False):
"""
Analyze and distribute safetensor model blocks across devices
Target for refactor back into one function once stability for CLIP is established.
Analyze and distribute safetensor model blocks across devices.
Supports CLIP head preservation when is_clip=True.
"""
DEVICE_RATIOS_DISTORCH = {}
device_table = {}
@@ -180,13 +246,10 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
distorch_alloc, virtual_vram_str = allocations_string.split('#')
compute_device = virtual_vram_str.split(';')[0]
logger.info(f"[MultiGPU_DisTorch2] Compute Device: {compute_device}")
logger.debug(f"[MultiGPU DisTorch V2] Compute Device: {compute_device}")
if not distorch_alloc:
mode = "fraction"
logger.info("[MultiGPU_DisTorch2] Expert String Examples:")
logger.info(" Direct(byte) Mode - cuda:0,500mb;cuda:1,3.0g;cpu,5gb* -> '*' cpu = over/underflow device, put 0.50gb on cuda0, 3.00gb on cuda1, and 5.00gb (or the rest) on cpu")
logger.info(" Ratio(%) Mode - cuda:0,8%;cuda:1,8%;cpu,4% -> 8:8:4 ratio, put 40% on cuda0, 40% on cuda1, and 20% on cpu")
distorch_alloc = calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str)
elif any(c in distorch_alloc.lower() for c in ['g', 'm', 'k', 'b']):
@@ -202,12 +265,13 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
if device not in present_devices:
distorch_alloc += f";{device},0.0"
logger.info(f"[MultiGPU_DisTorch2] Final Allocation String: {distorch_alloc}")
eq_line = "=" * 50
dash_line = "-" * 50
fmt_assign = "{:<18}{:>7}{:>14}{:>10}"
logger.info(eq_line)
logger.info(f"[MultiGPU DisTorch V2] Final Allocation String:\n{distorch_alloc}")
for allocation in distorch_alloc.split(';'):
if ',' not in allocation:
continue
@@ -257,13 +321,23 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
total_memory = 0
raw_block_list = model_patcher._load_list()
total_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
MIN_BLOCK_THRESHOLD = total_memory * 0.0001
logger.debug(f"[MultiGPU_DisTorch2] Total model memory: {total_memory} bytes")
logger.debug(f"[MultiGPU_DisTorch2] Tiny block threshold (0.01%): {MIN_BLOCK_THRESHOLD} bytes")
logger.debug(f"[MultiGPU DisTorch V2] Total model memory: {total_memory} bytes")
logger.debug(f"[MultiGPU DisTorch V2] Tiny block threshold (0.01%): {MIN_BLOCK_THRESHOLD} bytes")
# CLIP-specific: Extract head blocks and get pre-assignments
head_memory = 0
block_assignments = {}
if is_clip:
head_blocks, distributable_raw, block_assignments, head_memory = \
_extract_clip_head_blocks(raw_block_list, compute_device)
logger.info(f"[MultiGPU DisTorch V2 CLIP] Preserving {len(head_blocks)} head layer(s) ({head_memory/(1024**2):.2f} MB) on compute device: {compute_device}")
else:
distributable_raw = raw_block_list
# Build all_blocks list for summary (using full raw_block_list)
all_blocks = []
for module_size, module_name, module_object, params in raw_block_list:
block_type = type(module_object).__name__
@@ -272,12 +346,17 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
memory_by_type[block_type] += module_size
all_blocks.append((module_name, module_object, block_type, module_size))
block_list = [b for b in all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
tiny_block_list = [b for b in all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
# Use distributable blocks for actual allocation (for CLIP, this excludes heads)
distributable_all_blocks = []
for module_size, module_name, module_object, params in distributable_raw:
distributable_all_blocks.append((module_name, module_object, type(module_object).__name__, module_size))
block_list = [b for b in distributable_all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
tiny_block_list = [b for b in distributable_all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
logger.debug(f"[MultiGPU_DisTorch2] Total blocks: {len(all_blocks)}")
logger.debug(f"[MultiGPU_DisTorch2] Distributable blocks: {len(block_list)}")
logger.debug(f"[MultiGPU_DisTorch2] Tiny blocks (<0.01%): {len(tiny_block_list)}")
logger.debug(f"[MultiGPU DisTorch V2] Total blocks: {len(all_blocks)}")
logger.debug(f"[MultiGPU DisTorch V2] Distributable blocks: {len(block_list)}")
logger.debug(f"[MultiGPU DisTorch V2] Tiny blocks (<0.01%): {len(tiny_block_list)}")
logger.info(" DisTorch2 Model Layer Distribution")
logger.info(dash_line)
@@ -304,6 +383,11 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
for dev in donor_devices
}
# CLIP-specific: Adjust compute_device quota to account for locked head blocks
if is_clip and compute_device in donor_quotas and head_memory > 0:
donor_quotas[compute_device] = max(0, donor_quotas[compute_device] - head_memory)
logger.debug(f"[MultiGPU DisTorch V2 CLIP] Adjusted {compute_device} quota by -{head_memory/(1024**2):.2f} MB for head preservation")
# Iterate from the TAIL of the model, assigning blocks to donors until their quotas are filled.
for block_name, module, block_type, block_memory in reversed(block_list):
assigned_to_donor = False
@@ -340,7 +424,7 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
tiny_mem_percent = (tiny_block_memory / total_memory) * 100 if total_memory > 0 else 0
device_label = f"{compute_device} (<0.01%)"
logger.info(fmt_assign.format(device_label, str(len(tiny_block_list)), f"{tiny_mem_mb:.2f}", f"{tiny_mem_percent:.1f}%"))
logger.debug(f"[MultiGPU_DisTorch2] Tiny block memory breakdown: {tiny_block_memory} bytes ({tiny_mem_mb:.2f} MB), which is {tiny_mem_percent:.4f}% of total model memory.")
logger.debug(f"[MultiGPU DisTorch V2] Tiny block memory breakdown: {tiny_block_memory} bytes ({tiny_mem_mb:.2f} MB), which is {tiny_mem_percent:.4f}% of total model memory.")
total_assigned_memory = 0
device_memories = {}
@@ -373,211 +457,6 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
"block_assignments": block_assignments
}
def analyze_safetensor_loading_clip(model_patcher, allocations_string):
"""
CLIP-SPECIFIC: A 1:1 clone of the working UNET allocation logic with the
single required modification to preserve head-blocks on the compute device.
All other logic and UX (logging, etc.) is identical to the original.
Target for refactor once stability for CLIP is established.
"""
DEVICE_RATIOS_DISTORCH = {}
device_table = {}
distorch_alloc = allocations_string
virtual_vram_gb = 0.0
distorch_alloc, virtual_vram_str = allocations_string.split('#')
compute_device = virtual_vram_str.split(';')[0]
logger.info(f"[MultiGPU_DisTorch2_CLIP] CLIP Compute Device: {compute_device}")
if not distorch_alloc:
mode = "fraction"
logger.info("[MultiGPU_DisTorch2_CLIP] Expert String Examples:")
logger.info(" Direct(byte) Mode - cuda:0,500mb;cuda:1,3.0g;cpu,5gb* -> '*' cpu = over/underflow device, put 0.50gb on cuda0, 3.00gb on cuda1, and 5.00gb (or the rest) on cpu")
logger.info(" Ratio(%) Mode - cuda:0,8%;cuda:1,8%;cpu,4% -> 8:8:4 ratio, put 40% on cuda0, 40% on cuda1, and 20% on cpu")
distorch_alloc = calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str)
elif any(c in distorch_alloc.lower() for c in ['g', 'm', 'k', 'b']):
mode = "byte"
distorch_alloc = calculate_fraction_from_byte_expert_string(model_patcher, distorch_alloc)
elif "%" in distorch_alloc:
mode = "ratio"
distorch_alloc = calculate_fraction_from_ratio_expert_string(model_patcher, distorch_alloc)
all_devices = get_device_list()
present_devices = {item.split(',')[0] for item in distorch_alloc.split(';') if ',' in item}
for device in all_devices:
if device not in present_devices:
distorch_alloc += f";{device},0.0"
logger.info(f"[MultiGPU_DisTorch2_CLIP] Final CLIP Allocation String: {distorch_alloc}")
eq_line = "=" * 50
dash_line = "-" * 50
fmt_assign = "{:<18}{:>7}{:>14}{:>10}"
for allocation in distorch_alloc.split(';'):
if ',' not in allocation:
continue
dev_name, fraction = allocation.split(',')
fraction = float(fraction)
total_mem_bytes = mm.get_total_memory(torch.device(dev_name))
alloc_gb = (total_mem_bytes * fraction) / (1024**3)
DEVICE_RATIOS_DISTORCH[dev_name] = alloc_gb
device_table[dev_name] = {
"fraction": fraction,
"total_gb": total_mem_bytes / (1024**3),
"alloc_gb": alloc_gb
}
logger.info(eq_line)
logger.info(" DisTorch2 CLIP Model Device Allocations")
logger.info(eq_line)
fmt_rosetta = "{:<8}{:>9}{:>9}{:>11}{:>10}"
logger.info(fmt_rosetta.format("Device", "VRAM GB", "Dev %", "Model GB", "Dist %"))
logger.info(dash_line)
sorted_devices = sorted(device_table.keys(), key=lambda d: (d == "cpu", d))
total_allocated_model_bytes = sum(d["alloc_gb"] * (1024**3) for d in device_table.values())
for dev in sorted_devices:
total_dev_gb = device_table[dev]["total_gb"]
alloc_fraction = device_table[dev]["fraction"]
alloc_gb = device_table[dev]["alloc_gb"]
dist_ratio_percent = (alloc_gb * (1024**3) / total_allocated_model_bytes) * 100 if total_allocated_model_bytes > 0 else 0
logger.info(fmt_rosetta.format(
dev,
f"{total_dev_gb:.2f}",
f"{alloc_fraction*100:.1f}%",
f"{alloc_gb:.2f}",
f"{dist_ratio_percent:.1f}%"
))
logger.info(dash_line)
block_summary = {}
memory_by_type = defaultdict(int)
raw_block_list = model_patcher._load_list()
total_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
# Split the model into head and distributable parts
head_keywords = ['embed', 'wte', 'wpe', 'token_embedding', 'position_embedding']
head_blocks = []
distributable_blocks_raw = []
head_memory = 0
for module_size, module_name, module_object, params in raw_block_list:
if any(keyword in module_name.lower() for keyword in head_keywords):
head_blocks.append((module_size, module_name, module_object, params))
else:
distributable_blocks_raw.append((module_size, module_name, module_object, params))
MIN_BLOCK_THRESHOLD = total_memory * 0.0001
all_blocks = []
for module_size, module_name, module_object, params in raw_block_list:
block_type = type(module_object).__name__
block_summary[block_type] = block_summary.get(block_type, 0) + 1
memory_by_type[block_type] += module_size
all_blocks.append((module_name, module_object, block_type, module_size))
# Use the distributable part for actual allocation logic
distributable_all_blocks = []
for module_size, module_name, module_object, params in distributable_blocks_raw:
distributable_all_blocks.append((module_name, module_object, type(module_object).__name__, module_size))
block_list = [b for b in distributable_all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
tiny_block_list = [b for b in distributable_all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
logger.info(" DisTorch2 CLIP Model Layer Distribution")
logger.info(dash_line)
fmt_layer = "{:<18}{:>7}{:>14}{:>10}"
logger.info(fmt_layer.format("Layer Type", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
for layer_type, count in block_summary.items():
mem_mb = memory_by_type[layer_type] / (1024 * 1024)
mem_percent = (memory_by_type[layer_type] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_layer.format(layer_type[:18], str(count), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
logger.info(dash_line)
block_assignments = {}
# Pre-assign head blocks and calculate their memory usage
for module_size, module_name, module_object, params in head_blocks:
block_assignments[module_name] = compute_device
head_memory += module_size
if head_blocks:
logger.info(f"[MultiGPU_DisTorch2_CLIP] Preserving {len(head_blocks)} head layer(s) ({head_memory / (1024*1024):.2f} MB) on compute device: {compute_device}")
donor_devices = [d for d in sorted_devices]
donor_quotas = {
dev: device_table[dev]["alloc_gb"] * (1024**3)
for dev in donor_devices
}
# Adjust compute_device quota to account for the locked head
if compute_device in donor_quotas:
donor_quotas[compute_device] = max(0, donor_quotas[compute_device] - head_memory)
for block_name, module, block_type, block_memory in reversed(block_list):
assigned_to_donor = False
for donor in donor_devices:
if donor_quotas[donor] >= block_memory:
block_assignments[block_name] = donor
donor_quotas[donor] -= block_memory
assigned_to_donor = True
break # Move to the next block
if not assigned_to_donor:
block_assignments[block_name] = compute_device
for block_name, module, block_type, block_memory in tiny_block_list:
block_assignments[block_name] = compute_device
device_assignments = {device: [] for device in DEVICE_RATIOS_DISTORCH.keys()}
for block_name, device in block_assignments.items():
# Find the block in the original list to get all its info
for b_name, b_module, b_type, b_mem in all_blocks:
if b_name == block_name:
device_assignments[device].append((b_name, b_module, b_type, b_mem))
break
logger.info("DisTorch2 CLIP Model Final Device/Layer Assignments")
logger.info(dash_line)
logger.info(fmt_assign.format("Device", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
device_memories = defaultdict(int)
device_counts = defaultdict(int)
for device, blocks in device_assignments.items():
for b_name, b_module, b_type, b_mem in blocks:
device_memories[device] += b_mem
device_counts[device] += 1
sorted_assignments = sorted(device_memories.keys(), key=lambda d: (d == "cpu", d))
for dev in sorted_assignments:
if device_counts[dev] == 0:
continue
mem_mb = device_memories[dev] / (1024 * 1024)
mem_percent = (device_memories[dev] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_assign.format(dev, str(device_counts[dev]), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
logger.info(dash_line)
return {
"device_assignments": device_assignments,
"block_assignments": block_assignments
}
def parse_memory_string(mem_str):
"""Parses a memory string (e.g., '4.0g', '512M') and returns bytes."""
mem_str = mem_str.strip().lower()
@@ -598,11 +477,7 @@ def parse_memory_string(mem_str):
return val
def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
"""
Converts a user-provided byte string (e.g., "cuda:1,4gb;cpu,*") into a
fractional VRAM allocation string that the main assignment logic can use.
This function strictly respects device order and byte quotas.
"""
"""Convert byte allocation string (e.g. 'cuda:1,4gb;cpu,*') to fractional VRAM allocation string respecting device order and byte quotas."""
raw_block_list = model_patcher._load_list()
total_model_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
remaining_model_bytes = total_model_memory
@@ -637,16 +512,16 @@ def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
if bytes_to_assign > 0:
final_byte_allocations[dev] = bytes_to_assign
remaining_model_bytes -= bytes_to_assign
logger.info(f"[MultiGPU_DisTorch2] Assigning {bytes_to_assign / (1024**2):.2f}MB of model to {dev} (requested {requested_bytes / (1024**2):.2f}MB).")
logger.info(f"[MultiGPU DisTorch V2] Assigning {bytes_to_assign / (1024**2):.2f}MB of model to {dev} (requested {requested_bytes / (1024**2):.2f}MB).")
if remaining_model_bytes <= 0:
logger.info("[MultiGPU_DisTorch2] All model blocks have been allocated. Subsequent devices in the string will receive no assignment.")
logger.info("[MultiGPU DisTorch V2] All model blocks have been allocated. Subsequent devices in the string will receive no assignment.")
break
# Assign any leftover model bytes to the wildcard device
if remaining_model_bytes > 0:
final_byte_allocations[wildcard_device] += remaining_model_bytes
logger.info(f"[MultiGPU_DisTorch2] Assigning remaining {remaining_model_bytes / (1024**2):.2f}MB of model to wildcard device '{wildcard_device}'.")
logger.info(f"[MultiGPU DisTorch V2] Assigning remaining {remaining_model_bytes / (1024**2):.2f}MB of model to wildcard device '{wildcard_device}'.")
# Convert the final byte allocations to VRAM fractions
allocation_parts = []
@@ -661,10 +536,7 @@ def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
return allocations_string
def calculate_fraction_from_ratio_expert_string(model_patcher, ratio_str):
"""
Converts a user-provided ratio string (which describes how to split the MODEL)
into a fraction string (which describes the fraction of DEVICE VRAM to use).
"""
"""Convert ratio allocation string (e.g. 'cuda:0,25%;cpu,75%') describing model split to fractional VRAM allocation string."""
raw_block_list = model_patcher._load_list()
total_model_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
@@ -704,7 +576,7 @@ def calculate_fraction_from_ratio_expert_string(model_patcher, ratio_str):
else:
put_part = ", ".join(put_parts[:-1]) + f", and {put_parts[-1]}"
logger.info(f"[MultiGPU_DisTorch2] Ratio(%) Mode - {ratio_str} -> {ratio_string} ratio, put {put_part}")
logger.info(f"[MultiGPU DisTorch V2] Ratio(%) Mode - {ratio_str} -> {ratio_string} ratio, put {put_part}")
allocations_string = ";".join(allocation_parts)
@@ -772,8 +644,8 @@ def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
# Warning if model too large
if model_size_gb > (recipient_vram * 0.9):
required_offload_gb = model_size_gb - (recipient_vram * 0.9)
logger.warning(f"[MultiGPU] WARNING: Model size ({model_size_gb:.2f}GB) is larger than 90% of available VRAM on {recipient_device} ({recipient_vram * 0.9:.2f}GB).")
logger.warning(f"[MultiGPU] To prevent an OOM error, set 'virtual_vram_gb' to at least {required_offload_gb:.2f}.")
logger.warning(f"\n\n[MultiGPU DisTorch V2] Model size ({model_size_gb:.2f}GB) is larger than 90% of available VRAM on: {recipient_device} ({recipient_vram * 0.9:.2f}GB).")
logger.warning(f"[MultiGPU DisTorch V2] To prevent an OOM error, set 'virtual_vram_gb' to at least {required_offload_gb:.2f}.\n\n")
new_on_recipient = max(0, model_size_gb - virtual_vram_gb)
@@ -789,294 +661,3 @@ def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
allocations_string = ";".join(allocation_parts)
return allocations_string
def override_class_with_distorch_safetensor_v2(cls):
"""DisTorch 2.0 wrapper for safetensor models"""
from . import current_device
class NodeOverrideDisTorchSafetensorV2(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
compute_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
@classmethod
def IS_CHANGED(s, *args, compute_device=None, virtual_vram_gb=4.0,
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
# Create a hash of our specific settings
settings_str = f"{compute_device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}"
return hashlib.sha256(settings_str.encode()).hexdigest()
def override(self, *args, compute_device=None, virtual_vram_gb=4.0,
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
from . import set_current_device
if compute_device is not None:
set_current_device(compute_device)
# Register our patched ModelPatcher
register_patched_safetensor_modelpatcher()
# Call original function
fn = getattr(super(), cls.FUNCTION)
# --- Check if we need to unload the model due to settings change ---
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
settings_str = f"{compute_device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}"
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
# Temporarily load to get hash without applying our patch
temp_out = fn(*args, **kwargs)
model_to_check = None
if hasattr(temp_out[0], 'model'):
model_to_check = temp_out[0]
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
model_to_check = temp_out[0].patcher
if model_to_check:
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
last_settings_hash = safetensor_settings_store.get(model_hash)
if last_settings_hash != settings_hash:
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
else:
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
out = fn(*args, **kwargs)
# Store high_precision_loras in the model for later retrieval
if hasattr(out[0], 'model'):
out[0].model._distorch_high_precision_loras = high_precision_loras
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
elif expert_mode_allocations: # Only include compute device if there's an expert string
vram_string = compute_device
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
return out
return NodeOverrideDisTorchSafetensorV2
def override_class_with_distorch_safetensor_v2_clip(cls):
"""DisTorch 2.0 wrapper for safetensor CLIP models"""
from . import current_device
class NodeOverrideDisTorchSafetensorV2Clip(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device}) # Changed from compute_device
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
@classmethod
def IS_CHANGED(s, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
# Create a hash of our specific settings
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}" # Changed from compute_device
return hashlib.sha256(settings_str.encode()).hexdigest()
def override(self, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
from . import set_current_text_encoder_device # Use text encoder device setter
if device is not None:
set_current_text_encoder_device(device)
kwargs['device'] = 'default' # Hardcode device setting like in standard clip wrapper
# Register our patched ModelPatcher
register_patched_safetensor_modelpatcher()
# Call original function
fn = getattr(super(), cls.FUNCTION)
# --- Check if we need to unload the model due to settings change ---
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}" # Changed from compute_device
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
# Temporarily load to get hash without applying our patch
temp_out = fn(*args, **kwargs)
model_to_check = None
if hasattr(temp_out[0], 'model'):
model_to_check = temp_out[0]
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
model_to_check = temp_out[0].patcher
if model_to_check:
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
last_settings_hash = safetensor_settings_store.get(model_hash)
if last_settings_hash != settings_hash:
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
else:
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
out = fn(*args, **kwargs)
# Store high_precision_loras in the model for later retrieval
if hasattr(out[0], 'model'):
out[0].model._distorch_high_precision_loras = high_precision_loras
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{device};{virtual_vram_gb};{donor_device}" # Changed from compute_device
elif expert_mode_allocations: # Only include device if there's an expert string
vram_string = device # Changed from compute_device
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
return out
return NodeOverrideDisTorchSafetensorV2Clip
def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
"""DisTorch 2.0 wrapper for safetensor CLIP models"""
from . import current_device
class NodeOverrideDisTorchSafetensorV2ClipNoDevice(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device}) # Changed from compute_device
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
@classmethod
def IS_CHANGED(s, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
# Create a hash of our specific settings
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}" # Changed from compute_device
return hashlib.sha256(settings_str.encode()).hexdigest()
def override(self, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
from . import set_current_text_encoder_device # Use text encoder device setter
if device is not None:
set_current_text_encoder_device(device)
# Register our patched ModelPatcher
register_patched_safetensor_modelpatcher()
# Call original function
fn = getattr(super(), cls.FUNCTION)
# --- Check if we need to unload the model due to settings change ---
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}" # Changed from compute_device
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
# Temporarily load to get hash without applying our patch
temp_out = fn(*args, **kwargs)
model_to_check = None
if hasattr(temp_out[0], 'model'):
model_to_check = temp_out[0]
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
model_to_check = temp_out[0].patcher
if model_to_check:
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
last_settings_hash = safetensor_settings_store.get(model_hash)
if last_settings_hash != settings_hash:
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
else:
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
out = fn(*args, **kwargs)
# Store high_precision_loras in the model for later retrieval
if hasattr(out[0], 'model'):
out[0].model._distorch_high_precision_loras = high_precision_loras
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{device};{virtual_vram_gb};{donor_device}" # Changed from compute_device
elif expert_mode_allocations: # Only include device if there's an expert string
vram_string = device # Changed from compute_device
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
return out
return NodeOverrideDisTorchSafetensorV2ClipNoDevice
@@ -1,920 +0,0 @@
{
"last_node_id": 115,
"last_link_id": 277,
"nodes": [
{
"id": 13,
"type": "SamplerCustomAdvanced",
"pos": [
815.8301391601562,
241.12867736816406
],
"size": [
292.4319763183594,
479.03521728515625
],
"flags": {
"collapsed": false
},
"order": 14,
"mode": 0,
"inputs": [
{
"name": "noise",
"type": "NOISE",
"link": 37,
"slot_index": 0
},
{
"name": "guider",
"type": "GUIDER",
"link": 30,
"slot_index": 1
},
{
"name": "sampler",
"type": "SAMPLER",
"link": 19,
"slot_index": 2
},
{
"name": "sigmas",
"type": "SIGMAS",
"link": 20,
"slot_index": 3
},
{
"name": "latent_image",
"type": "LATENT",
"link": 180,
"slot_index": 4
}
],
"outputs": [
{
"name": "output",
"type": "LATENT",
"links": [
210
],
"slot_index": 0,
"shape": 3
},
{
"name": "denoised_output",
"type": "LATENT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "SamplerCustomAdvanced"
},
"widgets_values": []
},
{
"id": 111,
"type": "DownloadAndLoadHyVideoTextEncoderMultiGPU",
"pos": [
-821.001220703125,
504.2577209472656
],
"size": [
371.9022521972656,
202
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "hyvid_text_encoder",
"type": "HYVIDTEXTENCODER",
"links": [
269
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DownloadAndLoadHyVideoTextEncoderMultiGPU"
},
"widgets_values": [
"xtuner/llava-llama-3-8b-v1_1-transformers",
"openai/clip-vit-large-patch14",
"bf16",
false,
2,
"disabled",
"cpu"
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 88,
"type": "VAELoaderMultiGPU",
"pos": [
-805.2030639648438,
373.4107360839844
],
"size": [
322.5263366699219,
82
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "VAE",
"type": "VAE",
"links": [
275
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAELoaderMultiGPU"
},
"widgets_values": [
"hunyuan_video_vae_bf16.safetensors",
"cuda:1"
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 109,
"type": "HyVideoTextImageEncode",
"pos": [
-374.6673278808594,
532.1463012695312
],
"size": [
295.6000061035156,
452.87860107421875
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "text_encoders",
"type": "HYVIDTEXTENCODER",
"link": 269
},
{
"name": "custom_prompt_template",
"type": "PROMPT_TEMPLATE",
"link": null,
"shape": 7
},
{
"name": "clip_l",
"type": "CLIP",
"link": null,
"shape": 7
},
{
"name": "image1",
"type": "IMAGE",
"link": 272,
"shape": 7
},
{
"name": "image2",
"type": "IMAGE",
"link": null,
"shape": 7
},
{
"name": "hyvid_cfg",
"type": "HYVID_CFG",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "hyvid_embeds",
"type": "HYVIDEMBEDS",
"links": [
270
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "HyVideoTextImageEncode"
},
"widgets_values": [
"The animal shown in <image> appears within its own natural setting, moving calmly or resting in place as a soft light casts delicate shadows across its form. Over the course of five seconds, it makes subtle shifts in posture or position, revealing small details of its features, such as the texture of its skin or fur, and the quiet rhythm of its breathing.",
"::4",
false,
"video",
""
]
},
{
"id": 112,
"type": "LoadImage",
"pos": [
-787.3131713867188,
795.4229736328125
],
"size": [
315,
314
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
272
],
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"pasted/image (224).png",
"image"
]
},
{
"id": 103,
"type": "Note",
"pos": [
-835.8104248046875,
-79.16381072998047
],
"size": [
353.56494140625,
190.77996826171875
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [],
"title": "This workflow requires ComfyUI-GGUF",
"properties": {},
"widgets_values": [
"**⚠️ Dependency Alert! ⚠️**\n\nThis workflow relies on nodes from the [ComfyUI-GGUF](https://github.com/city96/ComfyUI-GGUF) custom node repository to function correctly. \n\nSpecifically:\n\n*\"CLIPLoaderGGUFMultiGPU\" \n\nwill not work without this dependency installed. Please install ComfyUI-GGUF before attempting to run this workflow."
],
"color": "#332922",
"bgcolor": "#593930"
},
{
"id": 67,
"type": "ModelSamplingSD3",
"pos": [
-364.1572265625,
168.46791076660156
],
"size": [
210,
58
],
"flags": {
"collapsed": true
},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 276
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
252
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ModelSamplingSD3"
},
"widgets_values": [
7
]
},
{
"id": 17,
"type": "BasicScheduler",
"pos": [
-367.9955749511719,
236.91629028320312
],
"size": [
210,
109.8011474609375
],
"flags": {
"collapsed": true
},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 277,
"slot_index": 0
}
],
"outputs": [
{
"name": "SIGMAS",
"type": "SIGMAS",
"links": [
20
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "BasicScheduler"
},
"widgets_values": [
"simple",
20,
1
]
},
{
"id": 113,
"type": "HunyuanVideoEmbeddingsAdapter",
"pos": [
-369.37286376953125,
363.2193603515625
],
"size": [
283.43841552734375,
34.09494400024414
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "hyvid_embeds",
"type": "HYVIDEMBEDS",
"link": 270
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
271
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "HunyuanVideoEmbeddingsAdapter"
},
"widgets_values": []
},
{
"id": 26,
"type": "FluxGuidance",
"pos": [
-25.9213809967041,
325.2367858886719
],
"size": [
211.60000610351562,
58
],
"flags": {
"collapsed": true
},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "conditioning",
"type": "CONDITIONING",
"link": 271
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
129
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "FluxGuidance"
},
"widgets_values": [
6
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 22,
"type": "BasicGuider",
"pos": [
206.57337951660156,
209.95970153808594
],
"size": [
222.3482666015625,
46
],
"flags": {
"collapsed": true
},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 252,
"slot_index": 0
},
{
"name": "conditioning",
"type": "CONDITIONING",
"link": 129,
"slot_index": 1
}
],
"outputs": [
{
"name": "GUIDER",
"type": "GUIDER",
"links": [
30
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "BasicGuider"
},
"widgets_values": []
},
{
"id": 16,
"type": "KSamplerSelect",
"pos": [
334.6855163574219,
327.40887451171875
],
"size": [
210,
58
],
"flags": {
"collapsed": true
},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "SAMPLER",
"type": "SAMPLER",
"links": [
19
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "KSamplerSelect"
},
"widgets_values": [
"euler"
]
},
{
"id": 45,
"type": "EmptyHunyuanLatentVideo",
"pos": [
11.676980018615723,
448.76055908203125
],
"size": [
210,
130
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
180
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EmptyHunyuanLatentVideo"
},
"widgets_values": [
848,
480,
73,
1
]
},
{
"id": 73,
"type": "VAEDecodeTiled",
"pos": [
503.8817138671875,
509.217529296875
],
"size": [
210,
150
],
"flags": {},
"order": 15,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 210
},
{
"name": "vae",
"type": "VAE",
"link": 275
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
268
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecodeTiled"
},
"widgets_values": [
256,
64,
64,
8
]
},
{
"id": 102,
"type": "VHS_VideoCombine",
"pos": [
9.632485389709473,
698.3294677734375
],
"size": [
451.07391357421875,
334
],
"flags": {},
"order": 16,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 268
},
{
"name": "audio",
"type": "AUDIO",
"link": null,
"shape": 7
},
{
"name": "meta_batch",
"type": "VHS_BatchManager",
"link": null,
"shape": 7
},
{
"name": "vae",
"type": "VAE",
"link": null,
"shape": 7
}
],
"outputs": [
{
"name": "Filenames",
"type": "VHS_FILENAMES",
"links": null
}
],
"properties": {
"Node name for S&R": "VHS_VideoCombine"
},
"widgets_values": {
"frame_rate": 24,
"loop_count": 0,
"filename_prefix": "HunyuanVideo",
"format": "video/h264-mp4",
"pix_fmt": "yuv420p",
"crf": 19,
"save_metadata": true,
"trim_to_audio": false,
"pingpong": false,
"save_output": true,
"videopreview": {
"hidden": false,
"paused": false,
"params": {
"filename": "HunyuanVideo_00323.mp4",
"subfolder": "",
"type": "output",
"format": "video/h264-mp4",
"frame_rate": 24,
"workflow": "HunyuanVideo_00323.png",
"fullpath": "/home/johnj/ComfyUI/output/HunyuanVideo_00323.mp4"
},
"muted": false
}
}
},
{
"id": 115,
"type": "UnetLoaderGGUFDisTorchMultiGPU",
"pos": [
-810.4765014648438,
219.3965301513672
],
"size": [
342.1245422363281,
154
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
276,
277
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "UnetLoaderGGUFDisTorchMultiGPU"
},
"widgets_values": [
"hunyuan-video-t2v-720p-Q4_K_M.gguf",
"cuda:0",
4,
false,
""
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 25,
"type": "RandomNoise",
"pos": [
368.4217834472656,
72.61695861816406
],
"size": [
250.37998962402344,
82
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "NOISE",
"type": "NOISE",
"links": [
37
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "RandomNoise"
},
"widgets_values": [
5770521,
"fixed"
],
"color": "#2a363b",
"bgcolor": "#3f5159"
}
],
"links": [
[
19,
16,
0,
13,
2,
"SAMPLER"
],
[
20,
17,
0,
13,
3,
"SIGMAS"
],
[
30,
22,
0,
13,
1,
"GUIDER"
],
[
37,
25,
0,
13,
0,
"NOISE"
],
[
129,
26,
0,
22,
1,
"CONDITIONING"
],
[
180,
45,
0,
13,
4,
"LATENT"
],
[
210,
13,
0,
73,
0,
"LATENT"
],
[
252,
67,
0,
22,
0,
"MODEL"
],
[
268,
73,
0,
102,
0,
"IMAGE"
],
[
269,
111,
0,
109,
0,
"HYVIDTEXTENCODER"
],
[
270,
109,
0,
113,
0,
"HYVIDEMBEDS"
],
[
271,
113,
0,
26,
0,
"CONDITIONING"
],
[
272,
112,
0,
109,
3,
"IMAGE"
],
[
275,
88,
0,
73,
1,
"VAE"
],
[
276,
115,
0,
67,
0,
"MODEL"
],
[
277,
115,
0,
17,
0,
"MODEL"
]
],
"groups": [
{
"id": 2,
"title": "GGUFMultiGPU",
"bounding": [
-836.7138671875,
144.54360961914062,
403.62188720703125,
584.6732788085938
],
"color": "#8AA",
"font_size": 24,
"flags": {}
}
],
"config": {},
"extra": {
"ds": {
"scale": 1,
"offset": {
"0": 1075.0079345703125,
"1": 156.88380432128906
}
},
"groupNodes": {},
"ue_links": [],
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}
+364
View File
@@ -0,0 +1,364 @@
"""
Model Management Extensions for MultiGPU
Extends ComfyUI's model management with multi-device capabilities and lifecycle tracking.
"""
import torch
import logging
import hashlib
import psutil
import comfy.model_management as mm
import gc
from datetime import datetime, timezone
import server
import weakref
import platform
import ctypes
import comfy.model_patcher
from collections import defaultdict
logger = logging.getLogger("MultiGPU")
# ==========================================================================================
# GC Anchor System for Model Retention
# ==========================================================================================
# Global anchor set to prevent GC of models during selective unload
_MGPU_RETENTION_ANCHORS = set()
def add_retention_anchor(model_patcher, reason="keep_loaded"):
"""Add a model patcher to the GC anchor set to prevent premature garbage collection"""
if model_patcher is not None:
_MGPU_RETENTION_ANCHORS.add(model_patcher)
model_name = type(getattr(model_patcher, 'model', model_patcher)).__name__
logger.mgpu_mm_log(f"[GC_ANCHOR] Added retention anchor for {model_name}, reason: {reason}, total anchors: {len(_MGPU_RETENTION_ANCHORS)}")
def clear_all_retention_anchors(reason="manual_clear"):
"""Clear all retention anchors"""
count = len(_MGPU_RETENTION_ANCHORS)
_MGPU_RETENTION_ANCHORS.clear()
logger.mgpu_mm_log(f"[GC_ANCHOR] Cleared all {count} retention anchors, reason: {reason}")
# ==========================================================================================
# Model Analysis and Store Management (DisTorch V1 & V2)
# ==========================================================================================
# DisTorch V2 SafeTensor stores
safetensor_allocation_store = {}
safetensor_settings_store = {}
# DisTorch V1 GGUF stores (backwards compatibility)
model_allocation_store = {}
def create_safetensor_model_hash(model, caller):
"""Create a unique hash for a safetensor model to track allocations"""
if hasattr(model, 'model'):
actual_model = model.model
model_type = type(actual_model).__name__
model_size = model.model_size() if hasattr(model, 'model_size') else sum(p.numel() * p.element_size() for p in actual_model.parameters())
first_layers = str(list(model.model_state_dict().keys() if hasattr(model, 'model_state_dict') else actual_model.state_dict().keys())[:3])
else:
model_type = type(model).__name__
model_size = sum(p.numel() * p.element_size() for p in model.parameters())
first_layers = str(list(model.state_dict().keys())[:3])
identifier = f"{model_type}_{model_size}_{first_layers}"
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
logger.debug(f"[MultiGPU DisTorch V2] Created hash for {caller}: {final_hash[:8]}...")
return final_hash
def create_model_hash(model, caller):
"""Create a unique hash for a GGUF model to track allocations (DisTorch V1)"""
model_type = type(model.model).__name__
model_size = model.model_size()
first_layers = str(list(model.model_state_dict().keys())[:3])
identifier = f"{model_type}_{model_size}_{first_layers}"
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
logger.debug(f"[MultiGPU_DisTorch_HASH] Created hash for {caller}: {final_hash[:8]}...")
return final_hash
# ==========================================================================================
# Memory Logging Infrastructure
# ==========================================================================================
_MEM_SNAPSHOT_LAST = {}
_MEM_SNAPSHOT_SERIES = {}
def _capture_memory_snapshot():
"""Capture memory snapshot for CPU and all devices"""
# Import here to avoid circular dependency
from .device_utils import get_device_list
snapshot = {}
# CPU
vm = psutil.virtual_memory()
snapshot["cpu"] = (vm.used, vm.total)
# GPU devices
devices = [d for d in get_device_list() if d != "cpu"]
for dev_str in devices:
device = torch.device(dev_str)
total = mm.get_total_memory(device)
free_info = mm.get_free_memory(device, torch_free_too=True)
system_free = free_info[0] if isinstance(free_info, tuple) else free_info
used = max(0, total - system_free)
snapshot[dev_str] = (used, total)
return snapshot
def multigpu_memory_log(identifier, tag):
"""Record timestamped memory snapshot with clean aligned logging"""
if identifier == "print_summary":
for id_key in sorted(_MEM_SNAPSHOT_SERIES.keys()):
series = _MEM_SNAPSHOT_SERIES[id_key]
logger.mgpu_mm_log(f"=== memory summary: {id_key} ===")
for ts, tag_name, snap in series:
parts = []
cpu_used, cpu_total = snap.get("cpu", (0, 0))
parts.append(f"cpu|{cpu_used/(1024**3):.2f}")
for dev in sorted([k for k in snap.keys() if k != "cpu"]):
used, total = snap[dev]
parts.append(f"{dev}|{used/(1024**3):.2f}")
ts_str = ts.strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
tag_padded = f"{id_key}_{tag_name}".ljust(35)
logger.mgpu_mm_log(f"{ts_str} {tag_padded} {' '.join(parts)}")
return
ts = datetime.now(timezone.utc)
curr = _capture_memory_snapshot()
# Store in series
if identifier not in _MEM_SNAPSHOT_SERIES:
_MEM_SNAPSHOT_SERIES[identifier] = []
_MEM_SNAPSHOT_SERIES[identifier].append((ts, tag, curr))
# Clean aligned format: timestamp + padded tag + memory values
ts_str = ts.strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
tag_padded = f"{identifier}_{tag}".ljust(35)
parts = []
cpu_used, _ = curr.get("cpu", (0, 0))
parts.append(f"cpu|{cpu_used/(1024**3):.2f}")
for dev in sorted([k for k in curr.keys() if k != "cpu"]):
used, _ = curr[dev]
parts.append(f"{dev}|{used/(1024**3):.2f}")
logger.mgpu_mm_log(f"{ts_str} {tag_padded} {' '.join(parts)}")
_MEM_SNAPSHOT_LAST[identifier] = (tag, curr)
# ==========================================================================================
# Memory Management and Cleanup
# ==========================================================================================
CPU_MEMORY_THRESHOLD_PERCENT = 85.0
CPU_RESET_HYSTERESIS_PERCENT = 5.0
_last_cpu_usage_at_reset = 0.0
def trigger_executor_cache_reset(reason="policy", force=False):
"""Trigger PromptExecutor.reset() by setting 'free_memory' flag"""
global _last_cpu_usage_at_reset
prompt_server = server.PromptServer.instance
if prompt_server is None:
logger.debug("[MultiGPU_Memory_Management] PromptServer not initialized")
return
if prompt_server.prompt_queue.currently_running and not force:
logger.debug(f"[MultiGPU_Memory_Management] Skipping reset during execution (reason: {reason})")
return
multigpu_memory_log("executor_reset", f"pre-trigger ({reason})")
logger.info(f"[MultiGPU_Memory_Management] Triggering PromptExecutor cache reset. Reason: {reason}")
prompt_server.prompt_queue.set_flag("free_memory", True)
logger.debug("[MultiGPU_Memory_Management] 'free_memory' flag set")
vm = psutil.virtual_memory()
_last_cpu_usage_at_reset = vm.percent
multigpu_memory_log("executor_reset", f"post-trigger ({reason})")
def check_cpu_memory_threshold(threshold_percent=CPU_MEMORY_THRESHOLD_PERCENT):
"""Check CPU memory and trigger reset if threshold exceeded"""
if server.PromptServer.instance is None:
return
if server.PromptServer.instance.prompt_queue.currently_running:
return
vm = psutil.virtual_memory()
current_usage = vm.percent
if current_usage > threshold_percent:
if current_usage > (_last_cpu_usage_at_reset + CPU_RESET_HYSTERESIS_PERCENT):
logger.warning(f"[MultiGPU_Memory_Monitor] CPU usage ({current_usage:.1f}%) exceeds threshold ({threshold_percent:.1f}%)")
multigpu_memory_log("cpu_monitor", f"trigger:{current_usage:.1f}pct")
trigger_executor_cache_reset(reason="cpu_threshold_exceeded", force=False)
else:
logger.debug(f"[MultiGPU_Memory_Monitor] CPU usage high ({current_usage:.1f}%) but within hysteresis")
multigpu_memory_log("cpu_monitor", f"skip_hysteresis:{current_usage:.1f}pct")
def force_full_system_cleanup(reason="manual", force=True):
"""Mirror ComfyUI-Manager 'Free model and node cache' by setting unload_models=True and free_memory=True flags."""
vm = psutil.virtual_memory()
pre_cpu = vm.used
pre_models = len(mm.current_loaded_models)
multigpu_memory_log("full_cleanup", f"start:{reason}")
logger.mgpu_mm_log(f"[ManagerMatch] Requesting cleanup (reason={reason}) | pre_models={pre_models}, cpu_used_gib={pre_cpu/(1024**3):.2f}")
if server.PromptServer.instance is not None:
pq = server.PromptServer.instance.prompt_queue
if (not pq.currently_running) or force:
pq.set_flag("unload_models", True)
pq.set_flag("free_memory", True)
logger.mgpu_mm_log("[ManagerMatch] Flags set: unload_models=True, free_memory=True")
else:
logger.mgpu_mm_log("[ManagerMatch] Skipped - execution active and force=False")
vm = psutil.virtual_memory()
post_cpu = vm.used
post_models = len(mm.current_loaded_models)
delta_cpu_mb = (post_cpu - pre_cpu) / (1024**2)
multigpu_memory_log("full_cleanup", f"requested:{reason}")
summary = f"[ManagerMatch] Cleanup requested (reason={reason}) | models {pre_models}->{post_models}, cpu_delta_mb={delta_cpu_mb:.2f}"
logger.mgpu_mm_log(summary)
return summary
# ==========================================================================================
# Core Patching: unload_all_models
# ==========================================================================================
if not hasattr(mm.unload_all_models, '_mgpu_eject_distorch_patched'):
logger.info("[MultiGPU Core Patching] Patching mm.unload_all_models for DisTorch2 ejection support")
_mgpu_original_unload_all_models = mm.unload_all_models
def _mgpu_patched_unload_all_models():
"""Patched mm.unload_all_models with selective ejection support and comprehensive diagnostics."""
logger.mgpu_mm_log(f"[UNLOAD_START] Patched unload_all_models called - initial model count: {len(mm.current_loaded_models)}")
# Check if there are any DisTorch models that want to be unloaded
has_distorch_to_unload = any(
(hasattr(lm.model, '_mgpu_unload_distorch_model') and lm.model._mgpu_unload_distorch_model) or
(hasattr(getattr(lm.model, 'model', None), '_mgpu_unload_distorch_model') and lm.model.model._mgpu_unload_distorch_model)
for lm in mm.current_loaded_models
if lm.model is not None
)
if not has_distorch_to_unload:
logger.mgpu_mm_log("No DisTorch models requesting unload - clearing anchors and delegating to original unload_all_models")
clear_all_retention_anchors(reason="no_selective_unload_needed")
_mgpu_original_unload_all_models()
return
# Direct approach: iterate through loaded models and selectively unload
models_to_unload = []
kept_models = []
for i, lm in enumerate(mm.current_loaded_models):
mp = lm.model # weakref call to ModelPatcher
# DIAGNOSTIC: Log full object chain
lm_id = id(lm)
mp_id = id(mp)
inner_model = getattr(mp, 'model', None)
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
# Format inner_model_id properly for f-string
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_READ] Model {i}: lm_id=0x{lm_id:x}, mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
# FIX: Check flag on ModelPatcher (where it was set), not on inner model
# OLD BUG: unload_distorch_model = getattr(mp.model, '_mgpu_unload_distorch_model', False)
# NEW FIX: Check both locations to see which one has the flag
flag_on_mp = getattr(mp, '_mgpu_unload_distorch_model', None)
flag_on_inner = getattr(mp.model, '_mgpu_unload_distorch_model', None) if inner_model else None
logger.mgpu_mm_log(f"[FLAG_CHECK] Model {i} ({inner_model_name}): flag_on_mp={flag_on_mp}, flag_on_inner={flag_on_inner}")
# Use whichever location has the flag (for backwards compatibility during transition)
if flag_on_mp is not None:
unload_distorch_model = flag_on_mp
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from ModelPatcher (mp_id=0x{mp_id:x})")
elif flag_on_inner is not None:
unload_distorch_model = flag_on_inner
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from inner model (inner_model_id={inner_id_str})")
else:
unload_distorch_model = False
logger.mgpu_mm_log(f"[FLAG_SOURCE] No flag found - defaulting to False (keep loaded)")
logger.mgpu_mm_log(f"[DECISION] Model {i} ({inner_model_name}): unload_distorch_model={unload_distorch_model}")
if unload_distorch_model:
models_to_unload.append(lm)
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → models_to_unload")
else:
kept_models.append(lm)
add_retention_anchor(mp, "keep_loaded_protection")
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → kept_models")
# After the kept_models/models_to_unload evaluation
logger.mgpu_mm_log(f"[CATEGORIZE_SUMMARY] kept_models: {len(kept_models)}, models_to_unload: {len(models_to_unload)}, total: {len(mm.current_loaded_models)}")
if len(kept_models) == len(mm.current_loaded_models):
# All models are meant to be kept - no DisTorch selective unloading needed
logger.mgpu_mm_log("[DELEGATION] All models flagged to be kept - delegating to standard unload_all_models")
_mgpu_original_unload_all_models()
return
if kept_models:
logger.mgpu_mm_log(f"[SELECTIVE_UNLOAD] Proceeding with selective unload: retaining {len(kept_models)}, unloading {len(models_to_unload)}")
# Unload models flagged for unload
for lm in models_to_unload:
try:
model_name = type(lm.model.model).__name__ if lm.model and hasattr(lm.model, 'model') else 'Unknown'
logger.mgpu_mm_log(f"[UNLOAD_EXECUTE] Unloading model: {model_name} (lm_id=0x{id(lm):x})")
lm.model_unload(unpatch_weights=True)
except Exception as e:
logger.warning(f"[UNLOAD_ERROR] Error unloading model: {e}")
# WEAKREF TRACKING: Attach weakref callbacks to prove if kept models are GC'd
def model_deleted_callback(ref, model_name, model_id):
logger.mgpu_mm_log(f"[WEAKREF_DELETED] Kept model GARBAGE COLLECTED: {model_name} (id=0x{model_id:x})")
for i, lm in enumerate(kept_models):
mp = lm.model
inner_model = getattr(mp, 'model', None)
model_name = type(inner_model).__name__ if inner_model else 'Unknown'
model_id = id(lm)
weakref.ref(lm, lambda ref, name=model_name, mid=model_id: model_deleted_callback(ref, name, mid))
logger.mgpu_mm_log(f"[WEAKREF_ATTACHED] Tracking kept model {i}: {model_name} (lm_id=0x{model_id:x}, mp_id=0x{id(mp):x})")
# Remove unloaded models from current_loaded_models
mm.current_loaded_models = kept_models
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] Updated mm.current_loaded_models, new count: {len(mm.current_loaded_models)}")
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] mm.current_loaded_models id: 0x{id(mm.current_loaded_models):x}")
# DIAGNOSTIC: Log what's remaining
for i, lm in enumerate(mm.current_loaded_models):
mp = lm.model
inner_model = getattr(mp, 'model', None)
model_name = type(inner_model).__name__ if inner_model else "None"
logger.mgpu_mm_log(f"[REMAINING_MODEL] {i}: {model_name} (lm_id=0x{id(lm):x}, mp_id=0x{id(mp):x})")
else:
logger.mgpu_mm_log("[DELEGATION] No models with keep_loaded=True found - delegating to original unload_all_models")
_mgpu_original_unload_all_models()
mm.unload_all_models = _mgpu_patched_unload_all_models
mm.unload_all_models._mgpu_eject_distorch_patched = True
logger.info("[MultiGPU Core Patching] mm.unload_all_models patched successfully")
else:
logger.debug("[MultiGPU Core Patching] mm.unload_all_models already patched - skipping")
+52
View File
@@ -3,6 +3,7 @@ import folder_paths
from pathlib import Path
from nodes import NODE_CLASS_MAPPINGS
from .device_utils import get_device_list
from .model_management_mgpu import force_full_system_cleanup
class DeviceSelectorMultiGPU:
@classmethod
@@ -20,6 +21,7 @@ class DeviceSelectorMultiGPU:
CATEGORY = "multigpu"
def select_device(self, device):
"""Select target device from available device list."""
return (device,)
@@ -37,6 +39,7 @@ class HunyuanVideoEmbeddingsAdapter:
CATEGORY = "multigpu"
def adapt_embeddings(self, hyvid_embeds):
"""Adapt HunyuanVideo embeddings to standard ComfyUI conditioning format."""
cond = hyvid_embeds["prompt_embeds"]
pooled_dict = {
@@ -72,6 +75,7 @@ class UnetLoaderGGUF:
TITLE = "Unet Loader (GGUF)"
def load_unet(self, unet_name, dequant_dtype=None, patch_dtype=None, patch_on_device=None):
"""Load GGUF format UNet model."""
original_loader = NODE_CLASS_MAPPINGS["UnetLoaderGGUF"]()
return original_loader.load_unet(unet_name, dequant_dtype, patch_dtype, patch_on_device)
@@ -109,20 +113,24 @@ class CLIPLoaderGGUF:
@classmethod
def get_filename_list(s):
"""Get combined list of CLIP and CLIP_GGUF model files."""
files = []
files += folder_paths.get_filename_list("clip")
files += folder_paths.get_filename_list("clip_gguf")
return sorted(files)
def load_data(self, ckpt_paths):
"""Load CLIP model data from checkpoint paths."""
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
return original_loader.load_data(ckpt_paths)
def load_patcher(self, clip_paths, clip_type, clip_data):
"""Create ModelPatcher for CLIP model."""
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
return original_loader.load_patcher(clip_paths, clip_type, clip_data)
def load_clip(self, clip_name, type="stable_diffusion", device=None):
"""Load CLIP model from GGUF or standard format."""
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
return original_loader.load_clip(clip_name, type)
@@ -143,6 +151,7 @@ class DualCLIPLoaderGGUF(CLIPLoaderGGUF):
TITLE = "DualCLIPLoader (GGUF)"
def load_clip(self, clip_name1, clip_name2, type, device=None):
"""Load dual CLIP model configuration."""
original_loader = NODE_CLASS_MAPPINGS["DualCLIPLoaderGGUF"]()
clip = original_loader.load_clip(clip_name1, clip_name2, type)
clip[0].patcher.load(force_patch_weights=True)
@@ -164,6 +173,7 @@ class TripleCLIPLoaderGGUF(CLIPLoaderGGUF):
TITLE = "TripleCLIPLoader (GGUF)"
def load_clip(self, clip_name1, clip_name2, clip_name3, type="sd3"):
"""Load triple CLIP model configuration for SD3."""
original_loader = NODE_CLASS_MAPPINGS["TripleCLIPLoaderGGUF"]()
return original_loader.load_clip(clip_name1, clip_name2, clip_name3, type)
@@ -183,6 +193,7 @@ class QuadrupleCLIPLoaderGGUF(CLIPLoaderGGUF):
TITLE = "QuadrupleCLIPLoader (GGUF)"
def load_clip(self, clip_name1, clip_name2, clip_name3, clip_name4, type="stable_diffusion"):
"""Load quadruple CLIP model configuration."""
original_loader = NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderGGUF"]()
return original_loader.load_clip(clip_name1, clip_name2, clip_name3, clip_name4, type)
@@ -206,12 +217,15 @@ class LTXVLoader:
OUTPUT_NODE = False
def load(self, ckpt_name, dtype):
"""Load LTXV model and VAE with specified precision."""
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
return original_loader.load(ckpt_name, dtype)
def _load_unet(self, load_device, offload_device, weights, num_latent_channels, dtype, config=None ):
"""Load LTXV UNet with device-specific configuration."""
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
return original_loader._load_unet(load_device, offload_device, weights, num_latent_channels, dtype, config=None )
def _load_vae(self, weights, config=None):
"""Load LTXV VAE from weights."""
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
return original_loader._load_vae(weights, config=None)
@@ -238,6 +252,7 @@ class Florence2ModelLoader:
CATEGORY = "Florence2"
def loadmodel(self, model, precision, attention, lora=None):
"""Load Florence2 vision model with specified precision and attention mode."""
original_loader = NODE_CLASS_MAPPINGS["Florence2ModelLoader"]()
return original_loader.loadmodel(model, precision, attention, lora)
@@ -285,6 +300,7 @@ class DownloadAndLoadFlorence2Model:
CATEGORY = "Florence2"
def loadmodel(self, model, precision, attention, lora=None):
"""Download and load Florence2 model from HuggingFace."""
original_loader = NODE_CLASS_MAPPINGS["DownloadAndLoadFlorence2Model"]()
return original_loader.loadmodel(model, precision, attention, lora)
@@ -300,6 +316,7 @@ class CheckpointLoaderNF4:
def load_checkpoint(self, ckpt_name):
"""Load checkpoint in NF4 quantized format."""
original_loader = NODE_CLASS_MAPPINGS["CheckpointLoaderNF4"]()
return original_loader.load_checkpoint(ckpt_name)
@@ -316,6 +333,7 @@ class LoadFluxControlNet:
CATEGORY = "XLabsNodes"
def loadmodel(self, model_name, controlnet_path):
"""Load Flux ControlNet model."""
original_loader = NODE_CLASS_MAPPINGS["LoadFluxControlNet"]()
return original_loader.loadmodel(model_name, controlnet_path)
@@ -336,6 +354,7 @@ class MMAudioModelLoader:
CATEGORY = "MMAudio"
def loadmodel(self, mmaudio_model, base_precision):
"""Load MMAudio model with specified precision."""
original_loader = NODE_CLASS_MAPPINGS["MMAudioModelLoader"]()
return original_loader.loadmodel(mmaudio_model, base_precision)
@@ -363,6 +382,7 @@ class MMAudioFeatureUtilsLoader:
CATEGORY = "MMAudio"
def loadmodel(self, vae_model, precision, synchformer_model, clip_model, mode, bigvgan_vocoder_model=None):
"""Load MMAudio feature extraction utilities including VAE, Synchformer, and CLIP."""
original_loader = NODE_CLASS_MAPPINGS["MMAudioFeatureUtilsLoader"]()
return original_loader.loadmodel(vae_model, precision, synchformer_model, clip_model, mode, bigvgan_vocoder_model)
@@ -393,6 +413,7 @@ class MMAudioSampler:
CATEGORY = "MMAudio"
def sample(self, mmaudio_model, seed, feature_utils, duration, steps, cfg, prompt, negative_prompt, mask_away_clip, force_offload, images=None):
"""Sample audio from MMAudio model with conditioning."""
original_loader = NODE_CLASS_MAPPINGS["MMAudioSampler"]()
return original_loader.sample(mmaudio_model, seed, feature_utils, duration, steps, cfg, prompt, negative_prompt, mask_away_clip, force_offload, images)
@@ -406,6 +427,7 @@ class PulidModelLoader:
CATEGORY = "pulid"
def load_model(self, pulid_file):
"""Load PuLID identity preservation model."""
original_loader = NODE_CLASS_MAPPINGS["PulidModelLoader"]()
return original_loader.load_model(pulid_file)
@@ -423,6 +445,7 @@ class PulidInsightFaceLoader:
CATEGORY = "pulid"
def load_insightface(self, provider):
"""Load InsightFace face analysis model for PuLID."""
original_loader = NODE_CLASS_MAPPINGS["PulidInsightFaceLoader"]()
return original_loader.load_insightface(provider)
@@ -438,6 +461,7 @@ class PulidEvaClipLoader:
CATEGORY = "pulid"
def load_eva_clip(self):
"""Load EVA CLIP model for PuLID."""
original_loader = NODE_CLASS_MAPPINGS["PulidEvaClipLoader"]()
return original_loader.load_eva_clip()
@@ -472,6 +496,7 @@ class HyVideoModelLoader:
CATEGORY = "HunyuanVideoWrapper"
def loadmodel(self, model, base_precision, load_device, quantization, compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None, auto_cpu_offload=False):
"""Load HunyuanVideo model with specified precision and quantization."""
original_loader = NODE_CLASS_MAPPINGS["HyVideoModelLoader"]()
return original_loader.loadmodel(model, base_precision, load_device, quantization, compile_args, attention_mode, block_swap_args, lora, auto_cpu_offload)
@@ -497,6 +522,7 @@ class HyVideoVAELoader:
DESCRIPTION = "Loads Hunyuan VAE model from 'ComfyUI/models/vae'"
def loadmodel(self, model_name, precision, compile_args=None):
"""Load HunyuanVideo VAE model."""
original_loader = NODE_CLASS_MAPPINGS["HyVideoVAELoader"]()
return original_loader.loadmodel(model_name, precision, compile_args)
@@ -525,5 +551,31 @@ class DownloadAndLoadHyVideoTextEncoder:
DESCRIPTION = "Loads Hunyuan text_encoder model from 'ComfyUI/models/LLM'"
def loadmodel(self, llm_model, clip_model, precision, apply_final_norm=False, hidden_state_skip_layer=2, quantization="disabled"):
"""Download and load HunyuanVideo text encoder from HuggingFace."""
original_loader = NODE_CLASS_MAPPINGS["DownloadAndLoadHyVideoTextEncoder"]()
return original_loader.loadmodel(llm_model, clip_model, precision, apply_final_norm, hidden_state_skip_layer, quantization)
class UNetLoaderLP:
"""UNet Loader (Low Precision) - sets LoRA precision to False for CPU storage optimization"""
@classmethod
def INPUT_TYPES(s):
return {"required": { "unet_name": (folder_paths.get_filename_list("unet"), ),
}}
RETURN_TYPES = ("MODEL",)
FUNCTION = "load_unet"
CATEGORY = "loaders"
TITLE = "UNet Loader (LP)"
def load_unet(self, unet_name):
"""Load UNet with low-precision LoRA flag for CPU storage optimization."""
original_loader = NODE_CLASS_MAPPINGS["UNETLoader"]()
out = original_loader.load_unet(unet_name)
# Set the low-precision LoRA flag on the loaded model
if hasattr(out[0], 'model'):
out[0].model._distorch_high_precision_loras = False
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
out[0].patcher.model._distorch_high_precision_loras = False
return out
+1 -1
View File
@@ -1,7 +1,7 @@
[project]
name = "comfyui-multigpu"
description = "Provides a suite of custom nodes to manage multiple GPUs for ComfyUI, including advanced model offloading for both GGUF and Safetensor formats with DisTorch, and bespoke MultiGPU support for WanVideoWrapper and other custom nodes."
version = "2.4.7"
version = "2.5.0"
license = {file = "LICENSE"}
[project.urls]
+33 -1
View File
@@ -4,7 +4,7 @@ import sys
import inspect
import folder_paths
import comfy.model_management as mm
from .device_utils import get_device_list
from .device_utils import get_device_list, comfyui_memory_load
class WanVideoModelLoader:
@classmethod
@@ -89,8 +89,16 @@ class WanVideoModelLoader:
logging.debug(f"[MultiGPU] Both WanVideo modules patched successfully")
logging.debug(f"[MultiGPU] Calling original WanVideo loader")
try:
logging.info(comfyui_memory_load(f"pre-model-load:wan-model:{model}"))
except Exception:
pass
result = original_loader.loadmodel(model, base_precision, load_device, quantization,
compile_args, attention_mode, block_swap_args, lora, vram_management_args, extra_model=extra_model, fantasytalking_model=fantasytalking_model, multitalk_model=multitalk_model, fantasyportrait_model=fantasyportrait_model)
try:
logging.info(comfyui_memory_load(f"post-model-load:wan-model:{model}"))
except Exception:
pass
if result and len(result) > 0 and hasattr(result[0], 'model'):
model_obj = result[0]
@@ -156,7 +164,15 @@ class WanVideoVAELoader:
setattr(nodes_module, 'device', selected_device)
setattr(nodes_module, 'offload_device', selected_device)
try:
logging.info(comfyui_memory_load(f"pre-model-load:wan-vae:{model_name}"))
except Exception:
pass
result = original_loader.loadmodel(model_name, precision, compile_args)
try:
logging.info(comfyui_memory_load(f"post-model-load:wan-vae:{model_name}"))
except Exception:
pass
# Attach device info to VAE object for downstream nodes
if result and len(result) > 0:
@@ -219,7 +235,15 @@ class LoadWanVideoT5TextEncoder:
if device == "cpu":
setattr(nodes_module, 'offload_device', selected_device)
try:
logging.info(comfyui_memory_load(f"pre-model-load:wan-textenc:{model_name}"))
except Exception:
pass
result = original_loader.loadmodel(model_name, precision, load_device, quantization)
try:
logging.info(comfyui_memory_load(f"post-model-load:wan-textenc:{model_name}"))
except Exception:
pass
logging.info(f"[MultiGPU] WanVideo T5 Text encoder loaded on {selected_device}")
@@ -331,7 +355,15 @@ class LoadWanVideoClipTextEncoder:
if device == "cpu":
setattr(nodes_module, 'offload_device', selected_device)
try:
logging.info(comfyui_memory_load(f"pre-model-load:wan-clip:{model_name}"))
except Exception:
pass
result = original_loader.loadmodel(model_name, precision, load_device)
try:
logging.info(comfyui_memory_load(f"post-model-load:wan-clip:{model_name}"))
except Exception:
pass
logging.info(f"[MultiGPU] WanVideo CLIP encoder loaded on {selected_device}")
+52
View File
@@ -0,0 +1,52 @@
# CLIPLoaderDisTorch2MultiGPU
The `CLIPLoaderDisTorch2MultiGPU` node is used to load standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name` | `STRING` | The name of the CLIP model to load. |
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,52 @@
# CLIPLoaderGGUFDisTorch2MultiGPU
The `CLIPLoaderGGUFDisTorch2MultiGPU` node is used to load GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name` | `STRING` | The name of the CLIP model to load from combined clip and clip_gguf folders. |
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+19
View File
@@ -0,0 +1,19 @@
# CLIPLoaderGGUFMultiGPU
The `CLIPLoaderGGUFMultiGPU` node is used to load GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name` | `STRING` | The name of the CLIP model to load from combined clip and clip_gguf folders. |
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
+19
View File
@@ -0,0 +1,19 @@
# CLIPLoaderMultiGPU
The `CLIPLoaderMultiGPU` node is used to load CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name` | `STRING` | The name of the CLIP model to load. |
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
@@ -0,0 +1,51 @@
# CLIPVisionLoaderDisTorch2MultiGPU
The `CLIPVisionLoaderDisTorch2MultiGPU` node is used to load CLIP Vision models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger vision encoder models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip_vision` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_vision` | `STRING` | The name of the CLIP Vision model to load. |
| `device` | `STRING` | Target device for vision encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP_VISION` | `CLIP_VISION` | The loaded CLIP Vision model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# CLIPVisionLoaderMultiGPU
The `CLIPVisionLoaderMultiGPU` node is used to load CLIP Vision models with device selection capability, enabling users to specify which GPU or device should be used for vision encoder execution.
This node automatically detects models located in the `ComfyUI/models/clip_vision` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_vision` | `STRING` | The name of the CLIP Vision model to load. |
| `device` | `STRING` | Target device for vision encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP_VISION` | `CLIP_VISION` | The loaded CLIP Vision model. |
@@ -0,0 +1,55 @@
# CheckpointLoaderAdvancedDisTorch2MultiGPU
The `CheckpointLoaderAdvancedDisTorch2MultiGPU` node is used to load checkpoint models with advanced DisTorch2 distributed tensor allocation, providing granular control over UNet, CLIP, and VAE component allocation across multiple devices with independent virtual VRAM management.
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
| `unet_compute_device` | `STRING` | Target compute device for UNet distributed allocation (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
| `unet_virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes for UNet component distributed allocation (default: 4.0, range: 0.0-128.0). |
| `unet_donor_device` | `STRING` | Device to donate VRAM from when allocating UNet virtual memory (default: 'cpu'). |
| `clip_compute_device` | `STRING` | Target compute device for CLIP distributed allocation (default: 'cpu'). |
| `clip_virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes for CLIP component distributed allocation (default: 2.0, range: 0.0-128.0). |
| `clip_donor_device` | `STRING` | Device to donate VRAM from when allocating CLIP virtual memory (default: 'cpu'). |
| `vae_device` | `STRING` | Target device for the VAE component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
| `unet_expert_mode_allocations` | `STRING` | Advanced UNet allocation string for expert device/ratio distributions. |
| `clip_expert_mode_allocations` | `STRING` | Advanced CLIP allocation string for expert device/ratio distributions. |
| `high_precision_loras` | `BOOLEAN` | Whether to use high-precision LoRA patches (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet diffusion model with DisTorch2 distributed allocation. |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation. |
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
This advanced checkpoint loader provides independent DisTorch2 allocation control for UNet and CLIP components, while using standard device placement for VAE.
### Key Concepts
**Individual Component Control**: Each model component (UNet, CLIP, VAE) can have its own allocation strategy.
**Virtual VRAM Allocation**: Artificially increases the available VRAM on compute devices by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of each component should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `unet_compute_device`: `cuda:0`, `unet_virtual_vram_gb`: `8.0`, `unet_donor_device`: `cuda:1`
- `clip_compute_device`: `cuda:1`, `clip_virtual_vram_gb`: `2.0`, `clip_donor_device`: `cpu`
- Result: UNet loads as if cuda:0 has 8GB more VRAM, CLIP loads with cuda:1 having 2GB more capacity.
**Expert Ratio Allocation**:
- `unet_expert_mode_allocations`: `cuda:0,70%;cuda:1,30%`
- `clip_expert_mode_allocations`: `cuda:1,50%;cpu,50%`
- Distributes UNet with 70% on GPU 0, 30% on GPU 1, and CLIP with 50% on GPU 1, 50% on CPU.
@@ -0,0 +1,22 @@
# CheckpointLoaderAdvancedMultiGPU
The `CheckpointLoaderAdvancedMultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with granular device control, allowing individual placement of each model component on different GPUs or devices.
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
| `unet_device` | `STRING` | Target device for the UNet diffusion model component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
| `clip_device` | `STRING` | Target device for the CLIP text encoder component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
| `vae_device` | `STRING` | Target device for the VAE decoder/encoder component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
@@ -0,0 +1,53 @@
# CheckpointLoaderSimpleDisTorch2MultiGPU
The `CheckpointLoaderSimpleDisTorch2MultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet diffusion model with DisTorch2 distributed allocation applied. |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,20 @@
# CheckpointLoaderSimpleMultiGPU
The `CheckpointLoaderSimpleMultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
@@ -0,0 +1,51 @@
# ControlNetLoaderDisTorch2MultiGPU
The `ControlNetLoaderDisTorch2MultiGPU` node is used to load ControlNet models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger conditional generation models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/controlnet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `control_net_name` | `STRING` | The name of the ControlNet model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CONTROL_NET` | `CONTROL_NET` | The loaded ControlNet model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# ControlNetLoaderMultiGPU
The `ControlNetLoaderMultiGPU` node is used to load ControlNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/controlnet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `control_net_name` | `STRING` | The name of the ControlNet model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CONTROL_NET` | `CONTROL_NET` | The loaded ControlNet model. |
@@ -0,0 +1,51 @@
# DiffControlNetLoaderDisTorch2MultiGPU
The `DiffControlNetLoaderDisTorch2MultiGPU` node is used to load Diffusers ControlNet models (HuggingFace Hub repositories) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger conditional generation models across multiple GPUs.
This node loads ControlNet models directly from HuggingFace model repositories by specifying the repository ID (e.g., "diffusers/controlnet-canny-sdxl-1.0").
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers ControlNet model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CONTROL_NET` | `CONTROL_NET` | The loaded diffusers ControlNet model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# DiffControlNetLoaderMultiGPU
The `DiffControlNetLoaderMultiGPU` node is used to load Diffusers ControlNet models (HuggingFace Hub repositories) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node loads ControlNet models directly from HuggingFace model repositories by specifying the repository ID (e.g., "diffusers/controlnet-canny-sdxl-1.0").
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers ControlNet model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CONTROL_NET` | `CONTROL_NET` | The loaded diffusers ControlNet model. |
@@ -0,0 +1,51 @@
# DiffusersLoaderDisTorch2MultiGPU
The `DiffusersLoaderDisTorch2MultiGPU` node is used to load Diffusers models (HuggingFace Hub repositories) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger diffusion models across multiple GPUs.
This node loads models directly from HuggingFace model repositories by specifying the repository ID (e.g., "stabilityai/stable-diffusion-xl-base-1.0").
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers model to load (e.g., 'stabilityai/stable-diffusion-xl-base-1.0'). |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded diffusers model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# DiffusersLoaderMultiGPU
The `DiffusersLoaderMultiGPU` node is used to load Diffusers models (HuggingFace Hub repositories) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node loads models directly from HuggingFace model repositories by specifying the repository ID (e.g., "stabilityai/stable-diffusion-xl-base-1.0").
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers model to load (e.g., 'stabilityai/stable-diffusion-xl-base-1.0'). |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded diffusers model. |
@@ -0,0 +1,53 @@
# DualCLIPLoaderDisTorch2MultiGPU
The `DualCLIPLoaderDisTorch2MultiGPU` node is used to load dual standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,53 @@
# DualCLIPLoaderGGUFDisTorch2MultiGPU
The `DualCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load dual GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+20
View File
@@ -0,0 +1,20 @@
# DualCLIPLoaderGGUFMultiGPU
The `DualCLIPLoaderGGUFMultiGPU` node is used to load dual GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models. |
+20
View File
@@ -0,0 +1,20 @@
# DualCLIPLoaderMultiGPU
The `DualCLIPLoaderMultiGPU` node is used to load dual CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models. |
@@ -0,0 +1,54 @@
# QuadrupleCLIPLoaderDisTorch2MultiGPU
The `QuadrupleCLIPLoaderDisTorch2MultiGPU` node is used to load quadruple standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,54 @@
# QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU
The `QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load quadruple GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load from combined clip and clip_gguf folders. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,21 @@
# QuadrupleCLIPLoaderGGUFMultiGPU
The `QuadrupleCLIPLoaderGGUFMultiGPU` node is used to load quadruple GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load from combined clip and clip_gguf folders. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models. |
+21
View File
@@ -0,0 +1,21 @@
# QuadrupleCLIPLoaderMultiGPU
The `QuadrupleCLIPLoaderMultiGPU` node is used to load quadruple CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models. |
@@ -0,0 +1,53 @@
# TripleCLIPLoaderDisTorch2MultiGPU
The `TripleCLIPLoaderDisTorch2MultiGPU` node is used to load triple standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3 with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,53 @@
# TripleCLIPLoaderGGUFDisTorch2MultiGPU
The `TripleCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load triple GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3 with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+20
View File
@@ -0,0 +1,20 @@
# TripleCLIPLoaderGGUFMultiGPU
The `TripleCLIPLoaderGGUFMultiGPU` node is used to load triple GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3. |
+20
View File
@@ -0,0 +1,20 @@
# TripleCLIPLoaderMultiGPU
The `TripleCLIPLoaderMultiGPU` node is used to load triple CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3. |
+51
View File
@@ -0,0 +1,51 @@
# UNETLoaderDisTorch2MultiGPU
The `UNETLoaderDisTorch2MultiGPU` node is used to load UNet diffusion models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/unet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the UNet model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# UNETLoaderMultiGPU
The `UNETLoaderMultiGPU` node is used to load diffusion model UNet components with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/unet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the UNet model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
@@ -0,0 +1,54 @@
# UnetLoaderGGUFAdvancedDisTorch2MultiGPU
The `UnetLoaderGGUFAdvancedDisTorch2MultiGPU` node is used to load GGUF format UNet models with advanced quantization options and DisTorch2 distributed tensor allocation, enabling sophisticated multi-device VRAM management across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
| `dequant_dtype` | `STRING` | Target data type for model dequantization during loading (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
| `patch_dtype` | `STRING` | Data type for LoRA patches applied to the model (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
| `patch_on_device` | `BOOLEAN` | Whether to apply LoRA patches directly on the target device. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with advanced quantization settings and DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
@@ -0,0 +1,21 @@
# UnetLoaderGGUFAdvancedMultiGPU
The `UnetLoaderGGUFAdvancedMultiGPU` node is used to load GGUF format UNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
| `dequant_dtype` | `STRING` | Target data type for model dequantization during loading (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
| `patch_dtype` | `STRING` | Data type for LoRA patches applied to the model (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
| `patch_on_device` | `BOOLEAN` | Whether to apply LoRA patches directly on the target device. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with advanced quantization settings. |
@@ -0,0 +1,51 @@
# UnetLoaderGGUFDisTorch2MultiGPU
The `UnetLoaderGGUFDisTorch2MultiGPU` node is used to load GGUF format UNet models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# UnetLoaderGGUFMultiGPU
The `UnetLoaderGGUFMultiGPU` node is used to load GGUF format UNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `MODEL` | `MODEL` | The loaded GGUF format UNet model. |
+51
View File
@@ -0,0 +1,51 @@
# VAELoaderDisTorch2MultiGPU
The `VAELoaderDisTorch2MultiGPU` node is used to load VAE (Variational Autoencoder) models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
This node automatically detects models located in the `ComfyUI/models/vae` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `vae_name` | `STRING` | The name of the VAE model to load. |
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `VAE` | `VAE` | The loaded VAE decoder/encoder with DisTorch2 distributed allocation applied. |
## DisTorch2 Distributed Loading
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
### Key Concepts
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
### Allocation Examples
**Basic Virtual VRAM Mode**:
- `compute_device`: `cuda:0`
- `virtual_vram_gb`: `8.0`
- `donor_device`: `cuda:1`
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
**Expert Ratio Allocation**:
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
**Expert Byte Allocation**:
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
**Mixed Mode**:
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
+18
View File
@@ -0,0 +1,18 @@
# VAELoaderMultiGPU
The `VAELoaderMultiGPU` node is used to load VAE (Variational Autoencoder) models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
This node automatically detects models located in the `ComfyUI/models/vae` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
## Inputs
| Parameter | Data Type | Description |
| --- | --- | --- |
| `vae_name` | `STRING` | The name of the VAE model to load. |
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
## Outputs
| Output Name | Data Type | Description |
| --- | --- | --- |
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
+520
View File
@@ -0,0 +1,520 @@
"""
ComfyUI-MultiGPU Wrapper Functions
All node override/wrapper generation functions consolidated in one location
"""
import copy
import hashlib
import logging
from .device_utils import get_device_list
logger = logging.getLogger("MultiGPU")
# ============================================================================
# DISTORCH V2 SAFETENSOR WRAPPERS (DisTorch2 for .safetensors and .gguf)
# ============================================================================
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround):
"""Internal factory function creating DisTorch2 override class with parameterized device selection behavior."""
from .distorch_2 import (
register_patched_safetensor_modelpatcher,
safetensor_allocation_store,
safetensor_settings_store,
create_safetensor_model_hash
)
from .model_management_mgpu import force_full_system_cleanup
class NodeOverrideDisTorchSafetensorV2(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"][device_param_name] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
inputs["optional"]["keep_loaded"] = ("BOOLEAN", {"default": True})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
@classmethod
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
expert_mode_allocations="", keep_loaded=True, **kwargs):
device_value = kwargs.get(device_param_name)
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{keep_loaded}"
current_hash = hashlib.sha256(settings_str.encode()).hexdigest()
if not hasattr(cls, '_last_hash'):
cls._last_hash = current_hash
logger.mgpu_mm_log(f"IS_CHANGED first call: {current_hash[:8]}")
elif cls._last_hash != current_hash:
cls._last_hash = current_hash
logger.mgpu_mm_log(f"IS_CHANGED CHANGED: {current_hash[:8]} ← settings changed")
return current_hash
def override(self, *args, virtual_vram_gb=4.0, donor_device="cpu",
expert_mode_allocations="", keep_loaded=True, **kwargs):
device_value = kwargs.get(device_param_name)
unload_distorch_model = not keep_loaded
if device_value is not None:
device_setter_func(device_value)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in [device_param_name, 'virtual_vram_gb',
'donor_device', 'expert_mode_allocations',
'keep_loaded']}
if apply_device_kwarg_workaround:
clean_kwargs['device'] = 'default'
register_patched_safetensor_modelpatcher()
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{device_value};{virtual_vram_gb};{donor_device}"
elif expert_mode_allocations:
vram_string = device_value
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **clean_kwargs)
model_to_check = None
if hasattr(out[0], 'model'):
model_to_check = out[0]
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_to_check = out[0].patcher
if model_to_check:
model_hash = create_safetensor_model_hash(model_to_check, "override_store")
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}"
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
safetensor_allocation_store[model_hash] = full_allocation
safetensor_settings_store[model_hash] = settings_hash
logger.debug(f"[MultiGPU DisTorch V2] Stored allocation for model {model_hash[:8]}: {full_allocation}")
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
logger.mgpu_mm_log(f"[FLAG_SET_START] Setting '_mgpu_unload_distorch_model' to: {unload_distorch_model} (keep_loaded={keep_loaded})")
if hasattr(out[0], 'model'):
mp = out[0]
mp_id = id(mp)
inner_model = getattr(mp, 'model', None)
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
mp._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
if inner_model:
inner_model._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
mp = out[0].patcher
mp_id = id(mp)
inner_model = getattr(mp, 'model', None)
inner_model_id = id(inner_model) if inner_model else None
inner_model_name = type(inner_model).__name__ if inner_model else "None"
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
mp._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
if inner_model:
inner_model._mgpu_unload_distorch_model = unload_distorch_model
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
if unload_distorch_model:
logger.mgpu_mm_log("[FLAG_TRIGGER] unload_distorch_model=True, triggering full system cleanup")
force_full_system_cleanup(reason="policy_every_load", force=True)
return out
return NodeOverrideDisTorchSafetensorV2
def override_class_with_distorch_safetensor_v2(cls):
"""DisTorch 2.0 wrapper for safetensor UNet/VAE models"""
from . import set_current_device
return _create_distorch_safetensor_v2_override(
cls,
device_param_name="compute_device",
device_setter_func=set_current_device,
apply_device_kwarg_workaround=False
)
def override_class_with_distorch_safetensor_v2_clip(cls):
"""DisTorch 2.0 wrapper for safetensor CLIP models (with device kwarg workaround)"""
from . import set_current_text_encoder_device
return _create_distorch_safetensor_v2_override(
cls,
device_param_name="device",
device_setter_func=set_current_text_encoder_device,
apply_device_kwarg_workaround=True
)
def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
"""DisTorch 2.0 wrapper for safetensor Triple/Quad CLIP models (no device kwarg workaround)"""
from . import set_current_text_encoder_device
return _create_distorch_safetensor_v2_override(
cls,
device_param_name="device",
device_setter_func=set_current_text_encoder_device,
apply_device_kwarg_workaround=False
)
# ============================================================================
# DISTORCH V1 LEGACY WRAPPERS (Rewritten to call V2 backend)
# ============================================================================
def override_class_with_distorch_gguf(cls):
"""DisTorch V1 Legacy wrapper - maintains V1 UI but calls V2 backend"""
from . import set_current_device
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
class NodeOverrideDisTorchGGUFLegacy(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
return inputs
CATEGORY = "multigpu/legacy"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (Legacy)"
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
if device is not None:
set_current_device(device)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
'expert_mode_allocations']}
register_patched_safetensor_modelpatcher()
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **clean_kwargs)
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "v1_compat")
safetensor_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_compat")
safetensor_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchGGUFLegacy
def override_class_with_distorch_gguf_v2(cls):
"""DisTorch V2 wrapper for GGUF models"""
from . import set_current_device
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
class NodeOverrideDisTorchGGUFv2(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
compute_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
return inputs
CATEGORY = "multigpu/distorch_2"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
def override(self, *args, compute_device=None, virtual_vram_gb=4.0, donor_device="cpu", expert_mode_allocations="", **kwargs):
if compute_device is not None:
set_current_device(compute_device)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in ['compute_device', 'virtual_vram_gb',
'donor_device', 'expert_mode_allocations']}
register_patched_safetensor_modelpatcher()
vram_string = ""
if virtual_vram_gb > 0:
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
elif expert_mode_allocations:
vram_string = compute_device
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **clean_kwargs)
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "v2_gguf")
safetensor_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "v2_gguf")
safetensor_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchGGUFv2
def override_class_with_distorch_clip(cls):
"""DisTorch V1 wrapper for CLIP models - calls V2 backend"""
from . import set_current_text_encoder_device
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
class NodeOverrideDisTorchClip(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch)"
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
'expert_mode_allocations']}
register_patched_safetensor_modelpatcher()
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **clean_kwargs)
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "v1_clip")
safetensor_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_clip")
safetensor_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchClip
def override_class_with_distorch_clip_no_device(cls):
"""DisTorch V1 wrapper for Triple/Quad CLIP models - calls V2 backend"""
from . import set_current_text_encoder_device
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
class NodeOverrideDisTorchClipNoDevice(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch)"
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
# Strip MultiGPU-specific parameters before calling original function
clean_kwargs = {k: v for k, v in kwargs.items()
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
'expert_mode_allocations']}
register_patched_safetensor_modelpatcher()
vram_string = ""
if virtual_vram_gb > 0:
if use_other_vram:
available_devices = [d for d in get_device_list() if d != "cpu"]
other_devices = [d for d in available_devices if d != device]
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
device_string = ','.join(other_devices + ['cpu'])
vram_string = f"{device};{virtual_vram_gb};{device_string}"
else:
vram_string = f"{device};{virtual_vram_gb};cpu"
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **clean_kwargs)
if hasattr(out[0], 'model'):
model_hash = create_safetensor_model_hash(out[0], "v1_clip_nodev")
safetensor_allocation_store[model_hash] = full_allocation
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_clip_nodev")
safetensor_allocation_store[model_hash] = full_allocation
return out
return NodeOverrideDisTorchClipNoDevice
# Backward compatibility alias
override_class_with_distorch = override_class_with_distorch_gguf
# ============================================================================
# STANDARD MULTIGPU WRAPPERS (Device selection without DisTorch)
# ============================================================================
def override_class(cls):
"""Standard MultiGPU device override for UNet/VAE models"""
from . import set_current_device
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_device(device)
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride
def override_class_clip(cls):
"""Standard MultiGPU device override for CLIP models (with device kwarg workaround)"""
from . import set_current_text_encoder_device
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
kwargs['device'] = 'default'
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride
def override_class_clip_no_device(cls):
"""Standard MultiGPU device override for Triple/Quad CLIP models (no device kwarg workaround)"""
from . import set_current_text_encoder_device
class NodeOverride(cls):
@classmethod
def INPUT_TYPES(s):
inputs = copy.deepcopy(cls.INPUT_TYPES())
devices = get_device_list()
default_device = devices[1] if len(devices) > 1 else devices[0]
inputs["optional"] = inputs.get("optional", {})
inputs["optional"]["device"] = (devices, {"default": default_device})
return inputs
CATEGORY = "multigpu"
FUNCTION = "override"
def override(self, *args, device=None, **kwargs):
if device is not None:
set_current_text_encoder_device(device)
fn = getattr(super(), cls.FUNCTION)
out = fn(*args, **kwargs)
return out
return NodeOverride