+3
-1
@@ -1,3 +1,5 @@
|
||||
# Python and IDE
|
||||
__pycache__/
|
||||
.vscode/settings.json
|
||||
.clinerules
|
||||
.vscode
|
||||
memory-bank/
|
||||
@@ -112,6 +112,15 @@ Currently supported nodes (automatically detected if available):
|
||||
|
||||
All MultiGPU nodes available for your install can be found in the "multigpu" category in the node menu.
|
||||
|
||||
## Node Documentation
|
||||
|
||||
Detailed technical documentation is available for all **automatically-detected core MultiGPU and DisTorch2 nodes**, covering 36+ documented nodes with comprehensive parameter details, output specifications, and DisTorch2 allocation guidance where applicable.
|
||||
|
||||
- **To access documentation**: Click on any core MultiGPU or DisTorch2 node in ComfyUI and select "Help" (question mark inside a circle) from the resultant menu
|
||||
- **Coverage**: All standard ComfyUI loader nodes (UNet, VAE, Checkpoints, CLIP, ControlNet, Diffusers) plus popular GGUF loader variants
|
||||
- **Contents**: Input parameters with data types and descriptions, output specifications, usage examples, and DisTorch2 distributed loading explanations with allocation modes and strategies
|
||||
- **Note**: Documentation covers core ComfyUI-MultiGPU functionality only. Third-party custom node integrations (WanVideoWrapper, Florence2, etc.) have their own separate documentation.
|
||||
|
||||
## Example workflows
|
||||
|
||||
All workflows have been tested on a 2x 3090 + 1060ti linux setup, a 4070 win 11 setup, and a 3090/1070ti linux setup.
|
||||
|
||||
+59
-145
@@ -1,121 +1,73 @@
|
||||
import torch
|
||||
import logging
|
||||
import weakref
|
||||
import os
|
||||
import copy
|
||||
from pathlib import Path
|
||||
import folder_paths
|
||||
import comfy.model_management as mm
|
||||
import comfy.model_patcher
|
||||
from nodes import NODE_CLASS_MAPPINGS as GLOBAL_NODE_CLASS_MAPPINGS
|
||||
from .device_utils import get_device_list, is_accelerator_available
|
||||
from .device_utils import (
|
||||
get_device_list,
|
||||
is_accelerator_available,
|
||||
soft_empty_cache_multigpu,
|
||||
)
|
||||
from .model_management_mgpu import (
|
||||
trigger_executor_cache_reset,
|
||||
check_cpu_memory_threshold,
|
||||
multigpu_memory_log,
|
||||
force_full_system_cleanup,
|
||||
)
|
||||
|
||||
# --- DisTorch V2 Logging Configuration ---
|
||||
# Set to "E" for Engineering (DEBUG) or "P" for Production (INFO)
|
||||
LOG_LEVEL = "P"
|
||||
WEB_DIRECTORY = "./web"
|
||||
MGPU_MM_LOG = False
|
||||
DEBUG_LOG = False
|
||||
|
||||
# Configure logger
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
logger.propagate = False
|
||||
|
||||
if not logger.handlers:
|
||||
log_level = logging.DEBUG if LOG_LEVEL == "E" else logging.INFO
|
||||
log_level = logging.DEBUG if DEBUG_LOG else logging.INFO
|
||||
handler = logging.StreamHandler()
|
||||
formatter = logging.Formatter('%(message)s')
|
||||
handler.setFormatter(formatter)
|
||||
logger.addHandler(handler)
|
||||
logger.setLevel(log_level)
|
||||
logger.info(f"[MultiGPU Initialization] Logger initialized with level: {logging.getLevelName(log_level)}")
|
||||
|
||||
def mgpu_mm_log_method(self, msg):
|
||||
"""Add MultiGPU model management logging method to logger instance."""
|
||||
if MGPU_MM_LOG:
|
||||
self.info(f"[MultiGPU Model Management] {msg}")
|
||||
logger.mgpu_mm_log = mgpu_mm_log_method.__get__(logger, type(logger))
|
||||
|
||||
def check_module_exists(module_path):
|
||||
"""Check if a custom node module exists in ComfyUI custom_nodes directory."""
|
||||
full_path = os.path.join(folder_paths.get_folder_paths("custom_nodes")[0], module_path)
|
||||
logger.debug(f"[MultiGPU] Checking for module at {full_path}")
|
||||
if not os.path.exists(full_path):
|
||||
logger.debug(f"[MultiGPU] Module {module_path} not found - skipping")
|
||||
return False
|
||||
logger.debug(f"[MultiGPU] Found {module_path}, creating compatible MultiGPU nodes")
|
||||
return True
|
||||
|
||||
# Global device state management
|
||||
current_device = mm.get_torch_device()
|
||||
current_text_encoder_device = mm.text_encoder_device()
|
||||
|
||||
def set_current_device(device):
|
||||
"""Set the current device context for MultiGPU operations."""
|
||||
global current_device
|
||||
current_device = device
|
||||
logger.info(f"[MultiGPU Initialization] current_device set to: {device}")
|
||||
logger.debug(f"[MultiGPU Initialization] current_device set to: {device}")
|
||||
|
||||
def set_current_text_encoder_device(device):
|
||||
"""Set the current text encoder device context for CLIP models."""
|
||||
global current_text_encoder_device
|
||||
current_text_encoder_device = device
|
||||
logger.info(f"[MultiGPU Initialization] current_text_encoder_device set to: {device}")
|
||||
|
||||
def override_class(cls):
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
|
||||
if device is not None:
|
||||
set_current_device(device)
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
|
||||
def override_class_clip(cls):
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
kwargs['device'] = 'default'
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
|
||||
def override_class_clip_no_device(cls):
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
|
||||
logger.debug(f"[MultiGPU Initialization] current_text_encoder_device set to: {device}")
|
||||
|
||||
def get_torch_device_patched():
|
||||
"""Return MultiGPU-aware device selection for patched mm.get_torch_device."""
|
||||
device = None
|
||||
if (not is_accelerator_available() or mm.cpu_state == mm.CPUState.CPU or "cpu" in str(current_device).lower()):
|
||||
device = torch.device("cpu")
|
||||
@@ -126,6 +78,7 @@ def get_torch_device_patched():
|
||||
return device
|
||||
|
||||
def text_encoder_device_patched():
|
||||
"""Return MultiGPU-aware text encoder device for patched mm.text_encoder_device."""
|
||||
device = None
|
||||
if (not is_accelerator_available() or mm.cpu_state == mm.CPUState.CPU or "cpu" in str(current_text_encoder_device).lower()):
|
||||
device = torch.device("cpu")
|
||||
@@ -135,23 +88,12 @@ def text_encoder_device_patched():
|
||||
logger.debug(f"[MultiGPU Core Patching] text_encoder_device_patched returning device: {device} (current_text_encoder_device={current_text_encoder_device})")
|
||||
return device
|
||||
|
||||
|
||||
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device, mm.text_encoder_device, and mm.text_encoder_initial_device")
|
||||
logger.info(f"[MultiGPU Core Patching] Patching mm.get_torch_device and mm.text_encoder_device")
|
||||
logger.debug(f"[MultiGPU DEBUG] Initial current_device: {current_device}")
|
||||
logger.debug(f"[MultiGPU DEBUG] Initial current_text_encoder_device: {current_text_encoder_device}")
|
||||
mm.get_torch_device = get_torch_device_patched
|
||||
mm.text_encoder_device = text_encoder_device_patched
|
||||
|
||||
def check_module_exists(module_path):
|
||||
full_path = os.path.join(folder_paths.get_folder_paths("custom_nodes")[0], module_path)
|
||||
logger.debug(f"[MultiGPU] Checking for module at {full_path}")
|
||||
if not os.path.exists(full_path):
|
||||
logger.debug(f"[MultiGPU] Module {module_path} not found - skipping")
|
||||
return False
|
||||
logger.debug(f"[MultiGPU] Found {module_path}, creating compatible MultiGPU nodes")
|
||||
return True
|
||||
|
||||
# Import from nodes.py
|
||||
from .nodes import (
|
||||
DeviceSelectorMultiGPU,
|
||||
HunyuanVideoEmbeddingsAdapter,
|
||||
@@ -175,9 +117,9 @@ from .nodes import (
|
||||
HyVideoModelLoader,
|
||||
HyVideoVAELoader,
|
||||
DownloadAndLoadHyVideoTextEncoder,
|
||||
UNetLoaderLP,
|
||||
)
|
||||
|
||||
# Import from wanvideo.py
|
||||
from .wanvideo import (
|
||||
WanVideoModelLoader,
|
||||
WanVideoModelLoader_2,
|
||||
@@ -189,81 +131,63 @@ from .wanvideo import (
|
||||
WanVideoSampler
|
||||
)
|
||||
|
||||
# Import from distorch.py
|
||||
from .distorch import (
|
||||
model_allocation_store,
|
||||
create_model_hash,
|
||||
register_patched_ggufmodelpatcher,
|
||||
analyze_ggml_loading,
|
||||
calculate_vvram_allocation_string,
|
||||
from .wrappers import (
|
||||
override_class,
|
||||
override_class_clip,
|
||||
override_class_clip_no_device,
|
||||
override_class_with_distorch_gguf,
|
||||
override_class_with_distorch_gguf_v2,
|
||||
override_class_with_distorch_clip,
|
||||
override_class_with_distorch_clip_no_device,
|
||||
override_class_with_distorch
|
||||
override_class_with_distorch,
|
||||
override_class_with_distorch_safetensor_v2,
|
||||
override_class_with_distorch_safetensor_v2_clip,
|
||||
override_class_with_distorch_safetensor_v2_clip_no_device,
|
||||
)
|
||||
|
||||
# Import from distorch_2.py for DisTorch v2 SafeTensor support
|
||||
from .distorch_2 import (
|
||||
safetensor_allocation_store,
|
||||
create_safetensor_model_hash,
|
||||
register_patched_safetensor_modelpatcher,
|
||||
analyze_safetensor_loading,
|
||||
calculate_safetensor_vvram_allocation,
|
||||
override_class_with_distorch_safetensor_v2,
|
||||
override_class_with_distorch_safetensor_v2_clip,
|
||||
override_class_with_distorch_safetensor_v2_clip_no_device
|
||||
)
|
||||
|
||||
# Import advanced checkpoint loaders
|
||||
from .checkpoint_multigpu import (
|
||||
CheckpointLoaderAdvancedMultiGPU,
|
||||
CheckpointLoaderAdvancedDisTorch2MultiGPU
|
||||
)
|
||||
|
||||
# Initialize NODE_CLASS_MAPPINGS
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"DeviceSelectorMultiGPU": DeviceSelectorMultiGPU,
|
||||
"HunyuanVideoEmbeddingsAdapter": HunyuanVideoEmbeddingsAdapter,
|
||||
"CheckpointLoaderAdvancedMultiGPU": CheckpointLoaderAdvancedMultiGPU,
|
||||
"CheckpointLoaderAdvancedDisTorch2MultiGPU": CheckpointLoaderAdvancedDisTorch2MultiGPU,
|
||||
"UNetLoaderLP": UNetLoaderLP,
|
||||
}
|
||||
|
||||
# Standard MultiGPU nodes
|
||||
NODE_CLASS_MAPPINGS["UNETLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["UNETLoader"])
|
||||
NODE_CLASS_MAPPINGS["VAELoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["VAELoader"])
|
||||
NODE_CLASS_MAPPINGS["CLIPLoaderMultiGPU"] = override_class_clip(GLOBAL_NODE_CLASS_MAPPINGS["CLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["DualCLIPLoaderMultiGPU"] = override_class_clip(GLOBAL_NODE_CLASS_MAPPINGS["DualCLIPLoader"])
|
||||
if "TripleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["TripleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
|
||||
if "QuadrupleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["TripleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["CLIPVisionLoaderMultiGPU"] = override_class_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["CLIPVisionLoader"])
|
||||
NODE_CLASS_MAPPINGS["CheckpointLoaderSimpleMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["CheckpointLoaderSimple"])
|
||||
NODE_CLASS_MAPPINGS["ControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["ControlNetLoader"])
|
||||
if "DiffusersLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["DiffusersLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
|
||||
if "DiffControlNetLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["DiffControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
|
||||
|
||||
# DisTorch 2 SafeTensor nodes for FLUX and other safetensor models
|
||||
NODE_CLASS_MAPPINGS["DiffusersLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
|
||||
NODE_CLASS_MAPPINGS["DiffControlNetLoaderMultiGPU"] = override_class(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
|
||||
NODE_CLASS_MAPPINGS["UNETLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["UNETLoader"])
|
||||
NODE_CLASS_MAPPINGS["VAELoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["VAELoader"])
|
||||
NODE_CLASS_MAPPINGS["CLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip(GLOBAL_NODE_CLASS_MAPPINGS["CLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["DualCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip(GLOBAL_NODE_CLASS_MAPPINGS["DualCLIPLoader"])
|
||||
if "TripleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["TripleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
|
||||
if "QuadrupleCLIPLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["TripleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["TripleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["QuadrupleCLIPLoader"])
|
||||
NODE_CLASS_MAPPINGS["CLIPVisionLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2_clip_no_device(GLOBAL_NODE_CLASS_MAPPINGS["CLIPVisionLoader"])
|
||||
NODE_CLASS_MAPPINGS["CheckpointLoaderSimpleDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["CheckpointLoaderSimple"])
|
||||
NODE_CLASS_MAPPINGS["ControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["ControlNetLoader"])
|
||||
if "DiffusersLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["DiffusersLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
|
||||
if "DiffControlNetLoader" in GLOBAL_NODE_CLASS_MAPPINGS:
|
||||
NODE_CLASS_MAPPINGS["DiffControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
|
||||
NODE_CLASS_MAPPINGS["DiffusersLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffusersLoader"])
|
||||
NODE_CLASS_MAPPINGS["DiffControlNetLoaderDisTorch2MultiGPU"] = override_class_with_distorch_safetensor_v2(GLOBAL_NODE_CLASS_MAPPINGS["DiffControlNetLoader"])
|
||||
|
||||
# --- Registration Table ---
|
||||
logger.info("[MultiGPU] Initiating custom_node Registration. . .")
|
||||
dash_line = "-" * 47
|
||||
fmt_reg = "{:<30}{:>5}{:>10}"
|
||||
@@ -274,6 +198,7 @@ logger.info(dash_line)
|
||||
registration_data = []
|
||||
|
||||
def register_and_count(module_names, node_map):
|
||||
"""Register MultiGPU node wrappers for detected custom node modules."""
|
||||
found = False
|
||||
for name in module_names:
|
||||
if check_module_exists(name):
|
||||
@@ -290,26 +215,21 @@ def register_and_count(module_names, node_map):
|
||||
registration_data.append({"name": module_names[0], "found": "Y" if found else "N", "count": count})
|
||||
return found
|
||||
|
||||
# ComfyUI-LTXVideo
|
||||
ltx_nodes = {"LTXVLoaderMultiGPU": override_class(LTXVLoader)}
|
||||
register_and_count(["ComfyUI-LTXVideo", "comfyui-ltxvideo"], ltx_nodes)
|
||||
|
||||
# ComfyUI-Florence2
|
||||
florence_nodes = {
|
||||
"Florence2ModelLoaderMultiGPU": override_class(Florence2ModelLoader),
|
||||
"DownloadAndLoadFlorence2ModelMultiGPU": override_class(DownloadAndLoadFlorence2Model)
|
||||
}
|
||||
register_and_count(["ComfyUI-Florence2", "comfyui-florence2"], florence_nodes)
|
||||
|
||||
# ComfyUI_bitsandbytes_NF4
|
||||
nf4_nodes = {"CheckpointLoaderNF4MultiGPU": override_class(CheckpointLoaderNF4)}
|
||||
register_and_count(["ComfyUI_bitsandbytes_NF4", "comfyui_bitsandbytes_nf4"], nf4_nodes)
|
||||
|
||||
# x-flux-comfyui
|
||||
flux_controlnet_nodes = {"LoadFluxControlNetMultiGPU": override_class(LoadFluxControlNet)}
|
||||
register_and_count(["x-flux-comfyui"], flux_controlnet_nodes)
|
||||
|
||||
# ComfyUI-MMAudio
|
||||
mmaudio_nodes = {
|
||||
"MMAudioModelLoaderMultiGPU": override_class(MMAudioModelLoader),
|
||||
"MMAudioFeatureUtilsLoaderMultiGPU": override_class(MMAudioFeatureUtilsLoader),
|
||||
@@ -317,7 +237,6 @@ mmaudio_nodes = {
|
||||
}
|
||||
register_and_count(["ComfyUI-MMAudio", "comfyui-mmaudio"], mmaudio_nodes)
|
||||
|
||||
# ComfyUI-GGUF
|
||||
gguf_nodes = {
|
||||
"UnetLoaderGGUFDisTorchMultiGPU": override_class_with_distorch_gguf(UnetLoaderGGUF),
|
||||
"UnetLoaderGGUFAdvancedDisTorchMultiGPU": override_class_with_distorch_gguf(UnetLoaderGGUFAdvanced),
|
||||
@@ -340,7 +259,6 @@ gguf_nodes = {
|
||||
}
|
||||
register_and_count(["ComfyUI-GGUF", "comfyui-gguf"], gguf_nodes)
|
||||
|
||||
# PuLID_ComfyUI
|
||||
pulid_nodes = {
|
||||
"PulidModelLoaderMultiGPU": override_class(PulidModelLoader),
|
||||
"PulidInsightFaceLoaderMultiGPU": override_class(PulidInsightFaceLoader),
|
||||
@@ -348,7 +266,6 @@ pulid_nodes = {
|
||||
}
|
||||
register_and_count(["PuLID_ComfyUI", "pulid_comfyui"], pulid_nodes)
|
||||
|
||||
# ComfyUI-HunyuanVideoWrapper
|
||||
hunyuan_nodes = {
|
||||
"HyVideoModelLoaderMultiGPU": override_class(HyVideoModelLoader),
|
||||
"HyVideoVAELoaderMultiGPU": override_class(HyVideoVAELoader),
|
||||
@@ -356,7 +273,6 @@ hunyuan_nodes = {
|
||||
}
|
||||
register_and_count(["ComfyUI-HunyuanVideoWrapper", "comfyui-hunyuanvideowrapper"], hunyuan_nodes)
|
||||
|
||||
# ComfyUI-WanVideoWrapper
|
||||
wanvideo_nodes = {
|
||||
"WanVideoModelLoaderMultiGPU": WanVideoModelLoader,
|
||||
"WanVideoModelLoaderMultiGPU_2": WanVideoModelLoader_2,
|
||||
@@ -369,10 +285,8 @@ wanvideo_nodes = {
|
||||
}
|
||||
register_and_count(["ComfyUI-WanVideoWrapper", "comfyui-wanvideowrapper"], wanvideo_nodes)
|
||||
|
||||
# Print the registration table
|
||||
for item in registration_data:
|
||||
logger.info(fmt_reg.format(item['name'], item['found'], str(item['count'])))
|
||||
logger.info(dash_line)
|
||||
|
||||
|
||||
logger.info(f"[MultiGPU] Registration complete. Final mappings: {', '.join(NODE_CLASS_MAPPINGS.keys())}")
|
||||
logger.info(f"[MultiGPU] Registration complete. Final mappings: {', '.join(NODE_CLASS_MAPPINGS.keys())}")
|
||||
+22
-24
@@ -1,8 +1,3 @@
|
||||
"""
|
||||
Advanced Checkpoint Loaders for MultiGPU
|
||||
Provides device-specific and DisTorch2 sharding for checkpoint components
|
||||
"""
|
||||
|
||||
import torch
|
||||
import logging
|
||||
import hashlib
|
||||
@@ -13,6 +8,7 @@ import comfy.model_detection
|
||||
import comfy.clip_vision
|
||||
from comfy.sd import VAE, CLIP
|
||||
from .device_utils import get_device_list, soft_empty_cache_multigpu
|
||||
from .model_management_mgpu import multigpu_memory_log
|
||||
from .distorch_2 import safetensor_allocation_store, safetensor_settings_store, create_safetensor_model_hash, register_patched_safetensor_modelpatcher
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
@@ -23,24 +19,21 @@ checkpoint_distorch_config = {}
|
||||
original_load_state_dict_guess_config = None
|
||||
|
||||
def patch_load_state_dict_guess_config():
|
||||
"""
|
||||
Monkey patch the load_state_dict_guess_config function to replace its logic
|
||||
with a MultiGPU-aware implementation.
|
||||
"""
|
||||
"""Monkey patch comfy.sd.load_state_dict_guess_config with MultiGPU-aware checkpoint loading."""
|
||||
global original_load_state_dict_guess_config
|
||||
|
||||
if original_load_state_dict_guess_config is not None:
|
||||
logger.info("[MultiGPU] load_state_dict_guess_config is already patched.")
|
||||
logger.debug("[MultiGPU Checkpoint] load_state_dict_guess_config is already patched.")
|
||||
return
|
||||
|
||||
logger.info("[MultiGPU] Patching comfy.sd.load_state_dict_guess_config for advanced MultiGPU loading.")
|
||||
logger.info("[MultiGPU Core Patching] Patching comfy.sd.load_state_dict_guess_config for advanced MultiGPU loading.")
|
||||
original_load_state_dict_guess_config = comfy.sd.load_state_dict_guess_config
|
||||
comfy.sd.load_state_dict_guess_config = patched_load_state_dict_guess_config
|
||||
|
||||
def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True, output_clipvision=False,
|
||||
embedding_directory=None, output_model=True, model_options={},
|
||||
te_model_options={}, metadata=None):
|
||||
|
||||
"""Patched checkpoint loader with MultiGPU and DisTorch2 device placement support."""
|
||||
from . import set_current_device, set_current_text_encoder_device, current_device, current_text_encoder_device
|
||||
|
||||
sd_size = sum(p.numel() for p in sd.values() if hasattr(p, 'numel'))
|
||||
@@ -51,9 +44,9 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
if not device_config and not distorch_config:
|
||||
return original_load_state_dict_guess_config(sd, output_vae, output_clip, output_clipvision, embedding_directory, output_model, model_options, te_model_options, metadata)
|
||||
|
||||
logger.info("--- [MultiGPU] ENTERING Patched Checkpoint Loader ---")
|
||||
logger.info(f"Received Device Config: {device_config}")
|
||||
logger.info(f"Received DisTorch2 Config: {distorch_config}")
|
||||
logger.debug("[MultiGPU Checkpoint] ENTERING Patched Checkpoint Loader")
|
||||
logger.debug(f"[MultiGPU Checkpoint] Received Device Config: {device_config}")
|
||||
logger.debug(f"[MultiGPU Checkpoint] Received DisTorch2 Config: {distorch_config}")
|
||||
|
||||
clip = None
|
||||
clipvision = None
|
||||
@@ -63,7 +56,6 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
|
||||
original_main_device = current_device
|
||||
original_clip_device = current_text_encoder_device
|
||||
logger.info(f"Saved original device contexts: UNet/VAE='{original_main_device}', CLIP='{original_clip_device}'")
|
||||
|
||||
try:
|
||||
diffusion_model_prefix = comfy.model_detection.unet_prefix_from_state_dict(sd)
|
||||
@@ -80,7 +72,7 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
return None
|
||||
return (diffusion_model, None, VAE(sd={}), None)
|
||||
|
||||
logger.info(f"[MultiGPU] Detected Model Config: {type(model_config).__name__}, Parameters: {parameters/10**9:.2f}B")
|
||||
logger.debug(f"[MultiGPU] Detected Model Config: {type(model_config).__name__}, Parameters: {parameters/10**9:.2f}B")
|
||||
|
||||
unet_weight_dtype = list(model_config.supported_inference_dtypes)
|
||||
if model_config.scaled_fp8 is not None:
|
||||
@@ -105,10 +97,14 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
set_current_device(unet_compute_device)
|
||||
inital_load_device = mm.unet_inital_load_device(parameters, unet_dtype)
|
||||
|
||||
multigpu_memory_log(f"unet:{config_hash[:8]}", "pre-load")
|
||||
|
||||
model = model_config.get_model(sd, diffusion_model_prefix, device=inital_load_device)
|
||||
|
||||
soft_empty_cache_multigpu(logger)
|
||||
logger.mgpu_mm_log("Invoking soft_empty_cache_multigpu before UNet ModelPatcher setup")
|
||||
soft_empty_cache_multigpu()
|
||||
model_patcher = comfy.model_patcher.ModelPatcher(model, load_device=unet_compute_device, offload_device=mm.unet_offload_device())
|
||||
multigpu_memory_log(f"unet:{config_hash[:8]}", "post-model")
|
||||
|
||||
if distorch_config and 'unet_allocation' in distorch_config:
|
||||
register_patched_safetensor_modelpatcher()
|
||||
@@ -117,17 +113,20 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
safetensor_settings_store[model_hash] = distorch_config.get('unet_settings','')
|
||||
model.is_distorch = True
|
||||
model._distorch_high_precision_loras = distorch_config.get('high_precision_loras', True)
|
||||
logger.info(f"Stored DisTorch2 config for UNet (hash {model_hash[:8]}): {distorch_config['unet_allocation']}")
|
||||
logger.mgpu_mm_log(f"Stored DisTorch2 config for UNet (hash {model_hash[:8]}): {distorch_config['unet_allocation']}")
|
||||
|
||||
model.load_model_weights(sd, diffusion_model_prefix)
|
||||
multigpu_memory_log(f"unet:{config_hash[:8]}", "post-weights")
|
||||
|
||||
if output_vae:
|
||||
vae_target_device = torch.device(device_config.get('vae_device', original_main_device))
|
||||
set_current_device(vae_target_device) # Use main device context for VAE
|
||||
multigpu_memory_log(f"vae:{config_hash[:8]}", "pre-load")
|
||||
|
||||
vae_sd = comfy.utils.state_dict_prefix_replace(sd, {k: "" for k in model_config.vae_key_prefix}, filter_keys=True)
|
||||
vae_sd = model_config.process_vae_state_dict(vae_sd)
|
||||
vae = VAE(sd=vae_sd, metadata=metadata)
|
||||
multigpu_memory_log(f"vae:{config_hash[:8]}", "post-load")
|
||||
|
||||
if output_clip:
|
||||
clip_target_device = device_config.get('clip_device', original_clip_device)
|
||||
@@ -137,7 +136,9 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
if clip_target is not None:
|
||||
clip_sd = model_config.process_clip_state_dict(sd)
|
||||
if len(clip_sd) > 0:
|
||||
soft_empty_cache_multigpu(logger)
|
||||
logger.debug("[MultiGPU Checkpoint] Invoking soft_empty_cache_multigpu before CLIP construction")
|
||||
multigpu_memory_log(f"clip:{config_hash[:8]}", "pre-load")
|
||||
soft_empty_cache_multigpu()
|
||||
clip_params = comfy.utils.calculate_parameters(clip_sd)
|
||||
clip = CLIP(clip_target, embedding_directory=embedding_directory, tokenizer_data=clip_sd, parameters=clip_params, model_options=te_model_options)
|
||||
|
||||
@@ -155,22 +156,19 @@ def patched_load_state_dict_guess_config(sd, output_vae=True, output_clip=True,
|
||||
if len(m) > 0: logger.warning(f"CLIP missing keys: {m}")
|
||||
if len(u) > 0: logger.debug(f"CLIP unexpected keys: {u}")
|
||||
logger.info("CLIP Loaded.")
|
||||
multigpu_memory_log(f"clip:{config_hash[:8]}", "post-load")
|
||||
else:
|
||||
logger.warning("No CLIP/text encoder weights in checkpoint.")
|
||||
else:
|
||||
logger.warning("CLIP target not found in model config.")
|
||||
|
||||
finally:
|
||||
# --- Restore original device contexts and clean up ---
|
||||
set_current_device(original_main_device)
|
||||
set_current_text_encoder_device(original_clip_device)
|
||||
if config_hash in checkpoint_device_config:
|
||||
del checkpoint_device_config[config_hash]
|
||||
if config_hash in checkpoint_distorch_config:
|
||||
del checkpoint_distorch_config[config_hash]
|
||||
logger.info(f"Restored original device contexts. UNet/VAE='{original_main_device}', CLIP='{original_clip_device}'")
|
||||
logger.info("--- [MultiGPU] EXITING Patched Checkpoint Loader ---")
|
||||
|
||||
return (model_patcher, clip, vae, clipvision)
|
||||
|
||||
class CheckpointLoaderAdvancedMultiGPU:
|
||||
|
||||
+175
-400
@@ -1,18 +1,12 @@
|
||||
"""
|
||||
Device detection, management, and inspection utilities for ComfyUI-MultiGPU.
|
||||
Single source of truth for all device enumeration, compatibility checks, and state inspection.
|
||||
Handles all device types supported by ComfyUI core.
|
||||
"""
|
||||
|
||||
import torch
|
||||
import logging
|
||||
import hashlib
|
||||
import psutil
|
||||
import comfy.model_management as mm
|
||||
import gc
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
|
||||
# Module-level cache for device list (populated once on first call)
|
||||
_DEVICE_LIST_CACHE = None
|
||||
|
||||
def get_device_list():
|
||||
@@ -33,81 +27,61 @@ def get_device_list():
|
||||
"""
|
||||
global _DEVICE_LIST_CACHE
|
||||
|
||||
# Return cached result if already populated
|
||||
if _DEVICE_LIST_CACHE is not None:
|
||||
return _DEVICE_LIST_CACHE
|
||||
|
||||
# First time - do the actual detection
|
||||
devs = []
|
||||
|
||||
# CPU is always physically present and can store tensors
|
||||
devs.append("cpu")
|
||||
|
||||
# CUDA devices (NVIDIA GPUs)
|
||||
try:
|
||||
if hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available():
|
||||
device_count = torch.cuda.device_count()
|
||||
devs += [f"cuda:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} CUDA device(s)")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] CUDA detection failed: {e}")
|
||||
if hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available():
|
||||
device_count = torch.cuda.device_count()
|
||||
devs += [f"cuda:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} CUDA device(s)")
|
||||
|
||||
# XPU devices (Intel GPUs)
|
||||
try:
|
||||
# Try to import intel extension first (may be required for XPU support)
|
||||
import intel_extension_for_pytorch as ipex
|
||||
except ImportError:
|
||||
pass
|
||||
try:
|
||||
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
|
||||
device_count = torch.xpu.device_count()
|
||||
devs += [f"xpu:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} XPU device(s)")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] XPU detection failed: {e}")
|
||||
|
||||
# NPU devices (Ascend NPUs from Huawei)
|
||||
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
|
||||
device_count = torch.xpu.device_count()
|
||||
devs += [f"xpu:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} XPU device(s)")
|
||||
|
||||
try:
|
||||
import torch_npu
|
||||
if hasattr(torch, "npu") and hasattr(torch.npu, "is_available") and torch.npu.is_available():
|
||||
device_count = torch.npu.device_count()
|
||||
devs += [f"npu:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} NPU device(s)")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] NPU detection failed: {e}")
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# MLU devices (Cambricon MLUs)
|
||||
try:
|
||||
import torch_mlu
|
||||
if hasattr(torch, "mlu") and hasattr(torch.mlu, "is_available") and torch.mlu.is_available():
|
||||
device_count = torch.mlu.device_count()
|
||||
devs += [f"mlu:{i}" for i in range(device_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {device_count} MLU device(s)")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] MLU detection failed: {e}")
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# MPS device (Apple Metal - single device only)
|
||||
try:
|
||||
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
devs.append("mps")
|
||||
logger.debug("[MultiGPU_Device_Utils] Found MPS device")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] MPS detection failed: {e}")
|
||||
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
devs.append("mps")
|
||||
logger.debug("[MultiGPU_Device_Utils] Found MPS device")
|
||||
|
||||
# DirectML devices (Windows DirectML for AMD/Intel/NVIDIA)
|
||||
try:
|
||||
import torch_directml
|
||||
adapter_count = torch_directml.device_count()
|
||||
if adapter_count > 0:
|
||||
devs += [f"directml:{i}" for i in range(adapter_count)]
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Found {adapter_count} DirectML adapter(s)")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] DirectML detection failed: {e}")
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# IXUCA/CoreX devices (special accelerator)
|
||||
try:
|
||||
if hasattr(torch, "corex"):
|
||||
# CoreX typically exposes single device, but check if there's a count method
|
||||
if hasattr(torch.corex, "device_count"):
|
||||
device_count = torch.corex.device_count()
|
||||
devs += [f"corex:{i}" for i in range(device_count)]
|
||||
@@ -115,431 +89,232 @@ def get_device_list():
|
||||
else:
|
||||
devs.append("corex:0")
|
||||
logger.debug("[MultiGPU_Device_Utils] Found CoreX device")
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Device_Utils] CoreX detection failed: {e}")
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# Cache the result for future calls
|
||||
_DEVICE_LIST_CACHE = devs
|
||||
|
||||
# Log only once when initially populated
|
||||
logger.info(f"[MultiGPU_Device_Utils] Device list initialized: {devs}")
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Device list initialized: {devs}")
|
||||
|
||||
return devs
|
||||
|
||||
|
||||
def is_accelerator_available():
|
||||
"""
|
||||
Check if any accelerator device is available.
|
||||
Used by patched functions to determine CPU fallback.
|
||||
"""Check if any GPU or accelerator device is available including CUDA, XPU, NPU, MLU, MPS, DirectML, or CoreX."""
|
||||
if hasattr(torch, "cuda") and torch.cuda.is_available():
|
||||
return True
|
||||
|
||||
Returns True if any GPU/accelerator is available, False otherwise.
|
||||
"""
|
||||
# Check CUDA
|
||||
try:
|
||||
if torch.cuda.is_available():
|
||||
return True
|
||||
except:
|
||||
pass
|
||||
if hasattr(torch, "xpu") and hasattr(torch.xpu, "is_available") and torch.xpu.is_available():
|
||||
return True
|
||||
|
||||
# Check XPU (Intel GPU)
|
||||
try:
|
||||
if hasattr(torch, "xpu") and torch.xpu.is_available():
|
||||
return True
|
||||
except:
|
||||
pass
|
||||
|
||||
# Check NPU (Ascend)
|
||||
try:
|
||||
import torch_npu
|
||||
if hasattr(torch, "npu") and torch.npu.is_available():
|
||||
if hasattr(torch, "npu") and hasattr(torch.npu, "is_available") and torch.npu.is_available():
|
||||
return True
|
||||
except:
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# Check MLU (Cambricon)
|
||||
|
||||
try:
|
||||
import torch_mlu
|
||||
if hasattr(torch, "mlu") and torch.mlu.is_available():
|
||||
if hasattr(torch, "mlu") and hasattr(torch.mlu, "is_available") and torch.mlu.is_available():
|
||||
return True
|
||||
except:
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
# Check MPS (Apple Metal)
|
||||
try:
|
||||
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
return True
|
||||
except:
|
||||
pass
|
||||
|
||||
# Check DirectML
|
||||
|
||||
if hasattr(torch.backends, "mps") and torch.backends.mps.is_available():
|
||||
return True
|
||||
|
||||
try:
|
||||
import torch_directml
|
||||
if torch_directml.device_count() > 0:
|
||||
return True
|
||||
except:
|
||||
pass
|
||||
|
||||
# Check CoreX/IXUCA
|
||||
try:
|
||||
if hasattr(torch, "corex"):
|
||||
return True
|
||||
except:
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
if hasattr(torch, "corex"):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def is_device_compatible(device_string):
|
||||
"""
|
||||
Check if a device string represents a valid, available device.
|
||||
|
||||
Args:
|
||||
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
|
||||
|
||||
Returns:
|
||||
True if the device is available, False otherwise.
|
||||
"""
|
||||
"""Check if a device string represents a valid available device."""
|
||||
available_devices = get_device_list()
|
||||
return device_string in available_devices
|
||||
|
||||
|
||||
def get_device_type(device_string):
|
||||
"""
|
||||
Extract the device type from a device string.
|
||||
|
||||
Args:
|
||||
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
|
||||
|
||||
Returns:
|
||||
Device type string (e.g., "cuda", "cpu", "xpu", "npu", "mlu", "mps", "directml", "corex")
|
||||
"""
|
||||
"""Extract device type from device string (e.g. 'cuda' from 'cuda:0')."""
|
||||
if ":" in device_string:
|
||||
return device_string.split(":")[0]
|
||||
return device_string
|
||||
|
||||
|
||||
def parse_device_string(device_string):
|
||||
"""
|
||||
Parse a device string into type and index.
|
||||
|
||||
Args:
|
||||
device_string: Device identifier like "cuda:0", "cpu", "xpu:1", etc.
|
||||
|
||||
Returns:
|
||||
Tuple of (device_type, device_index) where index is None for non-indexed devices
|
||||
"""
|
||||
"""Parse device string into (device_type, device_index) tuple."""
|
||||
if ":" in device_string:
|
||||
parts = device_string.split(":")
|
||||
return parts[0], int(parts[1])
|
||||
return device_string, None
|
||||
|
||||
def soft_empty_cache_multigpu():
|
||||
"""Clear allocator caches across all devices using context managers to preserve calling thread device context."""
|
||||
from .model_management_mgpu import multigpu_memory_log
|
||||
|
||||
logger.mgpu_mm_log("soft_empty_cache_multigpu: starting GC and multi-device cache clear")
|
||||
|
||||
def soft_empty_cache_multigpu(logger):
|
||||
"""
|
||||
Replicate ComfyUI's cache clearing but for ALL devices in MultiGPU.
|
||||
MultiGPU adaptation of ComfyUI's soft_empty_cache() functionality.
|
||||
"""
|
||||
import gc
|
||||
|
||||
logger.info("[MultiGPU_Device_Utils] Preparing devices for optimized safetensor loading")
|
||||
|
||||
# Python GC (same as all implementations)
|
||||
gc.collect()
|
||||
logger.debug("[MultiGPU_Device_Utils] Performed garbage collection before safetensor loading")
|
||||
|
||||
# Clear cache for ALL devices (not just ComfyUI's single device)
|
||||
all_devices = get_device_list()
|
||||
logger.mgpu_mm_log(f"soft_empty_cache_multigpu: devices to clear = {all_devices}")
|
||||
|
||||
# Check global availability first to avoid unnecessary iteration if backend is missing
|
||||
is_cuda_available = hasattr(torch, "cuda") and hasattr(torch.cuda, "is_available") and torch.cuda.is_available()
|
||||
|
||||
for device_str in all_devices:
|
||||
if device_str.startswith("cuda:"):
|
||||
device_idx = int(device_str.split(":")[1])
|
||||
torch.cuda.set_device(device_idx)
|
||||
torch.cuda.empty_cache()
|
||||
torch.cuda.ipc_collect() # ComfyUI's CUDA optimization
|
||||
logger.debug(f"[MultiGPU_Device_Utils] Cleared cache + IPC for {device_str}")
|
||||
if is_cuda_available:
|
||||
device_idx = int(device_str.split(":")[1])
|
||||
logger.mgpu_mm_log(f"Clearing CUDA cache on {device_str} (idx={device_idx})")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
with torch.cuda.device(device_idx):
|
||||
torch.cuda.empty_cache()
|
||||
if hasattr(torch.cuda, "ipc_collect"):
|
||||
torch.cuda.ipc_collect()
|
||||
logger.mgpu_mm_log(f"Cleared CUDA cache (and IPC if available) on {device_str}")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
elif device_str == "mps":
|
||||
torch.mps.empty_cache()
|
||||
logger.debug("[MultiGPU_Device_Utils] Cleared cache for MPS")
|
||||
if hasattr(torch, "mps") and hasattr(torch.mps, "empty_cache"):
|
||||
logger.mgpu_mm_log("Clearing MPS cache")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
torch.mps.empty_cache()
|
||||
logger.mgpu_mm_log("Cleared MPS cache")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
elif device_str.startswith("xpu:"):
|
||||
torch.xpu.empty_cache()
|
||||
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Intel XPU")
|
||||
if hasattr(torch, "xpu") and hasattr(torch.xpu, "empty_cache"):
|
||||
logger.mgpu_mm_log(f"Clearing XPU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
torch.xpu.empty_cache()
|
||||
logger.mgpu_mm_log(f"Cleared XPU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
elif device_str.startswith("npu:"):
|
||||
torch.npu.empty_cache()
|
||||
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Ascend NPU")
|
||||
if hasattr(torch, "npu") and hasattr(torch.npu, "empty_cache"):
|
||||
logger.mgpu_mm_log(f"Clearing NPU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
torch.npu.empty_cache()
|
||||
logger.mgpu_mm_log(f"Cleared NPU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
elif device_str.startswith("mlu:"):
|
||||
torch.mlu.empty_cache()
|
||||
logger.debug("[MultiGPU_Device_Utils] Cleared cache for Cambricon MLU")
|
||||
if hasattr(torch, "mlu") and hasattr(torch.mlu, "empty_cache"):
|
||||
logger.mgpu_mm_log(f"Clearing MLU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
torch.mlu.empty_cache()
|
||||
logger.mgpu_mm_log(f"Cleared MLU cache on {device_str}")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
elif device_str.startswith("corex:"):
|
||||
torch.corex.empty_cache() # Hypothetical based on ComfyUI's ixuca support
|
||||
logger.debug("[MultiGPU_Device_Utils] Cleared cache for CoreX")
|
||||
if hasattr(torch, "corex") and hasattr(torch.corex, "empty_cache"):
|
||||
logger.mgpu_mm_log(f"Clearing CoreX cache on {device_str}")
|
||||
multigpu_memory_log("general", f"pre-empty:{device_str}")
|
||||
torch.corex.empty_cache()
|
||||
logger.mgpu_mm_log(f"Cleared CoreX cache on {device_str}")
|
||||
multigpu_memory_log("general", f"post-empty:{device_str}")
|
||||
|
||||
multigpu_memory_log("general", "post-soft-empty")
|
||||
|
||||
|
||||
# ==========================================================================================
|
||||
# Model Management Inspection Utilities (End-to-End Tracking)
|
||||
# Comprehensive Memory Management (VRAM + CPU + Store Pruning)
|
||||
# ==========================================================================================
|
||||
|
||||
def create_model_identifier(model_patcher):
|
||||
"""Creates a concise, unique identifier for a model patcher based on type and size."""
|
||||
if not model_patcher or not model_patcher.model:
|
||||
return "N/A (Detached)"
|
||||
logger.info("[MultiGPU Core Patching] Patching mm.soft_empty_cache for Comprehensive Memory Management (VRAM + CPU + Store Pruning)")
|
||||
|
||||
model = model_patcher.model
|
||||
model_type = type(model).__name__
|
||||
original_soft_empty_cache = mm.soft_empty_cache
|
||||
|
||||
# Try the fast path first (using size calculated by ModelPatcher)
|
||||
try:
|
||||
model_size = model_patcher.model_size()
|
||||
except Exception:
|
||||
model_size = 0
|
||||
def soft_empty_cache_distorch2_patched(force=False):
|
||||
"""Patched mm.soft_empty_cache managing VRAM across all devices, CPU RAM with adaptive thresholding, and DisTorch store pruning."""
|
||||
from .model_management_mgpu import multigpu_memory_log, check_cpu_memory_threshold, trigger_executor_cache_reset
|
||||
from .distorch_2 import safetensor_allocation_store, create_safetensor_model_hash
|
||||
|
||||
multigpu_memory_log("patched_soft_empty", f"start:force={force}")
|
||||
is_distorch_active = False
|
||||
|
||||
# If the fast path fails or returns 0, perform a safe deep inspection
|
||||
if model_size == 0:
|
||||
try:
|
||||
# Safely inspect parameters without triggering hooks/loads
|
||||
with model_patcher.use_ejected(skip_and_inject_on_exit_only=True):
|
||||
# We must iterate parameters() AND buffers() as both consume memory
|
||||
params = list(model.parameters()) + list(model.buffers())
|
||||
# Use data_ptr to handle potential weight tying/shared tensors correctly
|
||||
seen_tensors = set()
|
||||
for p in params:
|
||||
if p.data_ptr() not in seen_tensors:
|
||||
model_size += p.numel() * p.element_size()
|
||||
seen_tensors.add(p.data_ptr())
|
||||
except Exception as e:
|
||||
logger.debug(f"[MultiGPU_Inspection] Error during safe size calculation for identifier: {e}")
|
||||
return f"{model_type} (ID_Err)"
|
||||
# Detect DisTorch2-managed models
|
||||
logger.mgpu_mm_log(f"[DETECT_DEBUG] Checking DisTorch2 active status - loaded models: {len(mm.current_loaded_models)}, store entries: {len(safetensor_allocation_store)}")
|
||||
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
mp = lm.model # weakref call to ModelPatcher
|
||||
if mp is not None:
|
||||
try:
|
||||
model_hash = create_safetensor_model_hash(mp, "cache_patch_check")
|
||||
in_store = model_hash in safetensor_allocation_store
|
||||
alloc_value = safetensor_allocation_store.get(model_hash, "")
|
||||
model_name = type(getattr(mp, 'model', mp)).__name__
|
||||
unload_distorch_model = getattr(getattr(mp, 'model', None), '_mgpu_unload_distorch_model', False)
|
||||
|
||||
logger.mgpu_mm_log(f"[DETECT_DEBUG] Model {i}: {model_name}, hash={model_hash[:8]}, in_store={in_store}, alloc_value='{alloc_value}', unload_distorch_model={unload_distorch_model}")
|
||||
|
||||
if in_store and alloc_value:
|
||||
is_distorch_active = True
|
||||
logger.mgpu_mm_log(f"[DETECT_DEBUG] DisTorch2 ACTIVE detected on model: {model_name}")
|
||||
break
|
||||
except Exception as e:
|
||||
logger.mgpu_mm_log(f"[DETECT_DEBUG] Model {i}: Error during detection - {e}")
|
||||
|
||||
logger.mgpu_mm_log(f"[DETECT_DEBUG] Final DisTorch2 active status: {is_distorch_active}")
|
||||
|
||||
# Create a hash based on type and calculated size
|
||||
identifier = f"{model_type}_{model_size}"
|
||||
model_hash = hashlib.sha256(identifier.encode()).hexdigest()
|
||||
return f"{model_type} ({model_hash[:8]})"
|
||||
# Phase 2: adaptive CPU memory management
|
||||
check_cpu_memory_threshold()
|
||||
|
||||
# VRAM allocator management
|
||||
if is_distorch_active:
|
||||
logger.mgpu_mm_log("DisTorch2 active: clearing allocator caches on all devices (VRAM)")
|
||||
soft_empty_cache_multigpu()
|
||||
else:
|
||||
logger.mgpu_mm_log("DisTorch2 not active: delegating allocator cache clear (VRAM) to original mm.soft_empty_cache")
|
||||
original_soft_empty_cache(force)
|
||||
# Optional: return CPU heap to OS (not part of Comfy Core)
|
||||
|
||||
def analyze_tensor_locations(model_patcher):
|
||||
"""
|
||||
Analyzes the physical device placement of model tensors (parameters and buffers).
|
||||
This provides the Ground Truth location of the data, handling shared weights correctly.
|
||||
"""
|
||||
device_summary = {}
|
||||
seen_tensors = set()
|
||||
total_memory = 0
|
||||
# Phase 1/3: forced executor reset mirrors ComfyUI 'Free memory' semantics
|
||||
if force:
|
||||
logger.mgpu_mm_log("Force flag active: triggering executor cache reset (CPU)")
|
||||
trigger_executor_cache_reset(reason="forced_soft_empty", force=True)
|
||||
multigpu_memory_log("patched_soft_empty", "end")
|
||||
|
||||
if not model_patcher or not model_patcher.model:
|
||||
return {"error": "Model not available"}, 0
|
||||
mm.soft_empty_cache = soft_empty_cache_distorch2_patched
|
||||
|
||||
model = model_patcher.model
|
||||
# ==========================================================================================
|
||||
# Memory Inspection Utilities
|
||||
# ==========================================================================================
|
||||
|
||||
# Crucial: Use the ejector to ensure we can access the model weights safely
|
||||
# without interfering with injections, hooks, or triggering unintended loads (like in standard LowVRAM mode).
|
||||
try:
|
||||
with model_patcher.use_ejected(skip_and_inject_on_exit_only=True):
|
||||
# Helper to process tensors (parameters or buffers)
|
||||
def process_tensor(tensor):
|
||||
nonlocal total_memory
|
||||
# Use data_ptr() for unique identification of the underlying memory
|
||||
if tensor.data_ptr() in seen_tensors:
|
||||
return
|
||||
seen_tensors.add(tensor.data_ptr())
|
||||
def comfyui_memory_load(tag):
|
||||
"""Return single-line pipe-delimited snapshot of system and device memory usage in GiB."""
|
||||
# CPU RAM
|
||||
vm = psutil.virtual_memory()
|
||||
cpu_used_gib = vm.used / (1024.0 ** 3)
|
||||
cpu_total_gib = vm.total / (1024.0 ** 3)
|
||||
|
||||
if tensor.numel() > 0:
|
||||
tensor_mem = tensor.numel() * tensor.element_size()
|
||||
total_memory += tensor_mem
|
||||
segments = [f"tag={tag}", f"cpu={cpu_used_gib:.2f}/{cpu_total_gib:.2f}"]
|
||||
|
||||
if hasattr(tensor, 'device'):
|
||||
device = str(tensor.device)
|
||||
else:
|
||||
# Handle cases like NF4 quantization or other custom tensors
|
||||
device = "Unknown/Managed"
|
||||
# Enumerate non-CPU devices
|
||||
devices = [d for d in get_device_list() if d != "cpu"]
|
||||
|
||||
if device not in device_summary:
|
||||
device_summary[device] = {'tensors': 0, 'memory': 0}
|
||||
|
||||
device_summary[device]['tensors'] += 1
|
||||
device_summary[device]['memory'] += tensor_mem
|
||||
|
||||
# Iterate over all parameters (weights, biases)
|
||||
for param in model.parameters():
|
||||
process_tensor(param)
|
||||
|
||||
# Iterate over all buffers (like batch norm running stats)
|
||||
for buffer in model.buffers():
|
||||
process_tensor(buffer)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"[MultiGPU_Inspection] Error during tensor location analysis: {e}")
|
||||
return {"error": str(e)}, 0
|
||||
|
||||
return device_summary, total_memory
|
||||
|
||||
|
||||
def inspect_model_management_state(context_description=""):
|
||||
"""
|
||||
Provides a detailed, structured overview of the current state of ComfyUI's model management,
|
||||
including memory usage across all devices and the status, location, and patching of all loaded models.
|
||||
|
||||
Call this function anywhere in the code to get an immediate snapshot of the system state.
|
||||
"""
|
||||
|
||||
# Ensure logger configuration (handles calls before full MultiGPU init if needed)
|
||||
if not logger.handlers:
|
||||
handler = logging.StreamHandler()
|
||||
formatter = logging.Formatter('%(message)s')
|
||||
handler.setFormatter(formatter)
|
||||
logger.addHandler(handler)
|
||||
# Default to INFO if log level isn't set by main __init__.py
|
||||
if logger.level == logging.NOTSET:
|
||||
logger.setLevel(logging.INFO)
|
||||
|
||||
# We inspect the state without forcing GC or cache clearing, which might alter the state we want to observe.
|
||||
|
||||
logger.info("\n" + "=" * 100)
|
||||
logger.info(f" INSPECTION: ComfyUI Model Management State [Context: {context_description}]")
|
||||
logger.info("=" * 100)
|
||||
|
||||
# 1. Device Memory Overview
|
||||
# Provides context on available resources across the system.
|
||||
logger.info("--- [1] System Device Memory Overview (GB) ---")
|
||||
# Sys Free: Memory available to the OS. Torch Alloc: Memory reserved by PyTorch (Active + Cache).
|
||||
fmt_mem = "{:<12} | {:>10} | {:>10} | {:>10} | {:>15}"
|
||||
logger.info(fmt_mem.format("Device", "Total", "Sys Free", "Used", "Torch Alloc"))
|
||||
logger.info("-" * 70)
|
||||
|
||||
all_devices = get_device_list()
|
||||
# Sort devices for consistent display (CPU last)
|
||||
sorted_devices = sorted(all_devices, key=lambda d: (d == 'cpu', d))
|
||||
|
||||
for dev_str in sorted_devices:
|
||||
try:
|
||||
device = torch.device(dev_str)
|
||||
|
||||
if dev_str == "cpu":
|
||||
vm = psutil.virtual_memory()
|
||||
mem_total, mem_free_sys, mem_used = vm.total, vm.available, vm.used
|
||||
torch_alloc = 0 # Difficult to track accurately for CPU globally
|
||||
else:
|
||||
# Use ComfyUI's management functions which account for different backends (CUDA, XPU, etc.)
|
||||
mem_total = mm.get_total_memory(device)
|
||||
|
||||
# get_free_memory returns (system_free, torch_cache_free)
|
||||
free_info = mm.get_free_memory(device, torch_free_too=True)
|
||||
if isinstance(free_info, tuple):
|
||||
mem_free_sys = free_info[0]
|
||||
else:
|
||||
mem_free_sys = free_info # Fallback for backends that return single value (like MPS)
|
||||
|
||||
mem_used = mem_total - mem_free_sys
|
||||
|
||||
# Determine Torch Allocation (Reserved memory) - Specific checks for known backends
|
||||
torch_alloc = 0
|
||||
if device.type == 'cuda' and hasattr(torch.cuda, 'memory_stats'):
|
||||
stats = torch.cuda.memory_stats(device)
|
||||
torch_alloc = stats.get('reserved_bytes.all.current', 0)
|
||||
elif device.type == 'xpu' and hasattr(torch, 'xpu') and hasattr(torch.xpu, 'memory_stats'):
|
||||
stats = torch.xpu.memory_stats(device)
|
||||
torch_alloc = stats.get('reserved_bytes.all.current', 0)
|
||||
elif device.type == 'npu' and hasattr(torch, 'npu') and hasattr(torch.npu, 'memory_stats'):
|
||||
stats = torch.npu.memory_stats(device)
|
||||
torch_alloc = stats.get('reserved_bytes.all.current', 0)
|
||||
elif device.type == 'mlu' and hasattr(torch, 'mlu') and hasattr(torch.mlu, 'memory_stats'):
|
||||
stats = torch.mlu.memory_stats(device)
|
||||
torch_alloc = stats.get('reserved_bytes.all.current', 0)
|
||||
# MPS, DirectML, CoreX do not always expose detailed reserved memory stats easily.
|
||||
|
||||
logger.info(fmt_mem.format(
|
||||
dev_str,
|
||||
f"{mem_total / (1024**3):.2f}",
|
||||
f"{mem_free_sys / (1024**3):.2f}",
|
||||
f"{mem_used / (1024**3):.2f}",
|
||||
f"{torch_alloc / (1024**3):.2f}"
|
||||
))
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not retrieve memory stats for {dev_str}: {e}")
|
||||
|
||||
logger.info("-" * 70)
|
||||
|
||||
# 2. Loaded Models Inspection (Logical and Physical View)
|
||||
# mm.current_loaded_models holds the list of models ComfyUI is managing.
|
||||
loaded_models = mm.current_loaded_models
|
||||
logger.info(f"\n--- [2] Loaded Models Inspection (Count: {len(loaded_models)}) ---")
|
||||
|
||||
if not loaded_models:
|
||||
logger.info("No models currently managed by comfy.model_management.")
|
||||
logger.info("=" * 100)
|
||||
return
|
||||
|
||||
for i, lm in enumerate(loaded_models):
|
||||
logger.info(f"\nModel {i+1}/{len(loaded_models)}:")
|
||||
|
||||
# Check lifecycle status
|
||||
mp = lm.model # weakref call to ModelPatcher
|
||||
if mp is None:
|
||||
# ModelPatcher is gone. Check if the underlying model is still alive (potential leak)
|
||||
if lm.is_dead() and lm.real_model() is not None:
|
||||
logger.warning(f" [!] Status: LEAK DETECTED (Patcher GC'd, but underlying model {lm.real_model().__class__.__name__} persists)")
|
||||
else:
|
||||
logger.info(f" Status: Cleaned Up (Patcher and Model GC'd)")
|
||||
continue
|
||||
|
||||
model_id = create_model_identifier(mp)
|
||||
logger.info(f" Identifier: {model_id}")
|
||||
logger.info(f" Status: {'Active (In Use)' if lm.currently_used else 'Idle (Cache)'}")
|
||||
|
||||
# A. Logical View (What ComfyUI intends/tracks)
|
||||
logger.info(" [A] Logical View (ComfyUI Tracking):")
|
||||
|
||||
# Devices: Target (Compute) vs Offload (Storage)
|
||||
logger.info(f" Devices: Target={lm.device} | Offload={mp.offload_device} | Current (Model.device)={mp.current_loaded_device()}")
|
||||
|
||||
# Memory Footprint
|
||||
mem_total = lm.model_memory()
|
||||
mem_loaded = lm.model_loaded_memory()
|
||||
mem_offloaded = lm.model_offloaded_memory()
|
||||
logger.info(f" Memory (MB): Total={mem_total/(1024**2):.2f} | Loaded (on Target)={mem_loaded/(1024**2):.2f} | Offloaded={mem_offloaded/(1024**2):.2f}")
|
||||
|
||||
# Management Mode (LowVRAM/DisTorch)
|
||||
# model_lowvram indicates if ComfyUI is managing this model partially
|
||||
is_lowvram = getattr(mp.model, 'model_lowvram', False)
|
||||
lowvram_patches_pending = mp.lowvram_patch_counter()
|
||||
logger.info(f" Mode: {'Partial Load (LowVRAM/DisTorch)' if is_lowvram else 'Full Load'}")
|
||||
if is_lowvram:
|
||||
# This indicates how many weights are being managed by the partial loading system
|
||||
logger.info(f" Weights Managed by LowVRAM/DisTorch System: {lowvram_patches_pending}")
|
||||
|
||||
# Patching (LoRAs, etc.) - Tracking Attach/Detach
|
||||
num_weight_patches = len(mp.patches)
|
||||
# Check the UUID applied to the actual weights vs the UUID defined in the patcher
|
||||
current_weight_uuid = getattr(mp.model, 'current_weight_patches_uuid', None)
|
||||
weights_synced = (mp.patches_uuid == current_weight_uuid) and (current_weight_uuid is not None)
|
||||
|
||||
if num_weight_patches > 0:
|
||||
status = 'Applied & Synced' if weights_synced else 'Pending/Mismatch (Re-patch needed)'
|
||||
logger.info(f" Patches: {num_weight_patches} weight patches defined | Status: {status}")
|
||||
logger.info(f" UUIDs: Defined={str(mp.patches_uuid)[:8]}... | Applied={str(current_weight_uuid)[:8] if current_weight_uuid else 'None'}...")
|
||||
|
||||
# B. Physical View (Ground Truth Tensor Locations)
|
||||
logger.info(" [B] Physical View (Ground Truth Tensor Locations):")
|
||||
device_summary, calculated_total_mem = analyze_tensor_locations(mp)
|
||||
|
||||
if "error" in device_summary:
|
||||
logger.error(f" Analysis Error: {device_summary['error']}")
|
||||
continue
|
||||
|
||||
if not device_summary:
|
||||
logger.info(" No tensors found (e.g., fully offloaded CLIP or utility object).")
|
||||
# Append per-device VRAM used/total
|
||||
for dev_str in devices:
|
||||
device = torch.device(dev_str)
|
||||
total = mm.get_total_memory(device)
|
||||
free_info = mm.get_free_memory(device, torch_free_too=True)
|
||||
# free_info may be a tuple (system_free, torch_cache_free) or a single value
|
||||
if isinstance(free_info, tuple):
|
||||
system_free = free_info[0]
|
||||
else:
|
||||
# Sort devices (CPU last)
|
||||
sorted_devices = sorted(device_summary.keys(), key=lambda d: (d.startswith("cpu"), d))
|
||||
fmt_loc = " {:<15} | Tensors: {:>6} | Memory (MB): {:>10.2f} | Percent: {:>6.1f}%"
|
||||
for device in sorted_devices:
|
||||
data = device_summary[device]
|
||||
percent = (data['memory'] / calculated_total_mem) * 100 if calculated_total_mem > 0 else 0
|
||||
logger.info(fmt_loc.format(device, data['tensors'], data['memory']/(1024**2), percent))
|
||||
system_free = free_info
|
||||
used = max(0, (total or 0) - (system_free or 0))
|
||||
|
||||
# Verification Check
|
||||
if abs(calculated_total_mem - mem_total) > (1024*1024): # Allow 1MB difference
|
||||
logger.warning(f" [!] Verification WARNING: Physical memory ({calculated_total_mem/(1024**2):.2f}MB) differs from logical memory ({mem_total/(1024**2):.2f}MB).")
|
||||
used_gib = used / (1024.0 ** 3)
|
||||
total_gib = (total or 0) / (1024.0 ** 3)
|
||||
if total_gib > 0:
|
||||
segments.append(f"{dev_str}={used_gib:.2f}/{total_gib:.2f}")
|
||||
|
||||
logger.info("-" * 100)
|
||||
|
||||
logger.info("End of Inspection")
|
||||
logger.info("=" * 100)
|
||||
return "|".join(segments)
|
||||
|
||||
-525
@@ -1,525 +0,0 @@
|
||||
"""
|
||||
DisTorch GGUF/GGML Memory Management Module
|
||||
Contains all GGUF/GGML related code for distributed memory management
|
||||
"""
|
||||
|
||||
import sys
|
||||
import torch
|
||||
import logging
|
||||
import hashlib
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
import copy
|
||||
from collections import defaultdict
|
||||
import comfy.model_management as mm
|
||||
from .device_utils import get_device_list, soft_empty_cache_multigpu
|
||||
|
||||
# Global store for model allocations
|
||||
model_allocation_store = {}
|
||||
|
||||
|
||||
def create_model_hash(model, caller):
|
||||
"""Create a unique hash for a model to track allocations"""
|
||||
model_type = type(model.model).__name__
|
||||
model_size = model.model_size()
|
||||
first_layers = str(list(model.model_state_dict().keys())[:3])
|
||||
identifier = f"{model_type}_{model_size}_{first_layers}"
|
||||
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
|
||||
logger.debug(f"[MultiGPU_DisTorch_HASH] Created hash for {caller}: {final_hash[:8]}...")
|
||||
return final_hash
|
||||
|
||||
|
||||
def register_patched_ggufmodelpatcher():
|
||||
"""Register and patch the GGUFModelPatcher for distributed loading"""
|
||||
from nodes import NODE_CLASS_MAPPINGS
|
||||
original_loader = NODE_CLASS_MAPPINGS["UnetLoaderGGUF"]
|
||||
module = sys.modules[original_loader.__module__]
|
||||
|
||||
if not hasattr(module.GGUFModelPatcher, '_patched'):
|
||||
original_load = module.GGUFModelPatcher.load
|
||||
|
||||
def new_load(self, *args, force_patch_weights=False, **kwargs):
|
||||
global model_allocation_store
|
||||
|
||||
super(module.GGUFModelPatcher, self).load(*args, force_patch_weights=True, **kwargs)
|
||||
debug_hash = create_model_hash(self, "patcher")
|
||||
linked = []
|
||||
module_count = 0
|
||||
for n, m in self.model.named_modules():
|
||||
module_count += 1
|
||||
if hasattr(m, "weight"):
|
||||
device = getattr(m.weight, "device", None)
|
||||
if device is not None:
|
||||
linked.append((n, m))
|
||||
continue
|
||||
if hasattr(m, "bias"):
|
||||
device = getattr(m.bias, "device", None)
|
||||
if device is not None:
|
||||
linked.append((n, m))
|
||||
continue
|
||||
if linked:
|
||||
if hasattr(self, 'model'):
|
||||
debug_hash = create_model_hash(self, "patcher")
|
||||
debug_allocations = model_allocation_store.get(debug_hash)
|
||||
if debug_allocations:
|
||||
soft_empty_cache_multigpu(logger)
|
||||
device_assignments = analyze_ggml_loading(self.model, debug_allocations)['device_assignments']
|
||||
for device, layers in device_assignments.items():
|
||||
target_device = torch.device(device)
|
||||
for n, m, _ in layers:
|
||||
m.to(self.load_device).to(target_device)
|
||||
|
||||
self.mmap_released = True
|
||||
|
||||
module.GGUFModelPatcher.load = new_load
|
||||
module.GGUFModelPatcher._patched = True
|
||||
|
||||
|
||||
def analyze_ggml_loading(model, allocations_str):
|
||||
"""Analyze and distribute GGML model layers across devices"""
|
||||
DEVICE_RATIOS_DISTORCH = {}
|
||||
device_table = {}
|
||||
distorch_alloc = allocations_str
|
||||
virtual_vram_gb = 0.0
|
||||
|
||||
if '#' in allocations_str:
|
||||
distorch_alloc, virtual_vram_str = allocations_str.split('#')
|
||||
if not distorch_alloc:
|
||||
distorch_alloc = calculate_vvram_allocation_string(model, virtual_vram_str)
|
||||
|
||||
eq_line = "=" * 47
|
||||
dash_line = "-" * 47
|
||||
fmt_assign = "{:<12}{:>10}{:>14}{:>10}"
|
||||
|
||||
for allocation in distorch_alloc.split(';'):
|
||||
dev_name, fraction = allocation.split(',')
|
||||
fraction = float(fraction)
|
||||
total_mem_bytes = mm.get_total_memory(torch.device(dev_name))
|
||||
alloc_gb = (total_mem_bytes * fraction) / (1024**3)
|
||||
DEVICE_RATIOS_DISTORCH[dev_name] = alloc_gb
|
||||
device_table[dev_name] = {
|
||||
"fraction": fraction,
|
||||
"total_gb": total_mem_bytes / (1024**3),
|
||||
"alloc_gb": alloc_gb
|
||||
}
|
||||
|
||||
logger.info(eq_line)
|
||||
logger.info(" DisTorch Model Device Allocations")
|
||||
logger.info(eq_line)
|
||||
logger.info(fmt_assign.format("Device", "Alloc %", "Total (GB)", " Alloc (GB)"))
|
||||
logger.info(dash_line)
|
||||
|
||||
sorted_devices = sorted(device_table.keys(), key=lambda d: (d == "cpu", d))
|
||||
|
||||
for dev in sorted_devices:
|
||||
frac = device_table[dev]["fraction"]
|
||||
tot_gb = device_table[dev]["total_gb"]
|
||||
alloc_gb = device_table[dev]["alloc_gb"]
|
||||
logger.info(fmt_assign.format(dev,f"{int(frac * 100)}%",f"{tot_gb:.2f}",f"{alloc_gb:.2f}"))
|
||||
|
||||
logger.info(dash_line)
|
||||
|
||||
layer_summary = {}
|
||||
layer_list = []
|
||||
memory_by_type = defaultdict(int)
|
||||
total_memory = 0
|
||||
|
||||
for name, module in model.named_modules():
|
||||
if hasattr(module, "weight"):
|
||||
layer_type = type(module).__name__
|
||||
layer_summary[layer_type] = layer_summary.get(layer_type, 0) + 1
|
||||
layer_list.append((name, module, layer_type))
|
||||
layer_memory = 0
|
||||
if module.weight is not None:
|
||||
layer_memory += module.weight.numel() * module.weight.element_size()
|
||||
if hasattr(module, "bias") and module.bias is not None:
|
||||
layer_memory += module.bias.numel() * module.bias.element_size()
|
||||
memory_by_type[layer_type] += layer_memory
|
||||
total_memory += layer_memory
|
||||
|
||||
logger.info(" DisTorch Model Layer Distribution")
|
||||
logger.info(dash_line)
|
||||
fmt_layer = "{:<12}{:>10}{:>14}{:>10}"
|
||||
logger.info(fmt_layer.format("Layer Type", "Layers", "Memory (MB)", "% Total"))
|
||||
logger.info(dash_line)
|
||||
for layer_type, count in layer_summary.items():
|
||||
mem_mb = memory_by_type[layer_type] / (1024 * 1024)
|
||||
mem_percent = (memory_by_type[layer_type] / total_memory) * 100 if total_memory > 0 else 0
|
||||
logger.info(fmt_layer.format(layer_type,str(count),f"{mem_mb:.2f}",f"{mem_percent:.1f}%"))
|
||||
logger.info(dash_line)
|
||||
|
||||
nonzero_devices = [d for d, r in DEVICE_RATIOS_DISTORCH.items() if r > 0]
|
||||
nonzero_total_ratio = sum(DEVICE_RATIOS_DISTORCH[d] for d in nonzero_devices)
|
||||
device_assignments = {device: [] for device in DEVICE_RATIOS_DISTORCH.keys()}
|
||||
total_layers = len(layer_list)
|
||||
current_layer = 0
|
||||
|
||||
for idx, device in enumerate(nonzero_devices):
|
||||
ratio = DEVICE_RATIOS_DISTORCH[device]
|
||||
if idx == len(nonzero_devices) - 1:
|
||||
device_layer_count = total_layers - current_layer
|
||||
else:
|
||||
device_layer_count = int((ratio / nonzero_total_ratio) * total_layers)
|
||||
start_idx = current_layer
|
||||
end_idx = current_layer + device_layer_count
|
||||
device_assignments[device] = layer_list[start_idx:end_idx]
|
||||
current_layer += device_layer_count
|
||||
|
||||
logger.info("DisTorch Model Final Device/Layer Assignments")
|
||||
logger.info(dash_line)
|
||||
fmt_assign = "{:<12}{:>10}{:>14}{:>10}"
|
||||
logger.info(fmt_assign.format("Device", "Layers", "Memory (MB)", "% Total"))
|
||||
logger.info(dash_line)
|
||||
total_assigned_memory = 0
|
||||
device_memories = {}
|
||||
for device, layers in device_assignments.items():
|
||||
device_memory = 0
|
||||
for layer_type in layer_summary:
|
||||
type_layers = sum(1 for _, _, lt in layers if lt == layer_type)
|
||||
if layer_summary[layer_type] > 0:
|
||||
mem_per_layer = memory_by_type[layer_type] / layer_summary[layer_type]
|
||||
device_memory += mem_per_layer * type_layers
|
||||
device_memories[device] = device_memory
|
||||
total_assigned_memory += device_memory
|
||||
|
||||
sorted_assignments = sorted(device_assignments.keys(), key=lambda d: (d == "cpu", d))
|
||||
|
||||
for dev in sorted_assignments:
|
||||
layers = device_assignments[dev]
|
||||
mem_mb = device_memories[dev] / (1024 * 1024)
|
||||
mem_percent = (device_memories[dev] / total_memory) * 100 if total_memory > 0 else 0
|
||||
logger.info(fmt_assign.format(dev,str(len(layers)),f"{mem_mb:.2f}",f"{mem_percent:.1f}%"))
|
||||
logger.info(dash_line)
|
||||
|
||||
return {"device_assignments": device_assignments}
|
||||
|
||||
|
||||
def calculate_vvram_allocation_string(model, virtual_vram_str):
|
||||
"""Calculate virtual VRAM allocation string for distributed loading"""
|
||||
recipient_device, vram_amount, donors = virtual_vram_str.split(';')
|
||||
virtual_vram_gb = float(vram_amount)
|
||||
|
||||
eq_line = "=" * 47
|
||||
dash_line = "-" * 47
|
||||
fmt_assign = "{:<8} {:<6} {:>11} {:>9} {:>9}"
|
||||
|
||||
logger.info(eq_line)
|
||||
logger.info(" DisTorch Model Virtual VRAM Analysis")
|
||||
logger.info(eq_line)
|
||||
logger.info(fmt_assign.format("Object", "Role", "Original(GB)", "Total(GB)", "Virt(GB)"))
|
||||
logger.info(dash_line)
|
||||
|
||||
recipient_vram = mm.get_total_memory(torch.device(recipient_device)) / (1024**3)
|
||||
recipient_virtual = recipient_vram + virtual_vram_gb
|
||||
|
||||
logger.info(fmt_assign.format(recipient_device, 'recip', f"{recipient_vram:.2f}GB",f"{recipient_virtual:.2f}GB", f"+{virtual_vram_gb:.2f}GB"))
|
||||
|
||||
ram_donors = [d for d in donors.split(',') if d != 'cpu']
|
||||
remaining_vram_needed = virtual_vram_gb
|
||||
|
||||
donor_device_info = {}
|
||||
donor_allocations = {}
|
||||
|
||||
for donor in ram_donors:
|
||||
donor_vram = mm.get_total_memory(torch.device(donor)) / (1024**3)
|
||||
max_donor_capacity = donor_vram * 0.9
|
||||
|
||||
donation = min(remaining_vram_needed, max_donor_capacity)
|
||||
donor_virtual = donor_vram - donation
|
||||
remaining_vram_needed -= donation
|
||||
donor_allocations[donor] = donation
|
||||
|
||||
donor_device_info[donor] = (donor_vram, donor_virtual)
|
||||
logger.info(fmt_assign.format(donor, 'donor', f"{donor_vram:.2f}GB", f"{donor_virtual:.2f}GB", f"-{donation:.2f}GB"))
|
||||
|
||||
system_dram_gb = mm.get_total_memory(torch.device('cpu')) / (1024**3)
|
||||
cpu_donation = remaining_vram_needed
|
||||
cpu_virtual = system_dram_gb - cpu_donation
|
||||
donor_allocations['cpu'] = cpu_donation
|
||||
logger.info(fmt_assign.format('cpu', 'donor', f"{system_dram_gb:.2f}GB", f"{cpu_virtual:.2f}GB", f"-{cpu_donation:.2f}GB"))
|
||||
|
||||
logger.info(dash_line)
|
||||
|
||||
layer_summary = {}
|
||||
layer_list = []
|
||||
memory_by_type = defaultdict(int)
|
||||
total_memory = 0
|
||||
|
||||
for name, module in model.named_modules():
|
||||
if hasattr(module, "weight"):
|
||||
layer_type = type(module).__name__
|
||||
layer_summary[layer_type] = layer_summary.get(layer_type, 0) + 1
|
||||
layer_list.append((name, module, layer_type))
|
||||
layer_memory = 0
|
||||
if module.weight is not None:
|
||||
layer_memory += module.weight.numel() * module.weight.element_size()
|
||||
if hasattr(module, "bias") and module.bias is not None:
|
||||
layer_memory += module.bias.numel() * module.bias.element_size()
|
||||
memory_by_type[layer_type] += layer_memory
|
||||
total_memory += layer_memory
|
||||
|
||||
model_size_gb = total_memory / (1024**3)
|
||||
new_model_size_gb = max(0, model_size_gb - virtual_vram_gb)
|
||||
|
||||
logger.info(fmt_assign.format('model', 'model', f"{model_size_gb:.2f}GB",f"{new_model_size_gb:.2f}GB", f"-{virtual_vram_gb:.2f}GB"))
|
||||
|
||||
if model_size_gb > (recipient_vram * 0.9):
|
||||
on_recipient = recipient_vram * 0.9
|
||||
on_virtuals = model_size_gb - on_recipient
|
||||
logger.info(f"\nWarning: Model size is greater than 90% of recipient VRAM. {on_virtuals:.2f} GB of GGML Layers Offloaded Automatically to Virtual VRAM.\n")
|
||||
else:
|
||||
on_recipient = model_size_gb
|
||||
on_virtuals = 0
|
||||
|
||||
new_on_recipient = max(0, on_recipient - virtual_vram_gb)
|
||||
|
||||
allocation_parts = []
|
||||
recipient_percent = new_on_recipient / recipient_vram
|
||||
allocation_parts.append(f"{recipient_device},{recipient_percent:.4f}")
|
||||
|
||||
for donor in ram_donors:
|
||||
donor_vram = donor_device_info[donor][0]
|
||||
donor_percent = donor_allocations[donor] / donor_vram
|
||||
allocation_parts.append(f"{donor},{donor_percent:.4f}")
|
||||
|
||||
cpu_percent = donor_allocations['cpu'] / system_dram_gb
|
||||
allocation_parts.append(f"cpu,{cpu_percent:.4f}")
|
||||
|
||||
allocation_string = ";".join(allocation_parts)
|
||||
fmt_mem = "{:<20}{:>20}"
|
||||
logger.info(fmt_mem.format("\n v1 Expert String", allocation_string))
|
||||
|
||||
return allocation_string
|
||||
|
||||
|
||||
def override_class_with_distorch_gguf(cls):
|
||||
"""Legacy DisTorch wrapper for GGUF models for backward compatibility."""
|
||||
from . import current_device
|
||||
|
||||
class NodeOverrideDisTorchGGUFLegacy(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
|
||||
"multiline": False,
|
||||
"default": "",
|
||||
})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/legacy"
|
||||
FUNCTION = "override"
|
||||
if hasattr(cls, 'TITLE'):
|
||||
TITLE = f"{cls.TITLE} (Legacy)"
|
||||
else:
|
||||
TITLE = "Legacy DisTorch Node"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
|
||||
from . import set_current_device
|
||||
if device is not None:
|
||||
set_current_device(device)
|
||||
|
||||
register_patched_ggufmodelpatcher()
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_model_hash(out[0], "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_model_hash(out[0].patcher, "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchGGUFLegacy
|
||||
|
||||
|
||||
def override_class_with_distorch_gguf_v2(cls):
|
||||
"""DisTorch 2.0 wrapper for GGUF models."""
|
||||
from . import current_device
|
||||
|
||||
class NodeOverrideDisTorchGGUFv2(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
compute_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, compute_device=None, virtual_vram_gb=4.0,
|
||||
donor_device="cpu", expert_mode_allocations="", **kwargs):
|
||||
from . import set_current_device
|
||||
if compute_device is not None:
|
||||
set_current_device(compute_device)
|
||||
|
||||
register_patched_ggufmodelpatcher()
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_model_hash(out[0], "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_model_hash(out[0].patcher, "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchGGUFv2
|
||||
|
||||
|
||||
def override_class_with_distorch_clip(cls):
|
||||
"""DisTorch wrapper for CLIP models with GGUF support"""
|
||||
from . import current_text_encoder_device
|
||||
|
||||
class NodeOverrideDisTorch(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
|
||||
"multiline": False,
|
||||
"default": "",
|
||||
"tooltip": "Expert use only: Manual VRAM allocation string. Incorrect values can cause crashes. Do not modify unless you fully understand DisTorch memory management."
|
||||
})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
|
||||
from . import set_current_text_encoder_device
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
register_patched_ggufmodelpatcher()
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logging.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_model_hash(out[0], "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_model_hash(out[0].patcher, "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorch
|
||||
def override_class_with_distorch_clip_no_device(cls):
|
||||
"""DisTorch wrapper for CLIP models with GGUF support"""
|
||||
from . import current_text_encoder_device
|
||||
|
||||
class NodeOverrideDisTorchClipNoDevice(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {
|
||||
"multiline": False,
|
||||
"default": "",
|
||||
"tooltip": "Expert use only: Manual VRAM allocation string. Incorrect values can cause crashes. Do not modify unless you fully understand DisTorch memory management."
|
||||
})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations=None, use_other_vram=None, virtual_vram_gb=0.0, **kwargs):
|
||||
from . import set_current_text_encoder_device
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
register_patched_ggufmodelpatcher()
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logging.info(f"[MultiGPU_DisTorch] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_model_hash(out[0], "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_model_hash(out[0].patcher, "override")
|
||||
model_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchClipNoDevice
|
||||
|
||||
# Alias for backward compatibility
|
||||
override_class_with_distorch = override_class_with_distorch_gguf
|
||||
+127
-546
@@ -16,8 +16,9 @@ import inspect
|
||||
from collections import defaultdict
|
||||
import comfy.model_management as mm
|
||||
import comfy.model_patcher
|
||||
from . import current_device
|
||||
from .device_utils import get_device_list, soft_empty_cache_multigpu
|
||||
from .model_management_mgpu import multigpu_memory_log, force_full_system_cleanup
|
||||
|
||||
|
||||
safetensor_allocation_store = {}
|
||||
safetensor_settings_store = {}
|
||||
@@ -48,15 +49,61 @@ def create_safetensor_model_hash(model, caller):
|
||||
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
|
||||
|
||||
# DEBUG STATEMENT - ALWAYS LOG THE HASH
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Created hash for {caller}: {final_hash[:8]}...")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Created hash for {caller}: {final_hash[:8]}...")
|
||||
return final_hash
|
||||
|
||||
|
||||
def register_patched_safetensor_modelpatcher():
|
||||
"""Register and patch the ModelPatcher for distributed safetensor loading"""
|
||||
from comfy.model_patcher import wipe_lowvram_weight, move_weight_functions
|
||||
# Patch ComfyUI's ModelPatcher
|
||||
if not hasattr(comfy.model_patcher.ModelPatcher, '_distorch_patched'):
|
||||
|
||||
# Patch LoadedModel.model_memory_required to drive behavior purely by Phase 2 = unload_distorch_model flag
|
||||
from comfy.model_management import current_loaded_models
|
||||
|
||||
original_loaded_model_memory_required = None
|
||||
for cls in current_loaded_models.__class__.__mro__:
|
||||
if hasattr(cls, 'model_memory_required'):
|
||||
original_loaded_model_memory_required = cls.model_memory_required
|
||||
break
|
||||
|
||||
if original_loaded_model_memory_required is None:
|
||||
# Global patch of LoadedModel class if available
|
||||
import comfy.model_management as mm
|
||||
|
||||
original_loaded_model_memory_required = mm.LoadedModel.model_memory_required
|
||||
|
||||
def patched_loaded_model_memory_required(self, device):
|
||||
"""Drive unload behavior purely by unload_distorch_model flag"""
|
||||
multigpu_memory_log("unload_distorch_model_memory_check", "start")
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Memory assessment requested for model on device: {device}")
|
||||
|
||||
# Check if this is a DisTorch model with unload_distorch_model flag
|
||||
is_distorch_model = hasattr(getattr(getattr(self, 'model', None), 'model', None), '_mgpu_unload_distorch_model')
|
||||
|
||||
model_name = type(getattr(getattr(self, 'model', None), 'model', None)).__name__ if getattr(getattr(self, 'model', None), 'model', None) else "Unknown"
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] DisTorch model: {model_name}, is_distorch_model={is_distorch_model}")
|
||||
|
||||
if is_distorch_model:
|
||||
if self.model.model._mgpu_unload_distorch_model:
|
||||
total_device_memory = mm.get_total_memory(device)
|
||||
memory_gb = total_device_memory / (1024**3)
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=True - Reporting MAX memory ({memory_gb:.2f}GB) to force complete eviction")
|
||||
return total_device_memory
|
||||
else:
|
||||
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] _mgpu_unload_distorch_model=False - Reporting 0 bytes (prevents eviction)")
|
||||
return 0
|
||||
|
||||
# Not a DisTorch model - use original behavior
|
||||
logger.mgpu_mm_log("[IS_DISTORCH_MODEL] Non-DisTorch model - Using original Comfy memory calculation")
|
||||
original_result = original_loaded_model_memory_required(self, device)
|
||||
original_gb = original_result / (1024**3) if original_result else 0
|
||||
logger.mgpu_mm_log(f"[IS_DISTORCH_MODEL] Original calculation returned: {original_gb:.2f}GB")
|
||||
multigpu_memory_log("keep_loaded_memory_check", "end")
|
||||
return original_result
|
||||
|
||||
mm.LoadedModel.model_memory_required = patched_loaded_model_memory_required
|
||||
|
||||
original_partially_load = comfy.model_patcher.ModelPatcher.partially_load
|
||||
|
||||
def new_partially_load(self, device_to, extra_memory=0, full_load=False, force_patch_weights=False, **kwargs):
|
||||
@@ -64,11 +111,16 @@ def register_patched_safetensor_modelpatcher():
|
||||
global safetensor_allocation_store
|
||||
|
||||
debug_hash = create_safetensor_model_hash(self, "partial_load")
|
||||
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "pre-load")
|
||||
allocations = safetensor_allocation_store.get(debug_hash)
|
||||
|
||||
if not hasattr(self.model, '_distorch_high_precision_loras') or not allocations:
|
||||
# Set default precision flag before checking
|
||||
if not hasattr(self.model, '_distorch_high_precision_loras'):
|
||||
self.model._distorch_high_precision_loras = True
|
||||
|
||||
if not allocations:
|
||||
result = original_partially_load(self, device_to, extra_memory, force_patch_weights)
|
||||
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "post-load")
|
||||
if hasattr(self, '_distorch_block_assignments'):
|
||||
del self._distorch_block_assignments
|
||||
return result
|
||||
@@ -79,7 +131,7 @@ def register_patched_safetensor_modelpatcher():
|
||||
unpatch_weights = self.model.current_weight_patches_uuid is not None and (self.model.current_weight_patches_uuid != self.patches_uuid or force_patch_weights)
|
||||
|
||||
if unpatch_weights:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Patches changed or forced. Unpatching model.")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Patches changed or forced. Unpatching model.")
|
||||
self.unpatch_model(self.offload_device, unpatch_weights=True)
|
||||
|
||||
self.patch_model(load_weights=False)
|
||||
@@ -87,15 +139,10 @@ def register_patched_safetensor_modelpatcher():
|
||||
mem_counter = 0
|
||||
|
||||
is_clip_model = getattr(self, 'is_clip', False)
|
||||
if is_clip_model:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Using CLIP-specific allocation for model {debug_hash[:8]} (HEAD PRESERVATION ENABLED)")
|
||||
device_assignments = analyze_safetensor_loading_clip(self, allocations)
|
||||
else:
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Using standard allocation for model {debug_hash[:8]} (UNET/VAE - UNTOUCHED)")
|
||||
device_assignments = analyze_safetensor_loading(self, allocations)
|
||||
device_assignments = analyze_safetensor_loading(self, allocations, is_clip=is_clip_model)
|
||||
|
||||
model_original_dtype = comfy.utils.weight_dtype(self.model.state_dict())
|
||||
high_precision_loras = self.model._distorch_high_precision_loras
|
||||
high_precision_loras = getattr(self.model, "_distorch_high_precision_loras", True)
|
||||
loading = self._load_list()
|
||||
loading.sort(reverse=True)
|
||||
for module_size, module_name, module_object, params in loading:
|
||||
@@ -109,7 +156,7 @@ def register_patched_safetensor_modelpatcher():
|
||||
pass
|
||||
|
||||
if current_module_device is not None and str(current_module_device) != str(block_target_device):
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Moving already patched {module_name} to {block_target_device}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Moving already patched {module_name} to {block_target_device}")
|
||||
module_object.to(block_target_device)
|
||||
|
||||
mem_counter += module_size
|
||||
@@ -143,11 +190,11 @@ def register_patched_safetensor_modelpatcher():
|
||||
new_param = torch.nn.Parameter(cast_data.to(torch.float8_e4m3fn))
|
||||
new_param.requires_grad = param.requires_grad
|
||||
setattr(module_object, param_name, new_param)
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Cast {module_name}.{param_name} to FP8 for CPU storage")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Cast {module_name}.{param_name} to FP8 for CPU storage")
|
||||
|
||||
# Step 4: Move to ultimate destination based on DisTorch assignment
|
||||
if block_target_device != device_to:
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Moving {module_name} from {device_to} to {block_target_device}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Moving {module_name} from {device_to} to {block_target_device}")
|
||||
module_object.to(block_target_device)
|
||||
module_object.comfy_cast_weights = True
|
||||
|
||||
@@ -157,20 +204,39 @@ def register_patched_safetensor_modelpatcher():
|
||||
|
||||
self.model.current_weight_patches_uuid = self.patches_uuid
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] DisTorch loading completed. Total memory: {mem_counter / (1024 * 1024):.2f}MB")
|
||||
logger.info("[MultiGPU DisTorch V2] DisTorch loading completed.")
|
||||
logger.info(f"[MultiGPU DisTorch V2] Total memory: {mem_counter / (1024 * 1024):.2f}MB")
|
||||
multigpu_memory_log(f"safetensor:{debug_hash[:8]}", "post-load")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
comfy.model_patcher.ModelPatcher.partially_load = new_partially_load
|
||||
comfy.model_patcher.ModelPatcher._distorch_patched = True
|
||||
logger.info("[MultiGPU_DisTorch2] Successfully patched ModelPatcher.partially_load")
|
||||
logger.info("[MultiGPU Core Patching] Successfully patched ModelPatcher.partially_load")
|
||||
|
||||
def _extract_clip_head_blocks(raw_block_list, compute_device):
|
||||
"""Identify and pre-assign CLIP head blocks to compute device returning head_blocks, distributable_blocks, block_assignments, and head_memory."""
|
||||
head_keywords = ['embed', 'wte', 'wpe', 'token_embedding', 'position_embedding']
|
||||
head_blocks = []
|
||||
distributable_blocks = []
|
||||
head_memory = 0
|
||||
block_assignments = {}
|
||||
|
||||
for module_size, module_name, module_object, params in raw_block_list:
|
||||
if any(kw in module_name.lower() for kw in head_keywords):
|
||||
head_blocks.append((module_size, module_name, module_object, params))
|
||||
block_assignments[module_name] = compute_device
|
||||
head_memory += module_size
|
||||
else:
|
||||
distributable_blocks.append((module_size, module_name, module_object, params))
|
||||
|
||||
return head_blocks, distributable_blocks, block_assignments, head_memory
|
||||
|
||||
def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
def analyze_safetensor_loading(model_patcher, allocations_string, is_clip=False):
|
||||
"""
|
||||
Analyze and distribute safetensor model blocks across devices
|
||||
Target for refactor back into one function once stability for CLIP is established.
|
||||
Analyze and distribute safetensor model blocks across devices.
|
||||
Supports CLIP head preservation when is_clip=True.
|
||||
"""
|
||||
DEVICE_RATIOS_DISTORCH = {}
|
||||
device_table = {}
|
||||
@@ -180,13 +246,10 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
distorch_alloc, virtual_vram_str = allocations_string.split('#')
|
||||
|
||||
compute_device = virtual_vram_str.split(';')[0]
|
||||
logger.info(f"[MultiGPU_DisTorch2] Compute Device: {compute_device}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Compute Device: {compute_device}")
|
||||
|
||||
if not distorch_alloc:
|
||||
mode = "fraction"
|
||||
logger.info("[MultiGPU_DisTorch2] Expert String Examples:")
|
||||
logger.info(" Direct(byte) Mode - cuda:0,500mb;cuda:1,3.0g;cpu,5gb* -> '*' cpu = over/underflow device, put 0.50gb on cuda0, 3.00gb on cuda1, and 5.00gb (or the rest) on cpu")
|
||||
logger.info(" Ratio(%) Mode - cuda:0,8%;cuda:1,8%;cpu,4% -> 8:8:4 ratio, put 40% on cuda0, 40% on cuda1, and 20% on cpu")
|
||||
distorch_alloc = calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str)
|
||||
|
||||
elif any(c in distorch_alloc.lower() for c in ['g', 'm', 'k', 'b']):
|
||||
@@ -202,12 +265,13 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
if device not in present_devices:
|
||||
distorch_alloc += f";{device},0.0"
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] Final Allocation String: {distorch_alloc}")
|
||||
|
||||
eq_line = "=" * 50
|
||||
dash_line = "-" * 50
|
||||
fmt_assign = "{:<18}{:>7}{:>14}{:>10}"
|
||||
|
||||
logger.info(eq_line)
|
||||
logger.info(f"[MultiGPU DisTorch V2] Final Allocation String:\n{distorch_alloc}")
|
||||
|
||||
for allocation in distorch_alloc.split(';'):
|
||||
if ',' not in allocation:
|
||||
continue
|
||||
@@ -257,13 +321,23 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
total_memory = 0
|
||||
|
||||
raw_block_list = model_patcher._load_list()
|
||||
|
||||
total_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
|
||||
|
||||
MIN_BLOCK_THRESHOLD = total_memory * 0.0001
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Total model memory: {total_memory} bytes")
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Tiny block threshold (0.01%): {MIN_BLOCK_THRESHOLD} bytes")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Total model memory: {total_memory} bytes")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Tiny block threshold (0.01%): {MIN_BLOCK_THRESHOLD} bytes")
|
||||
|
||||
# CLIP-specific: Extract head blocks and get pre-assignments
|
||||
head_memory = 0
|
||||
block_assignments = {}
|
||||
if is_clip:
|
||||
head_blocks, distributable_raw, block_assignments, head_memory = \
|
||||
_extract_clip_head_blocks(raw_block_list, compute_device)
|
||||
logger.info(f"[MultiGPU DisTorch V2 CLIP] Preserving {len(head_blocks)} head layer(s) ({head_memory/(1024**2):.2f} MB) on compute device: {compute_device}")
|
||||
else:
|
||||
distributable_raw = raw_block_list
|
||||
|
||||
# Build all_blocks list for summary (using full raw_block_list)
|
||||
all_blocks = []
|
||||
for module_size, module_name, module_object, params in raw_block_list:
|
||||
block_type = type(module_object).__name__
|
||||
@@ -272,12 +346,17 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
memory_by_type[block_type] += module_size
|
||||
all_blocks.append((module_name, module_object, block_type, module_size))
|
||||
|
||||
block_list = [b for b in all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
|
||||
tiny_block_list = [b for b in all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
|
||||
# Use distributable blocks for actual allocation (for CLIP, this excludes heads)
|
||||
distributable_all_blocks = []
|
||||
for module_size, module_name, module_object, params in distributable_raw:
|
||||
distributable_all_blocks.append((module_name, module_object, type(module_object).__name__, module_size))
|
||||
|
||||
block_list = [b for b in distributable_all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
|
||||
tiny_block_list = [b for b in distributable_all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
|
||||
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Total blocks: {len(all_blocks)}")
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Distributable blocks: {len(block_list)}")
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Tiny blocks (<0.01%): {len(tiny_block_list)}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Total blocks: {len(all_blocks)}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Distributable blocks: {len(block_list)}")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Tiny blocks (<0.01%): {len(tiny_block_list)}")
|
||||
|
||||
logger.info(" DisTorch2 Model Layer Distribution")
|
||||
logger.info(dash_line)
|
||||
@@ -304,6 +383,11 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
for dev in donor_devices
|
||||
}
|
||||
|
||||
# CLIP-specific: Adjust compute_device quota to account for locked head blocks
|
||||
if is_clip and compute_device in donor_quotas and head_memory > 0:
|
||||
donor_quotas[compute_device] = max(0, donor_quotas[compute_device] - head_memory)
|
||||
logger.debug(f"[MultiGPU DisTorch V2 CLIP] Adjusted {compute_device} quota by -{head_memory/(1024**2):.2f} MB for head preservation")
|
||||
|
||||
# Iterate from the TAIL of the model, assigning blocks to donors until their quotas are filled.
|
||||
for block_name, module, block_type, block_memory in reversed(block_list):
|
||||
assigned_to_donor = False
|
||||
@@ -340,7 +424,7 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
tiny_mem_percent = (tiny_block_memory / total_memory) * 100 if total_memory > 0 else 0
|
||||
device_label = f"{compute_device} (<0.01%)"
|
||||
logger.info(fmt_assign.format(device_label, str(len(tiny_block_list)), f"{tiny_mem_mb:.2f}", f"{tiny_mem_percent:.1f}%"))
|
||||
logger.debug(f"[MultiGPU_DisTorch2] Tiny block memory breakdown: {tiny_block_memory} bytes ({tiny_mem_mb:.2f} MB), which is {tiny_mem_percent:.4f}% of total model memory.")
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Tiny block memory breakdown: {tiny_block_memory} bytes ({tiny_mem_mb:.2f} MB), which is {tiny_mem_percent:.4f}% of total model memory.")
|
||||
|
||||
total_assigned_memory = 0
|
||||
device_memories = {}
|
||||
@@ -373,211 +457,6 @@ def analyze_safetensor_loading(model_patcher, allocations_string):
|
||||
"block_assignments": block_assignments
|
||||
}
|
||||
|
||||
|
||||
def analyze_safetensor_loading_clip(model_patcher, allocations_string):
|
||||
"""
|
||||
CLIP-SPECIFIC: A 1:1 clone of the working UNET allocation logic with the
|
||||
single required modification to preserve head-blocks on the compute device.
|
||||
All other logic and UX (logging, etc.) is identical to the original.
|
||||
Target for refactor once stability for CLIP is established.
|
||||
"""
|
||||
DEVICE_RATIOS_DISTORCH = {}
|
||||
device_table = {}
|
||||
distorch_alloc = allocations_string
|
||||
virtual_vram_gb = 0.0
|
||||
|
||||
distorch_alloc, virtual_vram_str = allocations_string.split('#')
|
||||
|
||||
compute_device = virtual_vram_str.split(';')[0]
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2_CLIP] CLIP Compute Device: {compute_device}")
|
||||
|
||||
if not distorch_alloc:
|
||||
mode = "fraction"
|
||||
logger.info("[MultiGPU_DisTorch2_CLIP] Expert String Examples:")
|
||||
logger.info(" Direct(byte) Mode - cuda:0,500mb;cuda:1,3.0g;cpu,5gb* -> '*' cpu = over/underflow device, put 0.50gb on cuda0, 3.00gb on cuda1, and 5.00gb (or the rest) on cpu")
|
||||
logger.info(" Ratio(%) Mode - cuda:0,8%;cuda:1,8%;cpu,4% -> 8:8:4 ratio, put 40% on cuda0, 40% on cuda1, and 20% on cpu")
|
||||
distorch_alloc = calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str)
|
||||
|
||||
elif any(c in distorch_alloc.lower() for c in ['g', 'm', 'k', 'b']):
|
||||
mode = "byte"
|
||||
distorch_alloc = calculate_fraction_from_byte_expert_string(model_patcher, distorch_alloc)
|
||||
elif "%" in distorch_alloc:
|
||||
mode = "ratio"
|
||||
distorch_alloc = calculate_fraction_from_ratio_expert_string(model_patcher, distorch_alloc)
|
||||
|
||||
all_devices = get_device_list()
|
||||
present_devices = {item.split(',')[0] for item in distorch_alloc.split(';') if ',' in item}
|
||||
for device in all_devices:
|
||||
if device not in present_devices:
|
||||
distorch_alloc += f";{device},0.0"
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2_CLIP] Final CLIP Allocation String: {distorch_alloc}")
|
||||
|
||||
eq_line = "=" * 50
|
||||
dash_line = "-" * 50
|
||||
fmt_assign = "{:<18}{:>7}{:>14}{:>10}"
|
||||
|
||||
for allocation in distorch_alloc.split(';'):
|
||||
if ',' not in allocation:
|
||||
continue
|
||||
dev_name, fraction = allocation.split(',')
|
||||
fraction = float(fraction)
|
||||
total_mem_bytes = mm.get_total_memory(torch.device(dev_name))
|
||||
alloc_gb = (total_mem_bytes * fraction) / (1024**3)
|
||||
DEVICE_RATIOS_DISTORCH[dev_name] = alloc_gb
|
||||
device_table[dev_name] = {
|
||||
"fraction": fraction,
|
||||
"total_gb": total_mem_bytes / (1024**3),
|
||||
"alloc_gb": alloc_gb
|
||||
}
|
||||
|
||||
logger.info(eq_line)
|
||||
logger.info(" DisTorch2 CLIP Model Device Allocations")
|
||||
logger.info(eq_line)
|
||||
|
||||
fmt_rosetta = "{:<8}{:>9}{:>9}{:>11}{:>10}"
|
||||
logger.info(fmt_rosetta.format("Device", "VRAM GB", "Dev %", "Model GB", "Dist %"))
|
||||
logger.info(dash_line)
|
||||
|
||||
sorted_devices = sorted(device_table.keys(), key=lambda d: (d == "cpu", d))
|
||||
|
||||
total_allocated_model_bytes = sum(d["alloc_gb"] * (1024**3) for d in device_table.values())
|
||||
|
||||
for dev in sorted_devices:
|
||||
total_dev_gb = device_table[dev]["total_gb"]
|
||||
alloc_fraction = device_table[dev]["fraction"]
|
||||
alloc_gb = device_table[dev]["alloc_gb"]
|
||||
|
||||
dist_ratio_percent = (alloc_gb * (1024**3) / total_allocated_model_bytes) * 100 if total_allocated_model_bytes > 0 else 0
|
||||
|
||||
logger.info(fmt_rosetta.format(
|
||||
dev,
|
||||
f"{total_dev_gb:.2f}",
|
||||
f"{alloc_fraction*100:.1f}%",
|
||||
f"{alloc_gb:.2f}",
|
||||
f"{dist_ratio_percent:.1f}%"
|
||||
))
|
||||
|
||||
logger.info(dash_line)
|
||||
|
||||
block_summary = {}
|
||||
memory_by_type = defaultdict(int)
|
||||
|
||||
raw_block_list = model_patcher._load_list()
|
||||
total_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
|
||||
|
||||
# Split the model into head and distributable parts
|
||||
head_keywords = ['embed', 'wte', 'wpe', 'token_embedding', 'position_embedding']
|
||||
head_blocks = []
|
||||
distributable_blocks_raw = []
|
||||
head_memory = 0
|
||||
|
||||
for module_size, module_name, module_object, params in raw_block_list:
|
||||
if any(keyword in module_name.lower() for keyword in head_keywords):
|
||||
head_blocks.append((module_size, module_name, module_object, params))
|
||||
else:
|
||||
distributable_blocks_raw.append((module_size, module_name, module_object, params))
|
||||
|
||||
MIN_BLOCK_THRESHOLD = total_memory * 0.0001
|
||||
all_blocks = []
|
||||
|
||||
for module_size, module_name, module_object, params in raw_block_list:
|
||||
block_type = type(module_object).__name__
|
||||
block_summary[block_type] = block_summary.get(block_type, 0) + 1
|
||||
memory_by_type[block_type] += module_size
|
||||
all_blocks.append((module_name, module_object, block_type, module_size))
|
||||
|
||||
# Use the distributable part for actual allocation logic
|
||||
distributable_all_blocks = []
|
||||
for module_size, module_name, module_object, params in distributable_blocks_raw:
|
||||
distributable_all_blocks.append((module_name, module_object, type(module_object).__name__, module_size))
|
||||
|
||||
block_list = [b for b in distributable_all_blocks if b[3] >= MIN_BLOCK_THRESHOLD]
|
||||
tiny_block_list = [b for b in distributable_all_blocks if b[3] < MIN_BLOCK_THRESHOLD]
|
||||
|
||||
logger.info(" DisTorch2 CLIP Model Layer Distribution")
|
||||
logger.info(dash_line)
|
||||
fmt_layer = "{:<18}{:>7}{:>14}{:>10}"
|
||||
logger.info(fmt_layer.format("Layer Type", "Layers", "Memory (MB)", "% Total"))
|
||||
logger.info(dash_line)
|
||||
|
||||
for layer_type, count in block_summary.items():
|
||||
mem_mb = memory_by_type[layer_type] / (1024 * 1024)
|
||||
mem_percent = (memory_by_type[layer_type] / total_memory) * 100 if total_memory > 0 else 0
|
||||
logger.info(fmt_layer.format(layer_type[:18], str(count), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
|
||||
|
||||
logger.info(dash_line)
|
||||
|
||||
block_assignments = {}
|
||||
|
||||
# Pre-assign head blocks and calculate their memory usage
|
||||
for module_size, module_name, module_object, params in head_blocks:
|
||||
block_assignments[module_name] = compute_device
|
||||
head_memory += module_size
|
||||
if head_blocks:
|
||||
logger.info(f"[MultiGPU_DisTorch2_CLIP] Preserving {len(head_blocks)} head layer(s) ({head_memory / (1024*1024):.2f} MB) on compute device: {compute_device}")
|
||||
donor_devices = [d for d in sorted_devices]
|
||||
donor_quotas = {
|
||||
dev: device_table[dev]["alloc_gb"] * (1024**3)
|
||||
for dev in donor_devices
|
||||
}
|
||||
# Adjust compute_device quota to account for the locked head
|
||||
if compute_device in donor_quotas:
|
||||
donor_quotas[compute_device] = max(0, donor_quotas[compute_device] - head_memory)
|
||||
|
||||
for block_name, module, block_type, block_memory in reversed(block_list):
|
||||
assigned_to_donor = False
|
||||
for donor in donor_devices:
|
||||
if donor_quotas[donor] >= block_memory:
|
||||
block_assignments[block_name] = donor
|
||||
donor_quotas[donor] -= block_memory
|
||||
assigned_to_donor = True
|
||||
break # Move to the next block
|
||||
|
||||
if not assigned_to_donor:
|
||||
block_assignments[block_name] = compute_device
|
||||
|
||||
for block_name, module, block_type, block_memory in tiny_block_list:
|
||||
block_assignments[block_name] = compute_device
|
||||
|
||||
device_assignments = {device: [] for device in DEVICE_RATIOS_DISTORCH.keys()}
|
||||
for block_name, device in block_assignments.items():
|
||||
# Find the block in the original list to get all its info
|
||||
for b_name, b_module, b_type, b_mem in all_blocks:
|
||||
if b_name == block_name:
|
||||
device_assignments[device].append((b_name, b_module, b_type, b_mem))
|
||||
break
|
||||
|
||||
logger.info("DisTorch2 CLIP Model Final Device/Layer Assignments")
|
||||
logger.info(dash_line)
|
||||
logger.info(fmt_assign.format("Device", "Layers", "Memory (MB)", "% Total"))
|
||||
logger.info(dash_line)
|
||||
|
||||
device_memories = defaultdict(int)
|
||||
device_counts = defaultdict(int)
|
||||
for device, blocks in device_assignments.items():
|
||||
for b_name, b_module, b_type, b_mem in blocks:
|
||||
device_memories[device] += b_mem
|
||||
device_counts[device] += 1
|
||||
|
||||
sorted_assignments = sorted(device_memories.keys(), key=lambda d: (d == "cpu", d))
|
||||
|
||||
for dev in sorted_assignments:
|
||||
if device_counts[dev] == 0:
|
||||
continue
|
||||
mem_mb = device_memories[dev] / (1024 * 1024)
|
||||
mem_percent = (device_memories[dev] / total_memory) * 100 if total_memory > 0 else 0
|
||||
logger.info(fmt_assign.format(dev, str(device_counts[dev]), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
|
||||
|
||||
logger.info(dash_line)
|
||||
|
||||
return {
|
||||
"device_assignments": device_assignments,
|
||||
"block_assignments": block_assignments
|
||||
}
|
||||
|
||||
|
||||
def parse_memory_string(mem_str):
|
||||
"""Parses a memory string (e.g., '4.0g', '512M') and returns bytes."""
|
||||
mem_str = mem_str.strip().lower()
|
||||
@@ -598,11 +477,7 @@ def parse_memory_string(mem_str):
|
||||
return val
|
||||
|
||||
def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
|
||||
"""
|
||||
Converts a user-provided byte string (e.g., "cuda:1,4gb;cpu,*") into a
|
||||
fractional VRAM allocation string that the main assignment logic can use.
|
||||
This function strictly respects device order and byte quotas.
|
||||
"""
|
||||
"""Convert byte allocation string (e.g. 'cuda:1,4gb;cpu,*') to fractional VRAM allocation string respecting device order and byte quotas."""
|
||||
raw_block_list = model_patcher._load_list()
|
||||
total_model_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
|
||||
remaining_model_bytes = total_model_memory
|
||||
@@ -637,16 +512,16 @@ def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
|
||||
if bytes_to_assign > 0:
|
||||
final_byte_allocations[dev] = bytes_to_assign
|
||||
remaining_model_bytes -= bytes_to_assign
|
||||
logger.info(f"[MultiGPU_DisTorch2] Assigning {bytes_to_assign / (1024**2):.2f}MB of model to {dev} (requested {requested_bytes / (1024**2):.2f}MB).")
|
||||
logger.info(f"[MultiGPU DisTorch V2] Assigning {bytes_to_assign / (1024**2):.2f}MB of model to {dev} (requested {requested_bytes / (1024**2):.2f}MB).")
|
||||
|
||||
if remaining_model_bytes <= 0:
|
||||
logger.info("[MultiGPU_DisTorch2] All model blocks have been allocated. Subsequent devices in the string will receive no assignment.")
|
||||
logger.info("[MultiGPU DisTorch V2] All model blocks have been allocated. Subsequent devices in the string will receive no assignment.")
|
||||
break
|
||||
|
||||
# Assign any leftover model bytes to the wildcard device
|
||||
if remaining_model_bytes > 0:
|
||||
final_byte_allocations[wildcard_device] += remaining_model_bytes
|
||||
logger.info(f"[MultiGPU_DisTorch2] Assigning remaining {remaining_model_bytes / (1024**2):.2f}MB of model to wildcard device '{wildcard_device}'.")
|
||||
logger.info(f"[MultiGPU DisTorch V2] Assigning remaining {remaining_model_bytes / (1024**2):.2f}MB of model to wildcard device '{wildcard_device}'.")
|
||||
|
||||
# Convert the final byte allocations to VRAM fractions
|
||||
allocation_parts = []
|
||||
@@ -661,10 +536,7 @@ def calculate_fraction_from_byte_expert_string(model_patcher, byte_str):
|
||||
return allocations_string
|
||||
|
||||
def calculate_fraction_from_ratio_expert_string(model_patcher, ratio_str):
|
||||
"""
|
||||
Converts a user-provided ratio string (which describes how to split the MODEL)
|
||||
into a fraction string (which describes the fraction of DEVICE VRAM to use).
|
||||
"""
|
||||
"""Convert ratio allocation string (e.g. 'cuda:0,25%;cpu,75%') describing model split to fractional VRAM allocation string."""
|
||||
raw_block_list = model_patcher._load_list()
|
||||
total_model_memory = sum(module_size for module_size, _, _, _ in raw_block_list)
|
||||
|
||||
@@ -704,7 +576,7 @@ def calculate_fraction_from_ratio_expert_string(model_patcher, ratio_str):
|
||||
else:
|
||||
put_part = ", ".join(put_parts[:-1]) + f", and {put_parts[-1]}"
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] Ratio(%) Mode - {ratio_str} -> {ratio_string} ratio, put {put_part}")
|
||||
logger.info(f"[MultiGPU DisTorch V2] Ratio(%) Mode - {ratio_str} -> {ratio_string} ratio, put {put_part}")
|
||||
|
||||
allocations_string = ";".join(allocation_parts)
|
||||
|
||||
@@ -772,8 +644,8 @@ def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
|
||||
# Warning if model too large
|
||||
if model_size_gb > (recipient_vram * 0.9):
|
||||
required_offload_gb = model_size_gb - (recipient_vram * 0.9)
|
||||
logger.warning(f"[MultiGPU] WARNING: Model size ({model_size_gb:.2f}GB) is larger than 90% of available VRAM on {recipient_device} ({recipient_vram * 0.9:.2f}GB).")
|
||||
logger.warning(f"[MultiGPU] To prevent an OOM error, set 'virtual_vram_gb' to at least {required_offload_gb:.2f}.")
|
||||
logger.warning(f"\n\n[MultiGPU DisTorch V2] Model size ({model_size_gb:.2f}GB) is larger than 90% of available VRAM on: {recipient_device} ({recipient_vram * 0.9:.2f}GB).")
|
||||
logger.warning(f"[MultiGPU DisTorch V2] To prevent an OOM error, set 'virtual_vram_gb' to at least {required_offload_gb:.2f}.\n\n")
|
||||
|
||||
new_on_recipient = max(0, model_size_gb - virtual_vram_gb)
|
||||
|
||||
@@ -789,294 +661,3 @@ def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
|
||||
|
||||
allocations_string = ";".join(allocation_parts)
|
||||
return allocations_string
|
||||
|
||||
def override_class_with_distorch_safetensor_v2(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor models"""
|
||||
from . import current_device
|
||||
|
||||
class NodeOverrideDisTorchSafetensorV2(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
compute_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(s, *args, compute_device=None, virtual_vram_gb=4.0,
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
# Create a hash of our specific settings
|
||||
settings_str = f"{compute_device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}"
|
||||
return hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
def override(self, *args, compute_device=None, virtual_vram_gb=4.0,
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
|
||||
from . import set_current_device
|
||||
if compute_device is not None:
|
||||
set_current_device(compute_device)
|
||||
|
||||
# Register our patched ModelPatcher
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
# Call original function
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
|
||||
# --- Check if we need to unload the model due to settings change ---
|
||||
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
|
||||
settings_str = f"{compute_device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}"
|
||||
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
# Temporarily load to get hash without applying our patch
|
||||
temp_out = fn(*args, **kwargs)
|
||||
model_to_check = None
|
||||
if hasattr(temp_out[0], 'model'):
|
||||
model_to_check = temp_out[0]
|
||||
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
|
||||
model_to_check = temp_out[0].patcher
|
||||
|
||||
if model_to_check:
|
||||
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
|
||||
last_settings_hash = safetensor_settings_store.get(model_hash)
|
||||
|
||||
if last_settings_hash != settings_hash:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
|
||||
else:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
|
||||
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
# Store high_precision_loras in the model for later retrieval
|
||||
if hasattr(out[0], 'model'):
|
||||
out[0].model._distorch_high_precision_loras = high_precision_loras
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
|
||||
elif expert_mode_allocations: # Only include compute device if there's an expert string
|
||||
vram_string = compute_device
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchSafetensorV2
|
||||
|
||||
|
||||
def override_class_with_distorch_safetensor_v2_clip(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor CLIP models"""
|
||||
from . import current_device
|
||||
|
||||
class NodeOverrideDisTorchSafetensorV2Clip(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device}) # Changed from compute_device
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(s, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
# Create a hash of our specific settings
|
||||
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}" # Changed from compute_device
|
||||
return hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
def override(self, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
|
||||
from . import set_current_text_encoder_device # Use text encoder device setter
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
kwargs['device'] = 'default' # Hardcode device setting like in standard clip wrapper
|
||||
|
||||
# Register our patched ModelPatcher
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
# Call original function
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
|
||||
# --- Check if we need to unload the model due to settings change ---
|
||||
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
|
||||
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}" # Changed from compute_device
|
||||
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
# Temporarily load to get hash without applying our patch
|
||||
temp_out = fn(*args, **kwargs)
|
||||
model_to_check = None
|
||||
if hasattr(temp_out[0], 'model'):
|
||||
model_to_check = temp_out[0]
|
||||
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
|
||||
model_to_check = temp_out[0].patcher
|
||||
|
||||
if model_to_check:
|
||||
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
|
||||
last_settings_hash = safetensor_settings_store.get(model_hash)
|
||||
|
||||
if last_settings_hash != settings_hash:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
|
||||
else:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
|
||||
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
# Store high_precision_loras in the model for later retrieval
|
||||
if hasattr(out[0], 'model'):
|
||||
out[0].model._distorch_high_precision_loras = high_precision_loras
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{device};{virtual_vram_gb};{donor_device}" # Changed from compute_device
|
||||
elif expert_mode_allocations: # Only include device if there's an expert string
|
||||
vram_string = device # Changed from compute_device
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchSafetensorV2Clip
|
||||
|
||||
def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor CLIP models"""
|
||||
from . import current_device
|
||||
|
||||
class NodeOverrideDisTorchSafetensorV2ClipNoDevice(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device}) # Changed from compute_device
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
inputs["optional"]["high_precision_loras"] = ("BOOLEAN", {"default": True})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(s, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
# Create a hash of our specific settings
|
||||
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{high_precision_loras}" # Changed from compute_device
|
||||
return hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
def override(self, *args, device=None, virtual_vram_gb=4.0, # Changed from compute_device
|
||||
donor_device="cpu", expert_mode_allocations="", high_precision_loras=True, **kwargs):
|
||||
|
||||
from . import set_current_text_encoder_device # Use text encoder device setter
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
# Register our patched ModelPatcher
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
# Call original function
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
|
||||
# --- Check if we need to unload the model due to settings change ---
|
||||
# This logic is a bit redundant with IS_CHANGED, but provides clear logging
|
||||
settings_str = f"{device}{virtual_vram_gb}{donor_device}{expert_mode_allocations}" # Changed from compute_device
|
||||
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
# Temporarily load to get hash without applying our patch
|
||||
temp_out = fn(*args, **kwargs)
|
||||
model_to_check = None
|
||||
if hasattr(temp_out[0], 'model'):
|
||||
model_to_check = temp_out[0]
|
||||
elif hasattr(temp_out[0], 'patcher') and hasattr(temp_out[0].patcher, 'model'):
|
||||
model_to_check = temp_out[0].patcher
|
||||
|
||||
if model_to_check:
|
||||
model_hash = create_safetensor_model_hash(model_to_check, "override_check")
|
||||
last_settings_hash = safetensor_settings_store.get(model_hash)
|
||||
|
||||
if last_settings_hash != settings_hash:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings changed for model {model_hash[:8]}. Previous settings hash: {last_settings_hash}, New settings hash: {settings_hash}. Forcing reload.")
|
||||
else:
|
||||
logger.info(f"[MultiGPU_DisTorch2] Settings unchanged for model {model_hash[:8]}. Using cached model.")
|
||||
|
||||
out = fn(*args, **kwargs)
|
||||
|
||||
# Store high_precision_loras in the model for later retrieval
|
||||
if hasattr(out[0], 'model'):
|
||||
out[0].model._distorch_high_precision_loras = high_precision_loras
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
out[0].patcher.model._distorch_high_precision_loras = high_precision_loras
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{device};{virtual_vram_gb};{donor_device}" # Changed from compute_device
|
||||
elif expert_mode_allocations: # Only include device if there's an expert string
|
||||
vram_string = device # Changed from compute_device
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logger.info(f"[MultiGPU_DisTorch2] Full allocation string: {full_allocation}")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "override")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchSafetensorV2ClipNoDevice
|
||||
|
||||
@@ -1,920 +0,0 @@
|
||||
{
|
||||
"last_node_id": 115,
|
||||
"last_link_id": 277,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 13,
|
||||
"type": "SamplerCustomAdvanced",
|
||||
"pos": [
|
||||
815.8301391601562,
|
||||
241.12867736816406
|
||||
],
|
||||
"size": [
|
||||
292.4319763183594,
|
||||
479.03521728515625
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": false
|
||||
},
|
||||
"order": 14,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "noise",
|
||||
"type": "NOISE",
|
||||
"link": 37,
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "guider",
|
||||
"type": "GUIDER",
|
||||
"link": 30,
|
||||
"slot_index": 1
|
||||
},
|
||||
{
|
||||
"name": "sampler",
|
||||
"type": "SAMPLER",
|
||||
"link": 19,
|
||||
"slot_index": 2
|
||||
},
|
||||
{
|
||||
"name": "sigmas",
|
||||
"type": "SIGMAS",
|
||||
"link": 20,
|
||||
"slot_index": 3
|
||||
},
|
||||
{
|
||||
"name": "latent_image",
|
||||
"type": "LATENT",
|
||||
"link": 180,
|
||||
"slot_index": 4
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "output",
|
||||
"type": "LATENT",
|
||||
"links": [
|
||||
210
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
},
|
||||
{
|
||||
"name": "denoised_output",
|
||||
"type": "LATENT",
|
||||
"links": null,
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "SamplerCustomAdvanced"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 111,
|
||||
"type": "DownloadAndLoadHyVideoTextEncoderMultiGPU",
|
||||
"pos": [
|
||||
-821.001220703125,
|
||||
504.2577209472656
|
||||
],
|
||||
"size": [
|
||||
371.9022521972656,
|
||||
202
|
||||
],
|
||||
"flags": {},
|
||||
"order": 0,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "hyvid_text_encoder",
|
||||
"type": "HYVIDTEXTENCODER",
|
||||
"links": [
|
||||
269
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DownloadAndLoadHyVideoTextEncoderMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"xtuner/llava-llama-3-8b-v1_1-transformers",
|
||||
"openai/clip-vit-large-patch14",
|
||||
"bf16",
|
||||
false,
|
||||
2,
|
||||
"disabled",
|
||||
"cpu"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 88,
|
||||
"type": "VAELoaderMultiGPU",
|
||||
"pos": [
|
||||
-805.2030639648438,
|
||||
373.4107360839844
|
||||
],
|
||||
"size": [
|
||||
322.5263366699219,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "VAE",
|
||||
"type": "VAE",
|
||||
"links": [
|
||||
275
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "VAELoaderMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"hunyuan_video_vae_bf16.safetensors",
|
||||
"cuda:1"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 109,
|
||||
"type": "HyVideoTextImageEncode",
|
||||
"pos": [
|
||||
-374.6673278808594,
|
||||
532.1463012695312
|
||||
],
|
||||
"size": [
|
||||
295.6000061035156,
|
||||
452.87860107421875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "text_encoders",
|
||||
"type": "HYVIDTEXTENCODER",
|
||||
"link": 269
|
||||
},
|
||||
{
|
||||
"name": "custom_prompt_template",
|
||||
"type": "PROMPT_TEMPLATE",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "clip_l",
|
||||
"type": "CLIP",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "image1",
|
||||
"type": "IMAGE",
|
||||
"link": 272,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "image2",
|
||||
"type": "IMAGE",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "hyvid_cfg",
|
||||
"type": "HYVID_CFG",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "hyvid_embeds",
|
||||
"type": "HYVIDEMBEDS",
|
||||
"links": [
|
||||
270
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "HyVideoTextImageEncode"
|
||||
},
|
||||
"widgets_values": [
|
||||
"The animal shown in <image> appears within its own natural setting, moving calmly or resting in place as a soft light casts delicate shadows across its form. Over the course of five seconds, it makes subtle shifts in posture or position, revealing small details of its features, such as the texture of its skin or fur, and the quiet rhythm of its breathing.",
|
||||
"::4",
|
||||
false,
|
||||
"video",
|
||||
""
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 112,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-787.3131713867188,
|
||||
795.4229736328125
|
||||
],
|
||||
"size": [
|
||||
315,
|
||||
314
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
272
|
||||
],
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage"
|
||||
},
|
||||
"widgets_values": [
|
||||
"pasted/image (224).png",
|
||||
"image"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 103,
|
||||
"type": "Note",
|
||||
"pos": [
|
||||
-835.8104248046875,
|
||||
-79.16381072998047
|
||||
],
|
||||
"size": [
|
||||
353.56494140625,
|
||||
190.77996826171875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"title": "This workflow requires ComfyUI-GGUF",
|
||||
"properties": {},
|
||||
"widgets_values": [
|
||||
"**⚠️ Dependency Alert! ⚠️**\n\nThis workflow relies on nodes from the [ComfyUI-GGUF](https://github.com/city96/ComfyUI-GGUF) custom node repository to function correctly. \n\nSpecifically:\n\n*\"CLIPLoaderGGUFMultiGPU\" \n\nwill not work without this dependency installed. Please install ComfyUI-GGUF before attempting to run this workflow."
|
||||
],
|
||||
"color": "#332922",
|
||||
"bgcolor": "#593930"
|
||||
},
|
||||
{
|
||||
"id": 67,
|
||||
"type": "ModelSamplingSD3",
|
||||
"pos": [
|
||||
-364.1572265625,
|
||||
168.46791076660156
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
58
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": true
|
||||
},
|
||||
"order": 9,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 276
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"links": [
|
||||
252
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "ModelSamplingSD3"
|
||||
},
|
||||
"widgets_values": [
|
||||
7
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 17,
|
||||
"type": "BasicScheduler",
|
||||
"pos": [
|
||||
-367.9955749511719,
|
||||
236.91629028320312
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
109.8011474609375
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": true
|
||||
},
|
||||
"order": 10,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 277,
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "SIGMAS",
|
||||
"type": "SIGMAS",
|
||||
"links": [
|
||||
20
|
||||
],
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "BasicScheduler"
|
||||
},
|
||||
"widgets_values": [
|
||||
"simple",
|
||||
20,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 113,
|
||||
"type": "HunyuanVideoEmbeddingsAdapter",
|
||||
"pos": [
|
||||
-369.37286376953125,
|
||||
363.2193603515625
|
||||
],
|
||||
"size": [
|
||||
283.43841552734375,
|
||||
34.09494400024414
|
||||
],
|
||||
"flags": {},
|
||||
"order": 11,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "hyvid_embeds",
|
||||
"type": "HYVIDEMBEDS",
|
||||
"link": 270
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "CONDITIONING",
|
||||
"type": "CONDITIONING",
|
||||
"links": [
|
||||
271
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "HunyuanVideoEmbeddingsAdapter"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 26,
|
||||
"type": "FluxGuidance",
|
||||
"pos": [
|
||||
-25.9213809967041,
|
||||
325.2367858886719
|
||||
],
|
||||
"size": [
|
||||
211.60000610351562,
|
||||
58
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": true
|
||||
},
|
||||
"order": 12,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "conditioning",
|
||||
"type": "CONDITIONING",
|
||||
"link": 271
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "CONDITIONING",
|
||||
"type": "CONDITIONING",
|
||||
"links": [
|
||||
129
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "FluxGuidance"
|
||||
},
|
||||
"widgets_values": [
|
||||
6
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 22,
|
||||
"type": "BasicGuider",
|
||||
"pos": [
|
||||
206.57337951660156,
|
||||
209.95970153808594
|
||||
],
|
||||
"size": [
|
||||
222.3482666015625,
|
||||
46
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": true
|
||||
},
|
||||
"order": 13,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 252,
|
||||
"slot_index": 0
|
||||
},
|
||||
{
|
||||
"name": "conditioning",
|
||||
"type": "CONDITIONING",
|
||||
"link": 129,
|
||||
"slot_index": 1
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "GUIDER",
|
||||
"type": "GUIDER",
|
||||
"links": [
|
||||
30
|
||||
],
|
||||
"slot_index": 0,
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "BasicGuider"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 16,
|
||||
"type": "KSamplerSelect",
|
||||
"pos": [
|
||||
334.6855163574219,
|
||||
327.40887451171875
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
58
|
||||
],
|
||||
"flags": {
|
||||
"collapsed": true
|
||||
},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "SAMPLER",
|
||||
"type": "SAMPLER",
|
||||
"links": [
|
||||
19
|
||||
],
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "KSamplerSelect"
|
||||
},
|
||||
"widgets_values": [
|
||||
"euler"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 45,
|
||||
"type": "EmptyHunyuanLatentVideo",
|
||||
"pos": [
|
||||
11.676980018615723,
|
||||
448.76055908203125
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
130
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "LATENT",
|
||||
"type": "LATENT",
|
||||
"links": [
|
||||
180
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "EmptyHunyuanLatentVideo"
|
||||
},
|
||||
"widgets_values": [
|
||||
848,
|
||||
480,
|
||||
73,
|
||||
1
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 73,
|
||||
"type": "VAEDecodeTiled",
|
||||
"pos": [
|
||||
503.8817138671875,
|
||||
509.217529296875
|
||||
],
|
||||
"size": [
|
||||
210,
|
||||
150
|
||||
],
|
||||
"flags": {},
|
||||
"order": 15,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "samples",
|
||||
"type": "LATENT",
|
||||
"link": 210
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": 275
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"links": [
|
||||
268
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "VAEDecodeTiled"
|
||||
},
|
||||
"widgets_values": [
|
||||
256,
|
||||
64,
|
||||
64,
|
||||
8
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 102,
|
||||
"type": "VHS_VideoCombine",
|
||||
"pos": [
|
||||
9.632485389709473,
|
||||
698.3294677734375
|
||||
],
|
||||
"size": [
|
||||
451.07391357421875,
|
||||
334
|
||||
],
|
||||
"flags": {},
|
||||
"order": 16,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 268
|
||||
},
|
||||
{
|
||||
"name": "audio",
|
||||
"type": "AUDIO",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "meta_batch",
|
||||
"type": "VHS_BatchManager",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
},
|
||||
{
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": null,
|
||||
"shape": 7
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "Filenames",
|
||||
"type": "VHS_FILENAMES",
|
||||
"links": null
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "VHS_VideoCombine"
|
||||
},
|
||||
"widgets_values": {
|
||||
"frame_rate": 24,
|
||||
"loop_count": 0,
|
||||
"filename_prefix": "HunyuanVideo",
|
||||
"format": "video/h264-mp4",
|
||||
"pix_fmt": "yuv420p",
|
||||
"crf": 19,
|
||||
"save_metadata": true,
|
||||
"trim_to_audio": false,
|
||||
"pingpong": false,
|
||||
"save_output": true,
|
||||
"videopreview": {
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "HunyuanVideo_00323.mp4",
|
||||
"subfolder": "",
|
||||
"type": "output",
|
||||
"format": "video/h264-mp4",
|
||||
"frame_rate": 24,
|
||||
"workflow": "HunyuanVideo_00323.png",
|
||||
"fullpath": "/home/johnj/ComfyUI/output/HunyuanVideo_00323.mp4"
|
||||
},
|
||||
"muted": false
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 115,
|
||||
"type": "UnetLoaderGGUFDisTorchMultiGPU",
|
||||
"pos": [
|
||||
-810.4765014648438,
|
||||
219.3965301513672
|
||||
],
|
||||
"size": [
|
||||
342.1245422363281,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"links": [
|
||||
276,
|
||||
277
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "UnetLoaderGGUFDisTorchMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"hunyuan-video-t2v-720p-Q4_K_M.gguf",
|
||||
"cuda:0",
|
||||
4,
|
||||
false,
|
||||
""
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 25,
|
||||
"type": "RandomNoise",
|
||||
"pos": [
|
||||
368.4217834472656,
|
||||
72.61695861816406
|
||||
],
|
||||
"size": [
|
||||
250.37998962402344,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "NOISE",
|
||||
"type": "NOISE",
|
||||
"links": [
|
||||
37
|
||||
],
|
||||
"shape": 3
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "RandomNoise"
|
||||
},
|
||||
"widgets_values": [
|
||||
5770521,
|
||||
"fixed"
|
||||
],
|
||||
"color": "#2a363b",
|
||||
"bgcolor": "#3f5159"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
19,
|
||||
16,
|
||||
0,
|
||||
13,
|
||||
2,
|
||||
"SAMPLER"
|
||||
],
|
||||
[
|
||||
20,
|
||||
17,
|
||||
0,
|
||||
13,
|
||||
3,
|
||||
"SIGMAS"
|
||||
],
|
||||
[
|
||||
30,
|
||||
22,
|
||||
0,
|
||||
13,
|
||||
1,
|
||||
"GUIDER"
|
||||
],
|
||||
[
|
||||
37,
|
||||
25,
|
||||
0,
|
||||
13,
|
||||
0,
|
||||
"NOISE"
|
||||
],
|
||||
[
|
||||
129,
|
||||
26,
|
||||
0,
|
||||
22,
|
||||
1,
|
||||
"CONDITIONING"
|
||||
],
|
||||
[
|
||||
180,
|
||||
45,
|
||||
0,
|
||||
13,
|
||||
4,
|
||||
"LATENT"
|
||||
],
|
||||
[
|
||||
210,
|
||||
13,
|
||||
0,
|
||||
73,
|
||||
0,
|
||||
"LATENT"
|
||||
],
|
||||
[
|
||||
252,
|
||||
67,
|
||||
0,
|
||||
22,
|
||||
0,
|
||||
"MODEL"
|
||||
],
|
||||
[
|
||||
268,
|
||||
73,
|
||||
0,
|
||||
102,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
269,
|
||||
111,
|
||||
0,
|
||||
109,
|
||||
0,
|
||||
"HYVIDTEXTENCODER"
|
||||
],
|
||||
[
|
||||
270,
|
||||
109,
|
||||
0,
|
||||
113,
|
||||
0,
|
||||
"HYVIDEMBEDS"
|
||||
],
|
||||
[
|
||||
271,
|
||||
113,
|
||||
0,
|
||||
26,
|
||||
0,
|
||||
"CONDITIONING"
|
||||
],
|
||||
[
|
||||
272,
|
||||
112,
|
||||
0,
|
||||
109,
|
||||
3,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
275,
|
||||
88,
|
||||
0,
|
||||
73,
|
||||
1,
|
||||
"VAE"
|
||||
],
|
||||
[
|
||||
276,
|
||||
115,
|
||||
0,
|
||||
67,
|
||||
0,
|
||||
"MODEL"
|
||||
],
|
||||
[
|
||||
277,
|
||||
115,
|
||||
0,
|
||||
17,
|
||||
0,
|
||||
"MODEL"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 2,
|
||||
"title": "GGUFMultiGPU",
|
||||
"bounding": [
|
||||
-836.7138671875,
|
||||
144.54360961914062,
|
||||
403.62188720703125,
|
||||
584.6732788085938
|
||||
],
|
||||
"color": "#8AA",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 1,
|
||||
"offset": {
|
||||
"0": 1075.0079345703125,
|
||||
"1": 156.88380432128906
|
||||
}
|
||||
},
|
||||
"groupNodes": {},
|
||||
"ue_links": [],
|
||||
"VHS_latentpreview": false,
|
||||
"VHS_latentpreviewrate": 0,
|
||||
"VHS_MetadataImage": true,
|
||||
"VHS_KeepIntermediate": true
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
@@ -0,0 +1,364 @@
|
||||
"""
|
||||
Model Management Extensions for MultiGPU
|
||||
Extends ComfyUI's model management with multi-device capabilities and lifecycle tracking.
|
||||
"""
|
||||
|
||||
import torch
|
||||
import logging
|
||||
import hashlib
|
||||
import psutil
|
||||
import comfy.model_management as mm
|
||||
import gc
|
||||
from datetime import datetime, timezone
|
||||
import server
|
||||
import weakref
|
||||
import platform
|
||||
import ctypes
|
||||
import comfy.model_patcher
|
||||
from collections import defaultdict
|
||||
|
||||
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
|
||||
# ==========================================================================================
|
||||
# GC Anchor System for Model Retention
|
||||
# ==========================================================================================
|
||||
|
||||
# Global anchor set to prevent GC of models during selective unload
|
||||
_MGPU_RETENTION_ANCHORS = set()
|
||||
|
||||
def add_retention_anchor(model_patcher, reason="keep_loaded"):
|
||||
"""Add a model patcher to the GC anchor set to prevent premature garbage collection"""
|
||||
if model_patcher is not None:
|
||||
_MGPU_RETENTION_ANCHORS.add(model_patcher)
|
||||
model_name = type(getattr(model_patcher, 'model', model_patcher)).__name__
|
||||
logger.mgpu_mm_log(f"[GC_ANCHOR] Added retention anchor for {model_name}, reason: {reason}, total anchors: {len(_MGPU_RETENTION_ANCHORS)}")
|
||||
|
||||
def clear_all_retention_anchors(reason="manual_clear"):
|
||||
"""Clear all retention anchors"""
|
||||
count = len(_MGPU_RETENTION_ANCHORS)
|
||||
_MGPU_RETENTION_ANCHORS.clear()
|
||||
logger.mgpu_mm_log(f"[GC_ANCHOR] Cleared all {count} retention anchors, reason: {reason}")
|
||||
|
||||
# ==========================================================================================
|
||||
# Model Analysis and Store Management (DisTorch V1 & V2)
|
||||
# ==========================================================================================
|
||||
|
||||
# DisTorch V2 SafeTensor stores
|
||||
safetensor_allocation_store = {}
|
||||
safetensor_settings_store = {}
|
||||
|
||||
# DisTorch V1 GGUF stores (backwards compatibility)
|
||||
model_allocation_store = {}
|
||||
|
||||
def create_safetensor_model_hash(model, caller):
|
||||
"""Create a unique hash for a safetensor model to track allocations"""
|
||||
if hasattr(model, 'model'):
|
||||
actual_model = model.model
|
||||
model_type = type(actual_model).__name__
|
||||
model_size = model.model_size() if hasattr(model, 'model_size') else sum(p.numel() * p.element_size() for p in actual_model.parameters())
|
||||
first_layers = str(list(model.model_state_dict().keys() if hasattr(model, 'model_state_dict') else actual_model.state_dict().keys())[:3])
|
||||
else:
|
||||
model_type = type(model).__name__
|
||||
model_size = sum(p.numel() * p.element_size() for p in model.parameters())
|
||||
first_layers = str(list(model.state_dict().keys())[:3])
|
||||
|
||||
identifier = f"{model_type}_{model_size}_{first_layers}"
|
||||
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Created hash for {caller}: {final_hash[:8]}...")
|
||||
return final_hash
|
||||
|
||||
def create_model_hash(model, caller):
|
||||
"""Create a unique hash for a GGUF model to track allocations (DisTorch V1)"""
|
||||
model_type = type(model.model).__name__
|
||||
model_size = model.model_size()
|
||||
first_layers = str(list(model.model_state_dict().keys())[:3])
|
||||
identifier = f"{model_type}_{model_size}_{first_layers}"
|
||||
final_hash = hashlib.sha256(identifier.encode()).hexdigest()
|
||||
logger.debug(f"[MultiGPU_DisTorch_HASH] Created hash for {caller}: {final_hash[:8]}...")
|
||||
return final_hash
|
||||
|
||||
# ==========================================================================================
|
||||
# Memory Logging Infrastructure
|
||||
# ==========================================================================================
|
||||
|
||||
_MEM_SNAPSHOT_LAST = {}
|
||||
_MEM_SNAPSHOT_SERIES = {}
|
||||
|
||||
def _capture_memory_snapshot():
|
||||
"""Capture memory snapshot for CPU and all devices"""
|
||||
# Import here to avoid circular dependency
|
||||
from .device_utils import get_device_list
|
||||
|
||||
snapshot = {}
|
||||
|
||||
# CPU
|
||||
vm = psutil.virtual_memory()
|
||||
snapshot["cpu"] = (vm.used, vm.total)
|
||||
|
||||
# GPU devices
|
||||
devices = [d for d in get_device_list() if d != "cpu"]
|
||||
for dev_str in devices:
|
||||
device = torch.device(dev_str)
|
||||
total = mm.get_total_memory(device)
|
||||
free_info = mm.get_free_memory(device, torch_free_too=True)
|
||||
system_free = free_info[0] if isinstance(free_info, tuple) else free_info
|
||||
used = max(0, total - system_free)
|
||||
snapshot[dev_str] = (used, total)
|
||||
|
||||
return snapshot
|
||||
|
||||
def multigpu_memory_log(identifier, tag):
|
||||
"""Record timestamped memory snapshot with clean aligned logging"""
|
||||
if identifier == "print_summary":
|
||||
for id_key in sorted(_MEM_SNAPSHOT_SERIES.keys()):
|
||||
series = _MEM_SNAPSHOT_SERIES[id_key]
|
||||
logger.mgpu_mm_log(f"=== memory summary: {id_key} ===")
|
||||
for ts, tag_name, snap in series:
|
||||
parts = []
|
||||
cpu_used, cpu_total = snap.get("cpu", (0, 0))
|
||||
parts.append(f"cpu|{cpu_used/(1024**3):.2f}")
|
||||
for dev in sorted([k for k in snap.keys() if k != "cpu"]):
|
||||
used, total = snap[dev]
|
||||
parts.append(f"{dev}|{used/(1024**3):.2f}")
|
||||
ts_str = ts.strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
|
||||
tag_padded = f"{id_key}_{tag_name}".ljust(35)
|
||||
logger.mgpu_mm_log(f"{ts_str} {tag_padded} {' '.join(parts)}")
|
||||
return
|
||||
|
||||
ts = datetime.now(timezone.utc)
|
||||
curr = _capture_memory_snapshot()
|
||||
|
||||
# Store in series
|
||||
if identifier not in _MEM_SNAPSHOT_SERIES:
|
||||
_MEM_SNAPSHOT_SERIES[identifier] = []
|
||||
_MEM_SNAPSHOT_SERIES[identifier].append((ts, tag, curr))
|
||||
|
||||
# Clean aligned format: timestamp + padded tag + memory values
|
||||
ts_str = ts.strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z"
|
||||
tag_padded = f"{identifier}_{tag}".ljust(35)
|
||||
|
||||
parts = []
|
||||
cpu_used, _ = curr.get("cpu", (0, 0))
|
||||
parts.append(f"cpu|{cpu_used/(1024**3):.2f}")
|
||||
|
||||
for dev in sorted([k for k in curr.keys() if k != "cpu"]):
|
||||
used, _ = curr[dev]
|
||||
parts.append(f"{dev}|{used/(1024**3):.2f}")
|
||||
|
||||
logger.mgpu_mm_log(f"{ts_str} {tag_padded} {' '.join(parts)}")
|
||||
|
||||
_MEM_SNAPSHOT_LAST[identifier] = (tag, curr)
|
||||
|
||||
|
||||
# ==========================================================================================
|
||||
# Memory Management and Cleanup
|
||||
# ==========================================================================================
|
||||
|
||||
CPU_MEMORY_THRESHOLD_PERCENT = 85.0
|
||||
CPU_RESET_HYSTERESIS_PERCENT = 5.0
|
||||
_last_cpu_usage_at_reset = 0.0
|
||||
|
||||
|
||||
def trigger_executor_cache_reset(reason="policy", force=False):
|
||||
"""Trigger PromptExecutor.reset() by setting 'free_memory' flag"""
|
||||
global _last_cpu_usage_at_reset
|
||||
|
||||
prompt_server = server.PromptServer.instance
|
||||
if prompt_server is None:
|
||||
logger.debug("[MultiGPU_Memory_Management] PromptServer not initialized")
|
||||
return
|
||||
|
||||
if prompt_server.prompt_queue.currently_running and not force:
|
||||
logger.debug(f"[MultiGPU_Memory_Management] Skipping reset during execution (reason: {reason})")
|
||||
return
|
||||
|
||||
multigpu_memory_log("executor_reset", f"pre-trigger ({reason})")
|
||||
logger.info(f"[MultiGPU_Memory_Management] Triggering PromptExecutor cache reset. Reason: {reason}")
|
||||
|
||||
prompt_server.prompt_queue.set_flag("free_memory", True)
|
||||
logger.debug("[MultiGPU_Memory_Management] 'free_memory' flag set")
|
||||
|
||||
vm = psutil.virtual_memory()
|
||||
_last_cpu_usage_at_reset = vm.percent
|
||||
|
||||
multigpu_memory_log("executor_reset", f"post-trigger ({reason})")
|
||||
|
||||
def check_cpu_memory_threshold(threshold_percent=CPU_MEMORY_THRESHOLD_PERCENT):
|
||||
"""Check CPU memory and trigger reset if threshold exceeded"""
|
||||
if server.PromptServer.instance is None:
|
||||
return
|
||||
|
||||
if server.PromptServer.instance.prompt_queue.currently_running:
|
||||
return
|
||||
|
||||
vm = psutil.virtual_memory()
|
||||
current_usage = vm.percent
|
||||
|
||||
if current_usage > threshold_percent:
|
||||
if current_usage > (_last_cpu_usage_at_reset + CPU_RESET_HYSTERESIS_PERCENT):
|
||||
logger.warning(f"[MultiGPU_Memory_Monitor] CPU usage ({current_usage:.1f}%) exceeds threshold ({threshold_percent:.1f}%)")
|
||||
multigpu_memory_log("cpu_monitor", f"trigger:{current_usage:.1f}pct")
|
||||
trigger_executor_cache_reset(reason="cpu_threshold_exceeded", force=False)
|
||||
else:
|
||||
logger.debug(f"[MultiGPU_Memory_Monitor] CPU usage high ({current_usage:.1f}%) but within hysteresis")
|
||||
multigpu_memory_log("cpu_monitor", f"skip_hysteresis:{current_usage:.1f}pct")
|
||||
|
||||
def force_full_system_cleanup(reason="manual", force=True):
|
||||
"""Mirror ComfyUI-Manager 'Free model and node cache' by setting unload_models=True and free_memory=True flags."""
|
||||
vm = psutil.virtual_memory()
|
||||
pre_cpu = vm.used
|
||||
pre_models = len(mm.current_loaded_models)
|
||||
|
||||
multigpu_memory_log("full_cleanup", f"start:{reason}")
|
||||
logger.mgpu_mm_log(f"[ManagerMatch] Requesting cleanup (reason={reason}) | pre_models={pre_models}, cpu_used_gib={pre_cpu/(1024**3):.2f}")
|
||||
|
||||
if server.PromptServer.instance is not None:
|
||||
pq = server.PromptServer.instance.prompt_queue
|
||||
if (not pq.currently_running) or force:
|
||||
pq.set_flag("unload_models", True)
|
||||
pq.set_flag("free_memory", True)
|
||||
logger.mgpu_mm_log("[ManagerMatch] Flags set: unload_models=True, free_memory=True")
|
||||
else:
|
||||
logger.mgpu_mm_log("[ManagerMatch] Skipped - execution active and force=False")
|
||||
|
||||
vm = psutil.virtual_memory()
|
||||
post_cpu = vm.used
|
||||
post_models = len(mm.current_loaded_models)
|
||||
delta_cpu_mb = (post_cpu - pre_cpu) / (1024**2)
|
||||
|
||||
multigpu_memory_log("full_cleanup", f"requested:{reason}")
|
||||
summary = f"[ManagerMatch] Cleanup requested (reason={reason}) | models {pre_models}->{post_models}, cpu_delta_mb={delta_cpu_mb:.2f}"
|
||||
logger.mgpu_mm_log(summary)
|
||||
return summary
|
||||
|
||||
# ==========================================================================================
|
||||
# Core Patching: unload_all_models
|
||||
# ==========================================================================================
|
||||
|
||||
if not hasattr(mm.unload_all_models, '_mgpu_eject_distorch_patched'):
|
||||
logger.info("[MultiGPU Core Patching] Patching mm.unload_all_models for DisTorch2 ejection support")
|
||||
|
||||
_mgpu_original_unload_all_models = mm.unload_all_models
|
||||
|
||||
def _mgpu_patched_unload_all_models():
|
||||
"""Patched mm.unload_all_models with selective ejection support and comprehensive diagnostics."""
|
||||
|
||||
logger.mgpu_mm_log(f"[UNLOAD_START] Patched unload_all_models called - initial model count: {len(mm.current_loaded_models)}")
|
||||
|
||||
# Check if there are any DisTorch models that want to be unloaded
|
||||
has_distorch_to_unload = any(
|
||||
(hasattr(lm.model, '_mgpu_unload_distorch_model') and lm.model._mgpu_unload_distorch_model) or
|
||||
(hasattr(getattr(lm.model, 'model', None), '_mgpu_unload_distorch_model') and lm.model.model._mgpu_unload_distorch_model)
|
||||
for lm in mm.current_loaded_models
|
||||
if lm.model is not None
|
||||
)
|
||||
|
||||
if not has_distorch_to_unload:
|
||||
logger.mgpu_mm_log("No DisTorch models requesting unload - clearing anchors and delegating to original unload_all_models")
|
||||
clear_all_retention_anchors(reason="no_selective_unload_needed")
|
||||
_mgpu_original_unload_all_models()
|
||||
return
|
||||
|
||||
# Direct approach: iterate through loaded models and selectively unload
|
||||
models_to_unload = []
|
||||
kept_models = []
|
||||
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
mp = lm.model # weakref call to ModelPatcher
|
||||
|
||||
# DIAGNOSTIC: Log full object chain
|
||||
lm_id = id(lm)
|
||||
mp_id = id(mp)
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
|
||||
# Format inner_model_id properly for f-string
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_READ] Model {i}: lm_id=0x{lm_id:x}, mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
# FIX: Check flag on ModelPatcher (where it was set), not on inner model
|
||||
# OLD BUG: unload_distorch_model = getattr(mp.model, '_mgpu_unload_distorch_model', False)
|
||||
# NEW FIX: Check both locations to see which one has the flag
|
||||
flag_on_mp = getattr(mp, '_mgpu_unload_distorch_model', None)
|
||||
flag_on_inner = getattr(mp.model, '_mgpu_unload_distorch_model', None) if inner_model else None
|
||||
|
||||
logger.mgpu_mm_log(f"[FLAG_CHECK] Model {i} ({inner_model_name}): flag_on_mp={flag_on_mp}, flag_on_inner={flag_on_inner}")
|
||||
|
||||
# Use whichever location has the flag (for backwards compatibility during transition)
|
||||
if flag_on_mp is not None:
|
||||
unload_distorch_model = flag_on_mp
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from ModelPatcher (mp_id=0x{mp_id:x})")
|
||||
elif flag_on_inner is not None:
|
||||
unload_distorch_model = flag_on_inner
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] Using flag from inner model (inner_model_id={inner_id_str})")
|
||||
else:
|
||||
unload_distorch_model = False
|
||||
logger.mgpu_mm_log(f"[FLAG_SOURCE] No flag found - defaulting to False (keep loaded)")
|
||||
|
||||
logger.mgpu_mm_log(f"[DECISION] Model {i} ({inner_model_name}): unload_distorch_model={unload_distorch_model}")
|
||||
|
||||
if unload_distorch_model:
|
||||
models_to_unload.append(lm)
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → models_to_unload")
|
||||
else:
|
||||
kept_models.append(lm)
|
||||
add_retention_anchor(mp, "keep_loaded_protection")
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE] Model {i} ({inner_model_name}) → kept_models")
|
||||
|
||||
# After the kept_models/models_to_unload evaluation
|
||||
logger.mgpu_mm_log(f"[CATEGORIZE_SUMMARY] kept_models: {len(kept_models)}, models_to_unload: {len(models_to_unload)}, total: {len(mm.current_loaded_models)}")
|
||||
|
||||
if len(kept_models) == len(mm.current_loaded_models):
|
||||
# All models are meant to be kept - no DisTorch selective unloading needed
|
||||
logger.mgpu_mm_log("[DELEGATION] All models flagged to be kept - delegating to standard unload_all_models")
|
||||
_mgpu_original_unload_all_models()
|
||||
return
|
||||
|
||||
if kept_models:
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_UNLOAD] Proceeding with selective unload: retaining {len(kept_models)}, unloading {len(models_to_unload)}")
|
||||
|
||||
# Unload models flagged for unload
|
||||
for lm in models_to_unload:
|
||||
try:
|
||||
model_name = type(lm.model.model).__name__ if lm.model and hasattr(lm.model, 'model') else 'Unknown'
|
||||
logger.mgpu_mm_log(f"[UNLOAD_EXECUTE] Unloading model: {model_name} (lm_id=0x{id(lm):x})")
|
||||
lm.model_unload(unpatch_weights=True)
|
||||
except Exception as e:
|
||||
logger.warning(f"[UNLOAD_ERROR] Error unloading model: {e}")
|
||||
|
||||
# WEAKREF TRACKING: Attach weakref callbacks to prove if kept models are GC'd
|
||||
def model_deleted_callback(ref, model_name, model_id):
|
||||
logger.mgpu_mm_log(f"[WEAKREF_DELETED] Kept model GARBAGE COLLECTED: {model_name} (id=0x{model_id:x})")
|
||||
|
||||
for i, lm in enumerate(kept_models):
|
||||
mp = lm.model
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
model_name = type(inner_model).__name__ if inner_model else 'Unknown'
|
||||
model_id = id(lm)
|
||||
weakref.ref(lm, lambda ref, name=model_name, mid=model_id: model_deleted_callback(ref, name, mid))
|
||||
logger.mgpu_mm_log(f"[WEAKREF_ATTACHED] Tracking kept model {i}: {model_name} (lm_id=0x{model_id:x}, mp_id=0x{id(mp):x})")
|
||||
|
||||
# Remove unloaded models from current_loaded_models
|
||||
mm.current_loaded_models = kept_models
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] Updated mm.current_loaded_models, new count: {len(mm.current_loaded_models)}")
|
||||
logger.mgpu_mm_log(f"[SELECTIVE_COMPLETE] mm.current_loaded_models id: 0x{id(mm.current_loaded_models):x}")
|
||||
|
||||
# DIAGNOSTIC: Log what's remaining
|
||||
for i, lm in enumerate(mm.current_loaded_models):
|
||||
mp = lm.model
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
logger.mgpu_mm_log(f"[REMAINING_MODEL] {i}: {model_name} (lm_id=0x{id(lm):x}, mp_id=0x{id(mp):x})")
|
||||
else:
|
||||
logger.mgpu_mm_log("[DELEGATION] No models with keep_loaded=True found - delegating to original unload_all_models")
|
||||
_mgpu_original_unload_all_models()
|
||||
|
||||
mm.unload_all_models = _mgpu_patched_unload_all_models
|
||||
mm.unload_all_models._mgpu_eject_distorch_patched = True
|
||||
logger.info("[MultiGPU Core Patching] mm.unload_all_models patched successfully")
|
||||
else:
|
||||
logger.debug("[MultiGPU Core Patching] mm.unload_all_models already patched - skipping")
|
||||
@@ -3,6 +3,7 @@ import folder_paths
|
||||
from pathlib import Path
|
||||
from nodes import NODE_CLASS_MAPPINGS
|
||||
from .device_utils import get_device_list
|
||||
from .model_management_mgpu import force_full_system_cleanup
|
||||
|
||||
class DeviceSelectorMultiGPU:
|
||||
@classmethod
|
||||
@@ -20,6 +21,7 @@ class DeviceSelectorMultiGPU:
|
||||
CATEGORY = "multigpu"
|
||||
|
||||
def select_device(self, device):
|
||||
"""Select target device from available device list."""
|
||||
return (device,)
|
||||
|
||||
|
||||
@@ -37,6 +39,7 @@ class HunyuanVideoEmbeddingsAdapter:
|
||||
CATEGORY = "multigpu"
|
||||
|
||||
def adapt_embeddings(self, hyvid_embeds):
|
||||
"""Adapt HunyuanVideo embeddings to standard ComfyUI conditioning format."""
|
||||
cond = hyvid_embeds["prompt_embeds"]
|
||||
|
||||
pooled_dict = {
|
||||
@@ -72,6 +75,7 @@ class UnetLoaderGGUF:
|
||||
TITLE = "Unet Loader (GGUF)"
|
||||
|
||||
def load_unet(self, unet_name, dequant_dtype=None, patch_dtype=None, patch_on_device=None):
|
||||
"""Load GGUF format UNet model."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["UnetLoaderGGUF"]()
|
||||
return original_loader.load_unet(unet_name, dequant_dtype, patch_dtype, patch_on_device)
|
||||
|
||||
@@ -109,20 +113,24 @@ class CLIPLoaderGGUF:
|
||||
|
||||
@classmethod
|
||||
def get_filename_list(s):
|
||||
"""Get combined list of CLIP and CLIP_GGUF model files."""
|
||||
files = []
|
||||
files += folder_paths.get_filename_list("clip")
|
||||
files += folder_paths.get_filename_list("clip_gguf")
|
||||
return sorted(files)
|
||||
|
||||
def load_data(self, ckpt_paths):
|
||||
"""Load CLIP model data from checkpoint paths."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
|
||||
return original_loader.load_data(ckpt_paths)
|
||||
|
||||
def load_patcher(self, clip_paths, clip_type, clip_data):
|
||||
"""Create ModelPatcher for CLIP model."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
|
||||
return original_loader.load_patcher(clip_paths, clip_type, clip_data)
|
||||
|
||||
def load_clip(self, clip_name, type="stable_diffusion", device=None):
|
||||
"""Load CLIP model from GGUF or standard format."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["CLIPLoaderGGUF"]()
|
||||
return original_loader.load_clip(clip_name, type)
|
||||
|
||||
@@ -143,6 +151,7 @@ class DualCLIPLoaderGGUF(CLIPLoaderGGUF):
|
||||
TITLE = "DualCLIPLoader (GGUF)"
|
||||
|
||||
def load_clip(self, clip_name1, clip_name2, type, device=None):
|
||||
"""Load dual CLIP model configuration."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["DualCLIPLoaderGGUF"]()
|
||||
clip = original_loader.load_clip(clip_name1, clip_name2, type)
|
||||
clip[0].patcher.load(force_patch_weights=True)
|
||||
@@ -164,6 +173,7 @@ class TripleCLIPLoaderGGUF(CLIPLoaderGGUF):
|
||||
TITLE = "TripleCLIPLoader (GGUF)"
|
||||
|
||||
def load_clip(self, clip_name1, clip_name2, clip_name3, type="sd3"):
|
||||
"""Load triple CLIP model configuration for SD3."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["TripleCLIPLoaderGGUF"]()
|
||||
return original_loader.load_clip(clip_name1, clip_name2, clip_name3, type)
|
||||
|
||||
@@ -183,6 +193,7 @@ class QuadrupleCLIPLoaderGGUF(CLIPLoaderGGUF):
|
||||
TITLE = "QuadrupleCLIPLoader (GGUF)"
|
||||
|
||||
def load_clip(self, clip_name1, clip_name2, clip_name3, clip_name4, type="stable_diffusion"):
|
||||
"""Load quadruple CLIP model configuration."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["QuadrupleCLIPLoaderGGUF"]()
|
||||
return original_loader.load_clip(clip_name1, clip_name2, clip_name3, clip_name4, type)
|
||||
|
||||
@@ -206,12 +217,15 @@ class LTXVLoader:
|
||||
OUTPUT_NODE = False
|
||||
|
||||
def load(self, ckpt_name, dtype):
|
||||
"""Load LTXV model and VAE with specified precision."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
|
||||
return original_loader.load(ckpt_name, dtype)
|
||||
def _load_unet(self, load_device, offload_device, weights, num_latent_channels, dtype, config=None ):
|
||||
"""Load LTXV UNet with device-specific configuration."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
|
||||
return original_loader._load_unet(load_device, offload_device, weights, num_latent_channels, dtype, config=None )
|
||||
def _load_vae(self, weights, config=None):
|
||||
"""Load LTXV VAE from weights."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["LTXVLoader"]()
|
||||
return original_loader._load_vae(weights, config=None)
|
||||
|
||||
@@ -238,6 +252,7 @@ class Florence2ModelLoader:
|
||||
CATEGORY = "Florence2"
|
||||
|
||||
def loadmodel(self, model, precision, attention, lora=None):
|
||||
"""Load Florence2 vision model with specified precision and attention mode."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["Florence2ModelLoader"]()
|
||||
return original_loader.loadmodel(model, precision, attention, lora)
|
||||
|
||||
@@ -285,6 +300,7 @@ class DownloadAndLoadFlorence2Model:
|
||||
CATEGORY = "Florence2"
|
||||
|
||||
def loadmodel(self, model, precision, attention, lora=None):
|
||||
"""Download and load Florence2 model from HuggingFace."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["DownloadAndLoadFlorence2Model"]()
|
||||
return original_loader.loadmodel(model, precision, attention, lora)
|
||||
|
||||
@@ -300,6 +316,7 @@ class CheckpointLoaderNF4:
|
||||
|
||||
|
||||
def load_checkpoint(self, ckpt_name):
|
||||
"""Load checkpoint in NF4 quantized format."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["CheckpointLoaderNF4"]()
|
||||
return original_loader.load_checkpoint(ckpt_name)
|
||||
|
||||
@@ -316,6 +333,7 @@ class LoadFluxControlNet:
|
||||
CATEGORY = "XLabsNodes"
|
||||
|
||||
def loadmodel(self, model_name, controlnet_path):
|
||||
"""Load Flux ControlNet model."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["LoadFluxControlNet"]()
|
||||
return original_loader.loadmodel(model_name, controlnet_path)
|
||||
|
||||
@@ -336,6 +354,7 @@ class MMAudioModelLoader:
|
||||
CATEGORY = "MMAudio"
|
||||
|
||||
def loadmodel(self, mmaudio_model, base_precision):
|
||||
"""Load MMAudio model with specified precision."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["MMAudioModelLoader"]()
|
||||
return original_loader.loadmodel(mmaudio_model, base_precision)
|
||||
|
||||
@@ -363,6 +382,7 @@ class MMAudioFeatureUtilsLoader:
|
||||
CATEGORY = "MMAudio"
|
||||
|
||||
def loadmodel(self, vae_model, precision, synchformer_model, clip_model, mode, bigvgan_vocoder_model=None):
|
||||
"""Load MMAudio feature extraction utilities including VAE, Synchformer, and CLIP."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["MMAudioFeatureUtilsLoader"]()
|
||||
return original_loader.loadmodel(vae_model, precision, synchformer_model, clip_model, mode, bigvgan_vocoder_model)
|
||||
|
||||
@@ -393,6 +413,7 @@ class MMAudioSampler:
|
||||
CATEGORY = "MMAudio"
|
||||
|
||||
def sample(self, mmaudio_model, seed, feature_utils, duration, steps, cfg, prompt, negative_prompt, mask_away_clip, force_offload, images=None):
|
||||
"""Sample audio from MMAudio model with conditioning."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["MMAudioSampler"]()
|
||||
return original_loader.sample(mmaudio_model, seed, feature_utils, duration, steps, cfg, prompt, negative_prompt, mask_away_clip, force_offload, images)
|
||||
|
||||
@@ -406,6 +427,7 @@ class PulidModelLoader:
|
||||
CATEGORY = "pulid"
|
||||
|
||||
def load_model(self, pulid_file):
|
||||
"""Load PuLID identity preservation model."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["PulidModelLoader"]()
|
||||
return original_loader.load_model(pulid_file)
|
||||
|
||||
@@ -423,6 +445,7 @@ class PulidInsightFaceLoader:
|
||||
CATEGORY = "pulid"
|
||||
|
||||
def load_insightface(self, provider):
|
||||
"""Load InsightFace face analysis model for PuLID."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["PulidInsightFaceLoader"]()
|
||||
return original_loader.load_insightface(provider)
|
||||
|
||||
@@ -438,6 +461,7 @@ class PulidEvaClipLoader:
|
||||
CATEGORY = "pulid"
|
||||
|
||||
def load_eva_clip(self):
|
||||
"""Load EVA CLIP model for PuLID."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["PulidEvaClipLoader"]()
|
||||
return original_loader.load_eva_clip()
|
||||
|
||||
@@ -472,6 +496,7 @@ class HyVideoModelLoader:
|
||||
CATEGORY = "HunyuanVideoWrapper"
|
||||
|
||||
def loadmodel(self, model, base_precision, load_device, quantization, compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None, auto_cpu_offload=False):
|
||||
"""Load HunyuanVideo model with specified precision and quantization."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["HyVideoModelLoader"]()
|
||||
return original_loader.loadmodel(model, base_precision, load_device, quantization, compile_args, attention_mode, block_swap_args, lora, auto_cpu_offload)
|
||||
|
||||
@@ -497,6 +522,7 @@ class HyVideoVAELoader:
|
||||
DESCRIPTION = "Loads Hunyuan VAE model from 'ComfyUI/models/vae'"
|
||||
|
||||
def loadmodel(self, model_name, precision, compile_args=None):
|
||||
"""Load HunyuanVideo VAE model."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["HyVideoVAELoader"]()
|
||||
return original_loader.loadmodel(model_name, precision, compile_args)
|
||||
|
||||
@@ -525,5 +551,31 @@ class DownloadAndLoadHyVideoTextEncoder:
|
||||
DESCRIPTION = "Loads Hunyuan text_encoder model from 'ComfyUI/models/LLM'"
|
||||
|
||||
def loadmodel(self, llm_model, clip_model, precision, apply_final_norm=False, hidden_state_skip_layer=2, quantization="disabled"):
|
||||
"""Download and load HunyuanVideo text encoder from HuggingFace."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["DownloadAndLoadHyVideoTextEncoder"]()
|
||||
return original_loader.loadmodel(llm_model, clip_model, precision, apply_final_norm, hidden_state_skip_layer, quantization)
|
||||
|
||||
|
||||
class UNetLoaderLP:
|
||||
"""UNet Loader (Low Precision) - sets LoRA precision to False for CPU storage optimization"""
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {"required": { "unet_name": (folder_paths.get_filename_list("unet"), ),
|
||||
}}
|
||||
RETURN_TYPES = ("MODEL",)
|
||||
FUNCTION = "load_unet"
|
||||
CATEGORY = "loaders"
|
||||
TITLE = "UNet Loader (LP)"
|
||||
|
||||
def load_unet(self, unet_name):
|
||||
"""Load UNet with low-precision LoRA flag for CPU storage optimization."""
|
||||
original_loader = NODE_CLASS_MAPPINGS["UNETLoader"]()
|
||||
out = original_loader.load_unet(unet_name)
|
||||
|
||||
# Set the low-precision LoRA flag on the loaded model
|
||||
if hasattr(out[0], 'model'):
|
||||
out[0].model._distorch_high_precision_loras = False
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
out[0].patcher.model._distorch_high_precision_loras = False
|
||||
|
||||
return out
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "comfyui-multigpu"
|
||||
description = "Provides a suite of custom nodes to manage multiple GPUs for ComfyUI, including advanced model offloading for both GGUF and Safetensor formats with DisTorch, and bespoke MultiGPU support for WanVideoWrapper and other custom nodes."
|
||||
version = "2.4.7"
|
||||
version = "2.5.0"
|
||||
license = {file = "LICENSE"}
|
||||
|
||||
[project.urls]
|
||||
|
||||
+33
-1
@@ -4,7 +4,7 @@ import sys
|
||||
import inspect
|
||||
import folder_paths
|
||||
import comfy.model_management as mm
|
||||
from .device_utils import get_device_list
|
||||
from .device_utils import get_device_list, comfyui_memory_load
|
||||
|
||||
class WanVideoModelLoader:
|
||||
@classmethod
|
||||
@@ -89,8 +89,16 @@ class WanVideoModelLoader:
|
||||
logging.debug(f"[MultiGPU] Both WanVideo modules patched successfully")
|
||||
|
||||
logging.debug(f"[MultiGPU] Calling original WanVideo loader")
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"pre-model-load:wan-model:{model}"))
|
||||
except Exception:
|
||||
pass
|
||||
result = original_loader.loadmodel(model, base_precision, load_device, quantization,
|
||||
compile_args, attention_mode, block_swap_args, lora, vram_management_args, extra_model=extra_model, fantasytalking_model=fantasytalking_model, multitalk_model=multitalk_model, fantasyportrait_model=fantasyportrait_model)
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"post-model-load:wan-model:{model}"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if result and len(result) > 0 and hasattr(result[0], 'model'):
|
||||
model_obj = result[0]
|
||||
@@ -156,7 +164,15 @@ class WanVideoVAELoader:
|
||||
setattr(nodes_module, 'device', selected_device)
|
||||
setattr(nodes_module, 'offload_device', selected_device)
|
||||
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"pre-model-load:wan-vae:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
result = original_loader.loadmodel(model_name, precision, compile_args)
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"post-model-load:wan-vae:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Attach device info to VAE object for downstream nodes
|
||||
if result and len(result) > 0:
|
||||
@@ -219,7 +235,15 @@ class LoadWanVideoT5TextEncoder:
|
||||
if device == "cpu":
|
||||
setattr(nodes_module, 'offload_device', selected_device)
|
||||
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"pre-model-load:wan-textenc:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
result = original_loader.loadmodel(model_name, precision, load_device, quantization)
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"post-model-load:wan-textenc:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
logging.info(f"[MultiGPU] WanVideo T5 Text encoder loaded on {selected_device}")
|
||||
|
||||
@@ -331,7 +355,15 @@ class LoadWanVideoClipTextEncoder:
|
||||
if device == "cpu":
|
||||
setattr(nodes_module, 'offload_device', selected_device)
|
||||
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"pre-model-load:wan-clip:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
result = original_loader.loadmodel(model_name, precision, load_device)
|
||||
try:
|
||||
logging.info(comfyui_memory_load(f"post-model-load:wan-clip:{model_name}"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
logging.info(f"[MultiGPU] WanVideo CLIP encoder loaded on {selected_device}")
|
||||
|
||||
|
||||
@@ -0,0 +1,52 @@
|
||||
# CLIPLoaderDisTorch2MultiGPU
|
||||
|
||||
The `CLIPLoaderDisTorch2MultiGPU` node is used to load standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name` | `STRING` | The name of the CLIP model to load. |
|
||||
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,52 @@
|
||||
# CLIPLoaderGGUFDisTorch2MultiGPU
|
||||
|
||||
The `CLIPLoaderGGUFDisTorch2MultiGPU` node is used to load GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name` | `STRING` | The name of the CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,19 @@
|
||||
# CLIPLoaderGGUFMultiGPU
|
||||
|
||||
The `CLIPLoaderGGUFMultiGPU` node is used to load GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name` | `STRING` | The name of the CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
|
||||
@@ -0,0 +1,19 @@
|
||||
# CLIPLoaderMultiGPU
|
||||
|
||||
The `CLIPLoaderMultiGPU` node is used to load CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name` | `STRING` | The name of the CLIP model to load. |
|
||||
| `type` | `STRING` | The type of CLIP model (e.g., 'stable_diffusion', 'stable_diffusion_xl'). |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# CLIPVisionLoaderDisTorch2MultiGPU
|
||||
|
||||
The `CLIPVisionLoaderDisTorch2MultiGPU` node is used to load CLIP Vision models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger vision encoder models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip_vision` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_vision` | `STRING` | The name of the CLIP Vision model to load. |
|
||||
| `device` | `STRING` | Target device for vision encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP_VISION` | `CLIP_VISION` | The loaded CLIP Vision model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# CLIPVisionLoaderMultiGPU
|
||||
|
||||
The `CLIPVisionLoaderMultiGPU` node is used to load CLIP Vision models with device selection capability, enabling users to specify which GPU or device should be used for vision encoder execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip_vision` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_vision` | `STRING` | The name of the CLIP Vision model to load. |
|
||||
| `device` | `STRING` | Target device for vision encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP_VISION` | `CLIP_VISION` | The loaded CLIP Vision model. |
|
||||
@@ -0,0 +1,55 @@
|
||||
# CheckpointLoaderAdvancedDisTorch2MultiGPU
|
||||
|
||||
The `CheckpointLoaderAdvancedDisTorch2MultiGPU` node is used to load checkpoint models with advanced DisTorch2 distributed tensor allocation, providing granular control over UNet, CLIP, and VAE component allocation across multiple devices with independent virtual VRAM management.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
|
||||
| `unet_compute_device` | `STRING` | Target compute device for UNet distributed allocation (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
|
||||
| `unet_virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes for UNet component distributed allocation (default: 4.0, range: 0.0-128.0). |
|
||||
| `unet_donor_device` | `STRING` | Device to donate VRAM from when allocating UNet virtual memory (default: 'cpu'). |
|
||||
| `clip_compute_device` | `STRING` | Target compute device for CLIP distributed allocation (default: 'cpu'). |
|
||||
| `clip_virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes for CLIP component distributed allocation (default: 2.0, range: 0.0-128.0). |
|
||||
| `clip_donor_device` | `STRING` | Device to donate VRAM from when allocating CLIP virtual memory (default: 'cpu'). |
|
||||
| `vae_device` | `STRING` | Target device for the VAE component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
|
||||
| `unet_expert_mode_allocations` | `STRING` | Advanced UNet allocation string for expert device/ratio distributions. |
|
||||
| `clip_expert_mode_allocations` | `STRING` | Advanced CLIP allocation string for expert device/ratio distributions. |
|
||||
| `high_precision_loras` | `BOOLEAN` | Whether to use high-precision LoRA patches (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet diffusion model with DisTorch2 distributed allocation. |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model with DisTorch2 distributed allocation. |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
This advanced checkpoint loader provides independent DisTorch2 allocation control for UNet and CLIP components, while using standard device placement for VAE.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Individual Component Control**: Each model component (UNet, CLIP, VAE) can have its own allocation strategy.
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on compute devices by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of each component should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `unet_compute_device`: `cuda:0`, `unet_virtual_vram_gb`: `8.0`, `unet_donor_device`: `cuda:1`
|
||||
- `clip_compute_device`: `cuda:1`, `clip_virtual_vram_gb`: `2.0`, `clip_donor_device`: `cpu`
|
||||
- Result: UNet loads as if cuda:0 has 8GB more VRAM, CLIP loads with cuda:1 having 2GB more capacity.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `unet_expert_mode_allocations`: `cuda:0,70%;cuda:1,30%`
|
||||
- `clip_expert_mode_allocations`: `cuda:1,50%;cpu,50%`
|
||||
- Distributes UNet with 70% on GPU 0, 30% on GPU 1, and CLIP with 50% on GPU 1, 50% on CPU.
|
||||
@@ -0,0 +1,22 @@
|
||||
# CheckpointLoaderAdvancedMultiGPU
|
||||
|
||||
The `CheckpointLoaderAdvancedMultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with granular device control, allowing individual placement of each model component on different GPUs or devices.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
|
||||
| `unet_device` | `STRING` | Target device for the UNet diffusion model component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
|
||||
| `clip_device` | `STRING` | Target device for the CLIP text encoder component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
|
||||
| `vae_device` | `STRING` | Target device for the VAE decoder/encoder component (e.g., 'cuda:0', 'cuda:1', 'cpu'). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
|
||||
@@ -0,0 +1,53 @@
|
||||
# CheckpointLoaderSimpleDisTorch2MultiGPU
|
||||
|
||||
The `CheckpointLoaderSimpleDisTorch2MultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet diffusion model with DisTorch2 distributed allocation applied. |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,20 @@
|
||||
# CheckpointLoaderSimpleMultiGPU
|
||||
|
||||
The `CheckpointLoaderSimpleMultiGPU` node is used to load checkpoint models (complete diffusion models containing UNet, CLIP, and VAE components) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/checkpoints` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `ckpt_name` | `STRING` | The name of the checkpoint model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
|
||||
| `CLIP` | `CLIP` | The loaded CLIP text encoder model. |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# ControlNetLoaderDisTorch2MultiGPU
|
||||
|
||||
The `ControlNetLoaderDisTorch2MultiGPU` node is used to load ControlNet models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger conditional generation models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/controlnet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `control_net_name` | `STRING` | The name of the ControlNet model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CONTROL_NET` | `CONTROL_NET` | The loaded ControlNet model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# ControlNetLoaderMultiGPU
|
||||
|
||||
The `ControlNetLoaderMultiGPU` node is used to load ControlNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/controlnet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `control_net_name` | `STRING` | The name of the ControlNet model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CONTROL_NET` | `CONTROL_NET` | The loaded ControlNet model. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# DiffControlNetLoaderDisTorch2MultiGPU
|
||||
|
||||
The `DiffControlNetLoaderDisTorch2MultiGPU` node is used to load Diffusers ControlNet models (HuggingFace Hub repositories) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger conditional generation models across multiple GPUs.
|
||||
|
||||
This node loads ControlNet models directly from HuggingFace model repositories by specifying the repository ID (e.g., "diffusers/controlnet-canny-sdxl-1.0").
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers ControlNet model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CONTROL_NET` | `CONTROL_NET` | The loaded diffusers ControlNet model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# DiffControlNetLoaderMultiGPU
|
||||
|
||||
The `DiffControlNetLoaderMultiGPU` node is used to load Diffusers ControlNet models (HuggingFace Hub repositories) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node loads ControlNet models directly from HuggingFace model repositories by specifying the repository ID (e.g., "diffusers/controlnet-canny-sdxl-1.0").
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers ControlNet model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CONTROL_NET` | `CONTROL_NET` | The loaded diffusers ControlNet model. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# DiffusersLoaderDisTorch2MultiGPU
|
||||
|
||||
The `DiffusersLoaderDisTorch2MultiGPU` node is used to load Diffusers models (HuggingFace Hub repositories) with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger diffusion models across multiple GPUs.
|
||||
|
||||
This node loads models directly from HuggingFace model repositories by specifying the repository ID (e.g., "stabilityai/stable-diffusion-xl-base-1.0").
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers model to load (e.g., 'stabilityai/stable-diffusion-xl-base-1.0'). |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded diffusers model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# DiffusersLoaderMultiGPU
|
||||
|
||||
The `DiffusersLoaderMultiGPU` node is used to load Diffusers models (HuggingFace Hub repositories) with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node loads models directly from HuggingFace model repositories by specifying the repository ID (e.g., "stabilityai/stable-diffusion-xl-base-1.0").
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `model_path` | `STRING` | The HuggingFace repository ID or local path of the diffusers model to load (e.g., 'stabilityai/stable-diffusion-xl-base-1.0'). |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded diffusers model. |
|
||||
@@ -0,0 +1,53 @@
|
||||
# DualCLIPLoaderDisTorch2MultiGPU
|
||||
|
||||
The `DualCLIPLoaderDisTorch2MultiGPU` node is used to load dual standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,53 @@
|
||||
# DualCLIPLoaderGGUFDisTorch2MultiGPU
|
||||
|
||||
The `DualCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load dual GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,20 @@
|
||||
# DualCLIPLoaderGGUFMultiGPU
|
||||
|
||||
The `DualCLIPLoaderGGUFMultiGPU` node is used to load dual GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models. |
|
||||
@@ -0,0 +1,20 @@
|
||||
# DualCLIPLoaderMultiGPU
|
||||
|
||||
The `DualCLIPLoaderMultiGPU` node is used to load dual CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `type` | `STRING` | The type of CLIP model configuration for dual loading. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded dual CLIP text encoder models. |
|
||||
@@ -0,0 +1,54 @@
|
||||
# QuadrupleCLIPLoaderDisTorch2MultiGPU
|
||||
|
||||
The `QuadrupleCLIPLoaderDisTorch2MultiGPU` node is used to load quadruple standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
|
||||
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,54 @@
|
||||
# QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU
|
||||
|
||||
The `QuadrupleCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load quadruple GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,21 @@
|
||||
# QuadrupleCLIPLoaderGGUFMultiGPU
|
||||
|
||||
The `QuadrupleCLIPLoaderGGUFMultiGPU` node is used to load quadruple GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models. |
|
||||
@@ -0,0 +1,21 @@
|
||||
# QuadrupleCLIPLoaderMultiGPU
|
||||
|
||||
The `QuadrupleCLIPLoaderMultiGPU` node is used to load quadruple CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
|
||||
| `clip_name4` | `STRING` | The name of the fourth CLIP model to load. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded quadruple CLIP text encoder models. |
|
||||
@@ -0,0 +1,53 @@
|
||||
# TripleCLIPLoaderDisTorch2MultiGPU
|
||||
|
||||
The `TripleCLIPLoaderDisTorch2MultiGPU` node is used to load triple standard CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3 with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,53 @@
|
||||
# TripleCLIPLoaderGGUFDisTorch2MultiGPU
|
||||
|
||||
The `TripleCLIPLoaderGGUFDisTorch2MultiGPU` node is used to load triple GGUF format CLIP text encoder models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger text encoding models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3 with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,20 @@
|
||||
# TripleCLIPLoaderGGUFMultiGPU
|
||||
|
||||
The `TripleCLIPLoaderGGUFMultiGPU` node is used to load triple GGUF format CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` and `ComfyUI/models/clip_gguf` folders, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load from combined clip and clip_gguf folders. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3. |
|
||||
@@ -0,0 +1,20 @@
|
||||
# TripleCLIPLoaderMultiGPU
|
||||
|
||||
The `TripleCLIPLoaderMultiGPU` node is used to load triple CLIP text encoder models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/clip` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `clip_name1` | `STRING` | The name of the first CLIP model to load. |
|
||||
| `clip_name2` | `STRING` | The name of the second CLIP model to load. |
|
||||
| `clip_name3` | `STRING` | The name of the third CLIP model to load. |
|
||||
| `device` | `STRING` | Target device for text encoder compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `CLIP` | `CLIP` | The loaded triple CLIP text encoder models configured for SD3. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# UNETLoaderDisTorch2MultiGPU
|
||||
|
||||
The `UNETLoaderDisTorch2MultiGPU` node is used to load UNet diffusion models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the UNet model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# UNETLoaderMultiGPU
|
||||
|
||||
The `UNETLoaderMultiGPU` node is used to load diffusion model UNet components with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the UNet model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded UNet diffusion model. |
|
||||
@@ -0,0 +1,54 @@
|
||||
# UnetLoaderGGUFAdvancedDisTorch2MultiGPU
|
||||
|
||||
The `UnetLoaderGGUFAdvancedDisTorch2MultiGPU` node is used to load GGUF format UNet models with advanced quantization options and DisTorch2 distributed tensor allocation, enabling sophisticated multi-device VRAM management across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
|
||||
| `dequant_dtype` | `STRING` | Target data type for model dequantization during loading (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
|
||||
| `patch_dtype` | `STRING` | Data type for LoRA patches applied to the model (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
|
||||
| `patch_on_device` | `BOOLEAN` | Whether to apply LoRA patches directly on the target device. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with advanced quantization settings and DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,21 @@
|
||||
# UnetLoaderGGUFAdvancedMultiGPU
|
||||
|
||||
The `UnetLoaderGGUFAdvancedMultiGPU` node is used to load GGUF format UNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
|
||||
| `dequant_dtype` | `STRING` | Target data type for model dequantization during loading (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
|
||||
| `patch_dtype` | `STRING` | Data type for LoRA patches applied to the model (options: 'default', 'target', 'float32', 'float16', 'bfloat16'). |
|
||||
| `patch_on_device` | `BOOLEAN` | Whether to apply LoRA patches directly on the target device. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with advanced quantization settings. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# UnetLoaderGGUFDisTorch2MultiGPU
|
||||
|
||||
The `UnetLoaderGGUFDisTorch2MultiGPU` node is used to load GGUF format UNet models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded GGUF format UNet model with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# UnetLoaderGGUFMultiGPU
|
||||
|
||||
The `UnetLoaderGGUFMultiGPU` node is used to load GGUF format UNet models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/unet_gguf` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `unet_name` | `STRING` | The name of the GGUF format UNet model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `MODEL` | `MODEL` | The loaded GGUF format UNet model. |
|
||||
@@ -0,0 +1,51 @@
|
||||
# VAELoaderDisTorch2MultiGPU
|
||||
|
||||
The `VAELoaderDisTorch2MultiGPU` node is used to load VAE (Variational Autoencoder) models with DisTorch2 distributed tensor allocation, enabling advanced multi-device VRAM management to handle larger models across multiple GPUs.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/vae` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `vae_name` | `STRING` | The name of the VAE model to load. |
|
||||
| `compute_device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
| `virtual_vram_gb` | `FLOAT` | Amount of virtual VRAM in gigabytes to allocate for distributed tensor management (default: 4.0, range: 0.0-128.0). |
|
||||
| `donor_device` | `STRING` | Device to donate VRAM from when allocating virtual memory (default: 'cpu'). |
|
||||
| `expert_mode_allocations` | `STRING` | Advanced allocation string for expert users to manually specify device/ratio distributions (e.g., 'cuda:0,50%;cpu,*'). |
|
||||
| `keep_loaded` | `BOOLEAN` | Whether to keep the model loaded when triggering memory cleanup operations (default: true). |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder with DisTorch2 distributed allocation applied. |
|
||||
|
||||
## DisTorch2 Distributed Loading
|
||||
|
||||
DisTorch2 is an advanced memory management system that enables loading and running large diffusion models across multiple GPUs by intelligently distributing tensor allocations. Instead of loading an entire model on a single device, DisTorch2 splits the model's layers across available devices while maintaining computational efficiency.
|
||||
|
||||
### Key Concepts
|
||||
|
||||
**Virtual VRAM Allocation**: Artificially increases the available VRAM on the compute device by borrowing memory capacity from donor devices through intelligent tensor distribution.
|
||||
|
||||
**Expert Mode Allocations**: Advanced users can manually specify exactly how much of the model should be placed on each device using ratio or byte-based allocation strings.
|
||||
|
||||
### Allocation Examples
|
||||
|
||||
**Basic Virtual VRAM Mode**:
|
||||
- `compute_device`: `cuda:0`
|
||||
- `virtual_vram_gb`: `8.0`
|
||||
- `donor_device`: `cuda:1`
|
||||
- Result: Loads model as if cuda:0 had 8GB more VRAM available, using cuda:1 as memory donor.
|
||||
|
||||
**Expert Ratio Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,60%;cuda:1,30%;cpu,10%`
|
||||
- Distributes model layers with 60% on GPU 0, 30% on GPU 1, and 10% on CPU.
|
||||
|
||||
**Expert Byte Allocation**:
|
||||
- `expert_mode_allocations`: `cuda:0,4gb;cuda:1,2gb;cpu,*`
|
||||
- Allocates exactly 4GB to cuda:0, 2GB to cuda:1, and remaining to CPU.
|
||||
|
||||
**Mixed Mode**:
|
||||
Combines virtual VRAM with expert allocations for complex multi-device scenarios.
|
||||
@@ -0,0 +1,18 @@
|
||||
# VAELoaderMultiGPU
|
||||
|
||||
The `VAELoaderMultiGPU` node is used to load VAE (Variational Autoencoder) models with device selection capability, enabling users to specify which GPU or device should be used for model execution.
|
||||
|
||||
This node automatically detects models located in the `ComfyUI/models/vae` folder, and it will also read models from additional paths configured in the `extra_model_paths.yaml` file. Sometimes, you may need to **refresh the ComfyUI interface** to allow it to read the model files from the corresponding folder.
|
||||
|
||||
## Inputs
|
||||
|
||||
| Parameter | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `vae_name` | `STRING` | The name of the VAE model to load. |
|
||||
| `device` | `STRING` | Target device for compute operations (e.g., 'cuda:0', 'cuda:1', 'cpu'). Selected from available devices on your system. |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Output Name | Data Type | Description |
|
||||
| --- | --- | --- |
|
||||
| `VAE` | `VAE` | The loaded VAE decoder/encoder model. |
|
||||
+520
@@ -0,0 +1,520 @@
|
||||
"""
|
||||
ComfyUI-MultiGPU Wrapper Functions
|
||||
All node override/wrapper generation functions consolidated in one location
|
||||
"""
|
||||
|
||||
import copy
|
||||
import hashlib
|
||||
import logging
|
||||
from .device_utils import get_device_list
|
||||
|
||||
logger = logging.getLogger("MultiGPU")
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# DISTORCH V2 SAFETENSOR WRAPPERS (DisTorch2 for .safetensors and .gguf)
|
||||
# ============================================================================
|
||||
|
||||
def _create_distorch_safetensor_v2_override(cls, device_param_name, device_setter_func, apply_device_kwarg_workaround):
|
||||
"""Internal factory function creating DisTorch2 override class with parameterized device selection behavior."""
|
||||
from .distorch_2 import (
|
||||
register_patched_safetensor_modelpatcher,
|
||||
safetensor_allocation_store,
|
||||
safetensor_settings_store,
|
||||
create_safetensor_model_hash
|
||||
)
|
||||
from .model_management_mgpu import force_full_system_cleanup
|
||||
|
||||
class NodeOverrideDisTorchSafetensorV2(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"][device_param_name] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
inputs["optional"]["keep_loaded"] = ("BOOLEAN", {"default": True})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
@classmethod
|
||||
def IS_CHANGED(s, *args, virtual_vram_gb=4.0, donor_device="cpu",
|
||||
expert_mode_allocations="", keep_loaded=True, **kwargs):
|
||||
device_value = kwargs.get(device_param_name)
|
||||
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}{keep_loaded}"
|
||||
current_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
if not hasattr(cls, '_last_hash'):
|
||||
cls._last_hash = current_hash
|
||||
logger.mgpu_mm_log(f"IS_CHANGED first call: {current_hash[:8]}")
|
||||
elif cls._last_hash != current_hash:
|
||||
cls._last_hash = current_hash
|
||||
logger.mgpu_mm_log(f"IS_CHANGED CHANGED: {current_hash[:8]} ← settings changed")
|
||||
return current_hash
|
||||
|
||||
def override(self, *args, virtual_vram_gb=4.0, donor_device="cpu",
|
||||
expert_mode_allocations="", keep_loaded=True, **kwargs):
|
||||
|
||||
device_value = kwargs.get(device_param_name)
|
||||
unload_distorch_model = not keep_loaded
|
||||
|
||||
if device_value is not None:
|
||||
device_setter_func(device_value)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in [device_param_name, 'virtual_vram_gb',
|
||||
'donor_device', 'expert_mode_allocations',
|
||||
'keep_loaded']}
|
||||
|
||||
if apply_device_kwarg_workaround:
|
||||
clean_kwargs['device'] = 'default'
|
||||
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{device_value};{virtual_vram_gb};{donor_device}"
|
||||
elif expert_mode_allocations:
|
||||
vram_string = device_value
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **clean_kwargs)
|
||||
|
||||
model_to_check = None
|
||||
if hasattr(out[0], 'model'):
|
||||
model_to_check = out[0]
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_to_check = out[0].patcher
|
||||
|
||||
if model_to_check:
|
||||
model_hash = create_safetensor_model_hash(model_to_check, "override_store")
|
||||
settings_str = f"{device_value}{virtual_vram_gb}{donor_device}{expert_mode_allocations}"
|
||||
settings_hash = hashlib.sha256(settings_str.encode()).hexdigest()
|
||||
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
safetensor_settings_store[model_hash] = settings_hash
|
||||
logger.debug(f"[MultiGPU DisTorch V2] Stored allocation for model {model_hash[:8]}: {full_allocation}")
|
||||
|
||||
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_START] Setting '_mgpu_unload_distorch_model' to: {unload_distorch_model} (keep_loaded={keep_loaded})")
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
mp = out[0]
|
||||
mp_id = id(mp)
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
mp._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
|
||||
|
||||
if inner_model:
|
||||
inner_model._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
|
||||
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
mp = out[0].patcher
|
||||
mp_id = id(mp)
|
||||
inner_model = getattr(mp, 'model', None)
|
||||
inner_model_id = id(inner_model) if inner_model else None
|
||||
inner_model_name = type(inner_model).__name__ if inner_model else "None"
|
||||
inner_id_str = f"0x{inner_model_id:x}" if inner_model_id is not None else "None"
|
||||
|
||||
logger.mgpu_mm_log(f"[OBJECT_CHAIN_SET] ModelPatcher via patcher: mp_id=0x{mp_id:x}, inner_model_id={inner_id_str}, inner_model_type={inner_model_name}")
|
||||
|
||||
mp._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_LOCATION] Set on ModelPatcher (mp_id=0x{mp_id:x}): mp._mgpu_unload_distorch_model = {unload_distorch_model}")
|
||||
|
||||
if inner_model:
|
||||
inner_model._mgpu_unload_distorch_model = unload_distorch_model
|
||||
logger.mgpu_mm_log(f"[FLAG_SET_COMPAT] Also set on inner model (inner_model_id=0x{inner_model_id:x}) for compatibility")
|
||||
|
||||
if unload_distorch_model:
|
||||
logger.mgpu_mm_log("[FLAG_TRIGGER] unload_distorch_model=True, triggering full system cleanup")
|
||||
force_full_system_cleanup(reason="policy_every_load", force=True)
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchSafetensorV2
|
||||
|
||||
|
||||
def override_class_with_distorch_safetensor_v2(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor UNet/VAE models"""
|
||||
from . import set_current_device
|
||||
return _create_distorch_safetensor_v2_override(
|
||||
cls,
|
||||
device_param_name="compute_device",
|
||||
device_setter_func=set_current_device,
|
||||
apply_device_kwarg_workaround=False
|
||||
)
|
||||
|
||||
|
||||
def override_class_with_distorch_safetensor_v2_clip(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor CLIP models (with device kwarg workaround)"""
|
||||
from . import set_current_text_encoder_device
|
||||
return _create_distorch_safetensor_v2_override(
|
||||
cls,
|
||||
device_param_name="device",
|
||||
device_setter_func=set_current_text_encoder_device,
|
||||
apply_device_kwarg_workaround=True
|
||||
)
|
||||
|
||||
|
||||
def override_class_with_distorch_safetensor_v2_clip_no_device(cls):
|
||||
"""DisTorch 2.0 wrapper for safetensor Triple/Quad CLIP models (no device kwarg workaround)"""
|
||||
from . import set_current_text_encoder_device
|
||||
return _create_distorch_safetensor_v2_override(
|
||||
cls,
|
||||
device_param_name="device",
|
||||
device_setter_func=set_current_text_encoder_device,
|
||||
apply_device_kwarg_workaround=False
|
||||
)
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# DISTORCH V1 LEGACY WRAPPERS (Rewritten to call V2 backend)
|
||||
# ============================================================================
|
||||
|
||||
def override_class_with_distorch_gguf(cls):
|
||||
"""DisTorch V1 Legacy wrapper - maintains V1 UI but calls V2 backend"""
|
||||
from . import set_current_device
|
||||
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
|
||||
|
||||
class NodeOverrideDisTorchGGUFLegacy(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/legacy"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (Legacy)"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
|
||||
if device is not None:
|
||||
set_current_device(device)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
|
||||
'expert_mode_allocations']}
|
||||
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **clean_kwargs)
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "v1_compat")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_compat")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchGGUFLegacy
|
||||
|
||||
|
||||
def override_class_with_distorch_gguf_v2(cls):
|
||||
"""DisTorch V2 wrapper for GGUF models"""
|
||||
from . import set_current_device
|
||||
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
|
||||
|
||||
class NodeOverrideDisTorchGGUFv2(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
compute_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["compute_device"] = (devices, {"default": compute_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 128.0, "step": 0.1})
|
||||
inputs["optional"]["donor_device"] = (devices, {"default": "cpu"})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu/distorch_2"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch2)"
|
||||
|
||||
def override(self, *args, compute_device=None, virtual_vram_gb=4.0, donor_device="cpu", expert_mode_allocations="", **kwargs):
|
||||
if compute_device is not None:
|
||||
set_current_device(compute_device)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in ['compute_device', 'virtual_vram_gb',
|
||||
'donor_device', 'expert_mode_allocations']}
|
||||
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
vram_string = f"{compute_device};{virtual_vram_gb};{donor_device}"
|
||||
elif expert_mode_allocations:
|
||||
vram_string = compute_device
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
logger.info(f"[MultiGPU DisTorch V2] Full allocation string: {full_allocation}")
|
||||
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **clean_kwargs)
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "v2_gguf")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "v2_gguf")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchGGUFv2
|
||||
|
||||
|
||||
def override_class_with_distorch_clip(cls):
|
||||
"""DisTorch V1 wrapper for CLIP models - calls V2 backend"""
|
||||
from . import set_current_text_encoder_device
|
||||
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
|
||||
|
||||
class NodeOverrideDisTorchClip(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch)"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
|
||||
'expert_mode_allocations']}
|
||||
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **clean_kwargs)
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "v1_clip")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_clip")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchClip
|
||||
|
||||
|
||||
def override_class_with_distorch_clip_no_device(cls):
|
||||
"""DisTorch V1 wrapper for Triple/Quad CLIP models - calls V2 backend"""
|
||||
from . import set_current_text_encoder_device
|
||||
from .distorch_2 import register_patched_safetensor_modelpatcher, safetensor_allocation_store, create_safetensor_model_hash
|
||||
|
||||
class NodeOverrideDisTorchClipNoDevice(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
inputs["optional"]["virtual_vram_gb"] = ("FLOAT", {"default": 4.0, "min": 0.0, "max": 24.0, "step": 0.1})
|
||||
inputs["optional"]["use_other_vram"] = ("BOOLEAN", {"default": False})
|
||||
inputs["optional"]["expert_mode_allocations"] = ("STRING", {"multiline": False, "default": ""})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
TITLE = f"{cls.TITLE if hasattr(cls, 'TITLE') else cls.__name__} (DisTorch)"
|
||||
|
||||
def override(self, *args, device=None, expert_mode_allocations="", use_other_vram=False, virtual_vram_gb=0.0, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
|
||||
# Strip MultiGPU-specific parameters before calling original function
|
||||
clean_kwargs = {k: v for k, v in kwargs.items()
|
||||
if k not in ['device', 'virtual_vram_gb', 'use_other_vram',
|
||||
'expert_mode_allocations']}
|
||||
|
||||
register_patched_safetensor_modelpatcher()
|
||||
|
||||
vram_string = ""
|
||||
if virtual_vram_gb > 0:
|
||||
if use_other_vram:
|
||||
available_devices = [d for d in get_device_list() if d != "cpu"]
|
||||
other_devices = [d for d in available_devices if d != device]
|
||||
other_devices.sort(key=lambda x: int(x.split(':')[1] if ':' in x else x[-1]), reverse=False)
|
||||
device_string = ','.join(other_devices + ['cpu'])
|
||||
vram_string = f"{device};{virtual_vram_gb};{device_string}"
|
||||
else:
|
||||
vram_string = f"{device};{virtual_vram_gb};cpu"
|
||||
|
||||
full_allocation = f"{expert_mode_allocations}#{vram_string}" if expert_mode_allocations or vram_string else ""
|
||||
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **clean_kwargs)
|
||||
|
||||
if hasattr(out[0], 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0], "v1_clip_nodev")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
elif hasattr(out[0], 'patcher') and hasattr(out[0].patcher, 'model'):
|
||||
model_hash = create_safetensor_model_hash(out[0].patcher, "v1_clip_nodev")
|
||||
safetensor_allocation_store[model_hash] = full_allocation
|
||||
|
||||
return out
|
||||
|
||||
return NodeOverrideDisTorchClipNoDevice
|
||||
|
||||
|
||||
# Backward compatibility alias
|
||||
override_class_with_distorch = override_class_with_distorch_gguf
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# STANDARD MULTIGPU WRAPPERS (Device selection without DisTorch)
|
||||
# ============================================================================
|
||||
|
||||
def override_class(cls):
|
||||
"""Standard MultiGPU device override for UNet/VAE models"""
|
||||
from . import set_current_device
|
||||
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
if device is not None:
|
||||
set_current_device(device)
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
|
||||
|
||||
def override_class_clip(cls):
|
||||
"""Standard MultiGPU device override for CLIP models (with device kwarg workaround)"""
|
||||
from . import set_current_text_encoder_device
|
||||
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
kwargs['device'] = 'default'
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
|
||||
|
||||
def override_class_clip_no_device(cls):
|
||||
"""Standard MultiGPU device override for Triple/Quad CLIP models (no device kwarg workaround)"""
|
||||
from . import set_current_text_encoder_device
|
||||
|
||||
class NodeOverride(cls):
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
inputs = copy.deepcopy(cls.INPUT_TYPES())
|
||||
devices = get_device_list()
|
||||
default_device = devices[1] if len(devices) > 1 else devices[0]
|
||||
inputs["optional"] = inputs.get("optional", {})
|
||||
inputs["optional"]["device"] = (devices, {"default": default_device})
|
||||
return inputs
|
||||
|
||||
CATEGORY = "multigpu"
|
||||
FUNCTION = "override"
|
||||
|
||||
def override(self, *args, device=None, **kwargs):
|
||||
if device is not None:
|
||||
set_current_text_encoder_device(device)
|
||||
fn = getattr(super(), cls.FUNCTION)
|
||||
out = fn(*args, **kwargs)
|
||||
return out
|
||||
|
||||
return NodeOverride
|
||||
Reference in New Issue
Block a user