Eliminate unused safetensor loading analysis method and update example configurations, adding one with LoRAs as one of the tested configurations to avoid the issue seen during initial release.

This commit is contained in:
John Pollock
2025-08-24 04:29:32 -05:00
parent e28b040cda
commit 4367c892d8
5 changed files with 855 additions and 147 deletions
-147
View File
@@ -377,152 +377,6 @@ def analyze_safetensor_loading(model_patcher, allocations_str):
"device_assignments": device_assignments,
"block_assignments": block_assignments
}
def analyze_safetensor_loading_comfy(model_patcher, allocations_str):
"""
Analyze and distribute safetensor model blocks across devices utilizing model_patcher._load_list().sort(reverse=True) method like Comfy
"""
DEVICE_RATIOS_DISTORCH = {}
device_table = {}
distorch_alloc = allocations_str
virtual_vram_gb = 0.0
# Parse allocation string EXACTLY like GGML
if '#' in allocations_str:
distorch_alloc, virtual_vram_str = allocations_str.split('#')
if not distorch_alloc:
distorch_alloc = calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str)
# EXACT SAME FORMATTING AS GGML
eq_line = "=" * 50
dash_line = "-" * 50
fmt_assign = "{:<18}{:>7}{:>14}{:>10}"
# Parse device allocations
for allocation in distorch_alloc.split(';'):
if ',' not in allocation:
continue
dev_name, fraction = allocation.split(',')
fraction = float(fraction)
total_mem_bytes = mm.get_total_memory(torch.device(dev_name))
alloc_gb = (total_mem_bytes * fraction) / (1024**3)
DEVICE_RATIOS_DISTORCH[dev_name] = alloc_gb
device_table[dev_name] = {
"fraction": fraction,
"total_gb": total_mem_bytes / (1024**3),
"alloc_gb": alloc_gb
}
# IDENTICAL LOGGING TO DISTORCH
logger.info(eq_line)
logger.info(" DisTorch2 Model Device Allocations")
logger.info(eq_line)
logger.info(fmt_assign.format("Device", "Alloc %", "Total (GB)", " Alloc (GB)"))
logger.info(dash_line)
sorted_devices = sorted(device_table.keys(), key=lambda d: (d == "cpu", d))
for dev in sorted_devices:
frac = device_table[dev]["fraction"]
tot_gb = device_table[dev]["total_gb"]
alloc_gb = device_table[dev]["alloc_gb"]
logger.info(fmt_assign.format(dev,f"{int(frac * 100)}%",f"{tot_gb:.2f}",f"{alloc_gb:.2f}"))
logger.info(dash_line)
# Get the model blocks using ComfyUI's method
block_list = model_patcher._load_list()
block_list.sort(reverse=True)
# Log layer distribution
total_memory = sum(b[0] for b in block_list)
memory_by_type = defaultdict(int)
block_summary = defaultdict(int)
for module_size, module_name, module_object, params in block_list:
block_type = module_object.__class__.__name__
block_summary[block_type] += 1
memory_by_type[block_type] += module_size
# Log layer distribution - IDENTICAL FORMAT TO GGML
logger.info(" DisTorch2 Model Layer Distribution")
logger.info(dash_line)
fmt_layer = "{:<18}{:>7}{:>14}{:>10}"
logger.info(fmt_layer.format("Layer Type", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
for layer_type, count in block_summary.items():
mem_mb = memory_by_type[layer_type] / (1024 * 1024)
mem_percent = (memory_by_type[layer_type] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_layer.format(layer_type[:18], str(count), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
logger.info(dash_line)
# Distribute blocks sequentially
device_assignments = {device: [] for device in DEVICE_RATIOS_DISTORCH.keys()}
block_assignments = {}
compute_device = str(current_device)
# Calculate total memory to be offloaded to donor devices
total_offload_gb = sum(DEVICE_RATIOS_DISTORCH.get(d, 0) for d in sorted_devices if d != compute_device)
total_offload_bytes = total_offload_gb * (1024**3)
offloaded_bytes = 0
# Iterate through the sorted list (largest blocks first)
for module_size, module_name, module_object, params in block_list:
# Assign to donor device until target is met
if offloaded_bytes < total_offload_bytes:
# For now, simple offload to CPU, will expand for multi-donor
donor_device = "cpu"
for dev in sorted_devices:
if dev != compute_device:
donor_device = dev
break # Use first available donor
block_assignments[module_name] = donor_device
setattr(module_object, 'distorch2_cpu_offload', True) # Attach the attribute here
offloaded_bytes += module_size
else:
# Assign remaining blocks to the primary compute device
block_assignments[module_name] = compute_device
# Populate device_assignments from the final block_assignments
for module_size, module_name, module_object, params in block_list:
device = block_assignments[module_name]
if device not in device_assignments:
device_assignments[device] = []
device_assignments[device].append((module_name, module_object, module_object.__class__.__name__, module_size))
# Log final assignments - IDENTICAL FORMAT TO GGML
logger.info("DisTorch2 Model Final Device/Layer Assignments")
logger.info(dash_line)
logger.info(fmt_assign.format("Device", "Layers", "Memory (MB)", "% Total"))
logger.info(dash_line)
# Log distributed blocks
total_assigned_memory = 0
device_memories = {}
for device, blocks in device_assignments.items():
device_memory = sum(b[3] for b in blocks)
device_memories[device] = device_memory
total_assigned_memory += device_memory
sorted_assignments = sorted(device_memories.keys(), key=lambda d: (d == "cpu", d))
for dev in sorted_assignments:
if dev not in device_memories:
continue
mem_mb = device_memories[dev] / (1024 * 1024)
mem_percent = (device_memories[dev] / total_memory) * 100 if total_memory > 0 else 0
logger.info(fmt_assign.format(dev, str(len(device_assignments[dev])), f"{mem_mb:.2f}", f"{mem_percent:.1f}%"))
logger.info(dash_line)
return {
"device_assignments": device_assignments,
"block_assignments": block_assignments,
"lowvram_model_memory": total_assigned_memory,
}
def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
"""Calculate virtual VRAM allocation string for distributed safetensor loading"""
@@ -609,7 +463,6 @@ def calculate_safetensor_vvram_allocation(model_patcher, virtual_vram_str):
return allocation_string
def override_class_with_distorch_safetensor_v2(cls):
"""DisTorch 2.0 wrapper for safetensor models"""
from .nodes import get_device_list
+855
View File
@@ -0,0 +1,855 @@
{
"id": "25dfce61-3571-4c9a-a7f8-6a83951d66aa",
"revision": 0,
"last_node_id": 52,
"last_link_id": 97,
"nodes": [
{
"id": 10,
"type": "SaveImage",
"pos": [
4750,
-350
],
"size": [
2000,
1700
],
"flags": {},
"order": 14,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 8
}
],
"outputs": [],
"properties": {
"Node name for S&R": "SaveImage",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
"ComfyUI"
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 5,
"type": "EmptyHunyuanLatentVideo",
"pos": [
3650,
300
],
"size": [
450,
150
],
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
60
]
}
],
"properties": {
"Node name for S&R": "EmptyHunyuanLatentVideo",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
1536,
1536,
1,
1
],
"color": "#323",
"bgcolor": "#535"
},
{
"id": 4,
"type": "CLIPTextEncode",
"pos": [
3650,
100
],
"size": [
450,
150
],
"flags": {
"collapsed": false
},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 89
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
59,
64
]
}
],
"title": "Negative Prompt",
"properties": {
"Node name for S&R": "CLIPTextEncode",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
""
],
"color": "#2a363b",
"bgcolor": "#3f5159"
},
{
"id": 44,
"type": "LoraLoaderModelOnly",
"pos": [
3200,
100
],
"size": [
400,
150
],
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 78
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
91
]
}
],
"properties": {
"Node name for S&R": "LoraLoaderModelOnly",
"cnr_id": "comfy-core",
"ver": "0.3.46"
},
"widgets_values": [
"Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors",
0.6000000000000001
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 43,
"type": "LoraLoaderModelOnly",
"pos": [
3182.24267578125,
-103.5514907836914
],
"size": [
400,
150
],
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 97
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
78
]
}
],
"properties": {
"Node name for S&R": "LoraLoaderModelOnly",
"cnr_id": "comfy-core",
"ver": "0.3.46"
},
"widgets_values": [
"Wan2.1_T2V_14B_FusionX_LoRA.safetensors",
0.8000000000000002
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 8,
"type": "VAELoader",
"pos": [
4450,
-50
],
"size": [
250,
100
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "VAE",
"type": "VAE",
"links": [
9
]
}
],
"properties": {
"Node name for S&R": "VAELoader",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
"wan_2.1_vae.safetensors"
],
"color": "#323",
"bgcolor": "#535"
},
{
"id": 36,
"type": "KSamplerAdvanced",
"pos": [
4150,
-100
],
"size": [
250,
546
],
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 91
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 63
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 64
},
{
"name": "latent_image",
"type": "LATENT",
"link": 61
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
65
]
}
],
"properties": {
"Node name for S&R": "KSamplerAdvanced",
"cnr_id": "comfy-core",
"ver": "0.3.46"
},
"widgets_values": [
"enable",
575669122636822,
"randomize",
8,
1,
"euler",
"simple",
4,
8,
"disable"
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 35,
"type": "KSamplerAdvanced",
"pos": [
4152.58056640625,
-684.031005859375
],
"size": [
250,
546
],
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 88
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 58
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 59
},
{
"name": "latent_image",
"type": "LATENT",
"link": 60
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
61
]
}
],
"properties": {
"Node name for S&R": "KSamplerAdvanced",
"cnr_id": "comfy-core",
"ver": "0.3.46"
},
"widgets_values": [
"enable",
202136745531075,
"randomize",
8,
1,
"euler",
"simple",
0,
4,
"disable"
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 29,
"type": "LoraLoader",
"pos": [
3190.54638671875,
-428.14935302734375
],
"size": [
400,
150
],
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 39
},
{
"name": "clip",
"type": "CLIP",
"link": 40
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
88
]
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
89,
90
]
}
],
"properties": {
"Node name for S&R": "LoraLoader",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
"Wan21_T2V_14B_lightx2v_cfg_step_distill_lora_rank32.safetensors",
0.6000000000000001,
0.6000000000000001
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 3,
"type": "CLIPTextEncode",
"pos": [
3650,
-300
],
"size": [
450,
350
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 90
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
58,
63
]
}
],
"title": "Positive Prompt",
"properties": {
"Node name for S&R": "CLIPTextEncode",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
"A towering technological monolith in a cyberpunk cityscape at night, with \"DisTorch 2\" emblazoned across its surface in massive neon blue-green mixed with purple letters that illuminate the surrounding buildings. The text occupies the central third of the frame, crafted from glowing plasma tubes and crackling energy. \"now with .safetensors!\" in same styling but smaller, italic version of the font splashed in the lower right-hand corner of the image. Rain-slicked streets below reflect the brilliant signage, while holographic advertisements and flying vehicles populate the background. Moody atmospheric lighting, heavy contrast, photorealistic textures, cinematic color grading. "
],
"color": "#2a363b",
"bgcolor": "#3f5159"
},
{
"id": 9,
"type": "VAEDecode",
"pos": [
4450,
112.2901840209961
],
"size": [
250,
150
],
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 65
},
{
"name": "vae",
"type": "VAE",
"link": 9
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
8
]
}
],
"properties": {
"Node name for S&R": "VAEDecode",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [],
"color": "#323",
"bgcolor": "#535"
},
{
"id": 51,
"type": "CLIPLoaderMultiGPU",
"pos": [
2706.568603515625,
-368.65240478515625
],
"size": [
380.6119384765625,
106
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "CLIP",
"type": "CLIP",
"links": [
96
]
}
],
"properties": {
"Node name for S&R": "CLIPLoaderMultiGPU"
},
"widgets_values": [
"umt5_xxl_fp8_e4m3fn_scaled.safetensors",
"wan",
"cuda:1"
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 50,
"type": "UNETLoaderDisTorch2MultiGPU",
"pos": [
2650.002685546875,
-685.0429077148438
],
"size": [
440.8240051269531,
202
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
93
]
}
],
"properties": {
"Node name for S&R": "UNETLoaderDisTorch2MultiGPU"
},
"widgets_values": [
"wan2.2_t2v_high_noise_14B_fp8_scaled.safetensors",
"default",
"cuda:0",
4,
"cpu",
"",
true
],
"color": "#233",
"bgcolor": "#355"
},
{
"id": 30,
"type": "LoraLoader",
"pos": [
3182.037353515625,
-675.4197387695312
],
"size": [
400,
150
],
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 93
},
{
"name": "clip",
"type": "CLIP",
"link": 96
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
39
]
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
40
]
}
],
"properties": {
"Node name for S&R": "LoraLoader",
"cnr_id": "comfy-core",
"ver": "0.3.43"
},
"widgets_values": [
"Wan2.1_T2V_14B_FusionX_LoRA.safetensors",
0.8000000000000002,
0.8000000000000002
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 52,
"type": "UNETLoaderDisTorch2MultiGPU",
"pos": [
2666.862548828125,
-89.75868225097656
],
"size": [
401.1171569824219,
202
],
"flags": {},
"order": 4,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
97
]
}
],
"properties": {
"Node name for S&R": "UNETLoaderDisTorch2MultiGPU"
},
"widgets_values": [
"wan2.2_t2v_low_noise_14B_fp8_scaled.safetensors",
"default",
"cuda:1",
4,
"cpu",
"",
true
],
"color": "#233",
"bgcolor": "#355"
}
],
"links": [
[
8,
9,
0,
10,
0,
"IMAGE"
],
[
9,
8,
0,
9,
1,
"VAE"
],
[
39,
30,
0,
29,
0,
"MODEL"
],
[
40,
30,
1,
29,
1,
"CLIP"
],
[
58,
3,
0,
35,
1,
"CONDITIONING"
],
[
59,
4,
0,
35,
2,
"CONDITIONING"
],
[
60,
5,
0,
35,
3,
"LATENT"
],
[
61,
35,
0,
36,
3,
"LATENT"
],
[
63,
3,
0,
36,
1,
"CONDITIONING"
],
[
64,
4,
0,
36,
2,
"CONDITIONING"
],
[
65,
36,
0,
9,
0,
"LATENT"
],
[
78,
43,
0,
44,
0,
"MODEL"
],
[
88,
29,
0,
35,
0,
"MODEL"
],
[
89,
29,
1,
4,
0,
"CLIP"
],
[
90,
29,
1,
3,
0,
"CLIP"
],
[
91,
44,
0,
36,
0,
"MODEL"
],
[
93,
50,
0,
30,
0,
"MODEL"
],
[
96,
51,
0,
30,
1,
"CLIP"
],
[
97,
52,
0,
43,
0,
"MODEL"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.7051680186913379,
"offset": [
-2597.5068415226533,
881.6331759197561
]
},
"frontendVersion": "1.25.9",
"VHS_latentpreview": false,
"VHS_latentpreviewrate": 0,
"VHS_MetadataImage": true,
"VHS_KeepIntermediate": true
},
"version": 0.4
}