This commit is contained in:
smthemex
2026-08-08 14:37:57 +08:00
parent 4172da4e2e
commit 08d5ec8f4d
5 changed files with 98 additions and 17 deletions
+7
View File
@@ -0,0 +1,7 @@
__pycache__/
*.pyc
.pytest_cache/
test_outputs/
outputs/
+11 -1
View File
@@ -1,13 +1,23 @@
# ComfyUI_UniBlockSwap
A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or Bernini or other large models
A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax or Klein9B or Bernini or other large models
# Update
* Fix gguf loader cause high ram error,修复gguf加载时内存占用过大的bug,使用时注意避免推理过大分辨率或者时长过长,导致调用共享显存(如果调用了,就变慢了,不划算)
* Make it for ' low Vram and normal Ram' users to esay running ComfyUI origin workflows.(Support allmot all of comfyUI origin workflows)
* Support text encoder or diffusion models, is enable text encoder will need more Ram
# Installation
----
In the ./ComfyUI/custom_nodes directory, run the following:
```
git clone https://github.com/smthemex/ComfyUI_UniBlockSwap
```
# Example
* run minimax H3 5min 0.4 just need 4.5G Vram (要降低Te的占用需要加te模块,或者用comfyUI自带的Vbar,OOM再加TE swap,避免内存占用)
![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/example_minimax.png)
![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/minimax.png)
* run bernini int4 +loras ,512x384x120frames just need 9-10G Vram (if unpack node,notice batch size is wrong 注意官方模板解开后,batch size指向是错的,须改成1)
![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/bernini.png)
* run klein9B Q8 just need 4.8G Vram
+78 -14
View File
@@ -35,6 +35,49 @@ def _has_ggml_params(module):
return False
def _backup_ggml_refs(module):
"""Preserve the ORIGINAL mmap-backed GGMLTensor objects for every GGML
parameter in `module`.
Why a full-reference backup (not just .data): tensor_type / tensor_shape /
patches live on the GGMLTensor *object*, and a .to(...) round trip creates
a fresh tensor that loses the mmap mapping. We must keep the original object
alive so we can point the parameter back at it later.
"""
if getattr(module, "_ggml_mmap_backup", None) is not None:
return
backup = {}
for name, param in module.named_parameters(recurse=True):
t = param.data
if hasattr(t, "tensor_type"): # a GGMLTensor
backup[name] = t # keep the object alive, mmap intact
module._ggml_mmap_backup = backup
def _restore_ggml_refs(module):
"""Point params back at the original mmap GGMLTensors and drop any GPU
copies. This is a *pointer assignment* (p.data = orig), so NO anonymous
heap allocation happens -- unlike module.to(offload_device), which would
reallocate the dequantized weights as non-reclaimable RAM.
If a block was never GPU-loaded (no backup), fall back to .to(cpu) which is
a no-op for an already-mmap'd CPU tensor.
"""
backup = getattr(module, "_ggml_mmap_backup", None)
if not backup:
module.to(module.offload_device if hasattr(module, "offload_device") else "cpu")
return
params = dict(module.named_parameters(recurse=True))
for name, orig in backup.items():
p = params.get(name)
if p is not None:
p.data = orig
# free the GPU copy of the now-unreferenced tensor
if torch.cuda.is_available():
gc.collect()
torch.cuda.empty_cache()
def _free_to_meta(module):
"""Free param data to meta tensor - NO CPU copy created.
The module structure is preserved. next load() restores from backup."""
@@ -60,14 +103,19 @@ class SwappableModuleList(nn.ModuleList):
if self._loaded_swap_idx >= 0:
prev = self._loaded_swap_idx + self.non_swap_count
try:
prev_mod = self._modules[str(prev)]
# FREE previous block GPU memory
if _has_ggml_params(self._modules[str(prev)]):
# GGUF: move quantized data to CPU (preserves GGMLTensor attributes)
self._modules[str(prev)].to(self.offload_device)
if _has_ggml_params(prev_mod):
# GGUF: restore the original mmap-backed GGMLTensor by
# pointer assignment. This drops the GPU copy WITHOUT
# reallocating the weights as anonymous CPU RAM (which
# .to(offload_device) would do after a .to(cuda) round
# trip, blowing RAM from 40G to 60G).
_restore_ggml_refs(prev_mod)
else:
# Safetensor: set to meta (vbar restores automatically)
_free_to_meta(self._modules[str(prev)])
for m in self._modules[str(prev)].modules():
_free_to_meta(prev_mod)
for m in prev_mod.modules():
for attr in ('_v', '_prefetch', '_v_signature'):
if hasattr(m, attr):
try:
@@ -77,19 +125,32 @@ class SwappableModuleList(nn.ModuleList):
except Exception:
pass
# LOAD current block if GGUF
if _has_ggml_params(self._modules[str(idx)]):
self._modules[str(idx)].to(self.compute_device)
cur_mod = self._modules[str(idx)]
if _has_ggml_params(cur_mod):
# Snapshot the mmap reference so we can later restore it. We do NOT
# call cur_mod.to(compute_device) here: GGUF weights are dequantized
# per-layer on demand inside GGMLLayer.cast_bias_weight() when each
# op runs (self.weight.to(input.device)). Pre-moving the whole block
# to GPU would force a full dequantization of every layer at once,
# spiking VRAM and -- on the next swap -- a GPU->"CPU" round trip,
# both of which defeat the mmap model's whole point.
_backup_ggml_refs(cur_mod)
# else: safetensor - vbar handles restoration
self._loaded_swap_idx = local_idx
def offload_swap_blocks(self):
for i in range(self.non_swap_count, self.total_count):
try:
if _has_ggml_params(self._modules[str(i)]):
self._modules[str(i)].to(self.offload_device)
blk = self._modules[str(i)]
if _has_ggml_params(blk):
# Restore the original mmap-backed GGMLTensor (pointer
# assignment, no anonymous RAM). If a block was never
# GPU-loaded the backup is empty and the helper safely
# falls back to a no-op .to(cpu).
_restore_ggml_refs(blk)
else:
_free_to_meta(self._modules[str(i)])
for m in self._modules[str(i)].modules():
_free_to_meta(blk)
for m in blk.modules():
for attr in ('_v', '_prefetch', '_v_signature'):
if hasattr(m, attr):
try:
@@ -177,12 +238,14 @@ def install_block_swap(diffusion_model, compute_device, offload_device,
logger.info("UniBlockSwap: '%s' = %d blocks, swapping %d",
name, total, n)
# For GGUF: offload swap blocks to CPU immediately.
# For GGUF: the swap blocks already live in the mmap file-backed mapping
# on CPU. No copy is needed now; we just record the original references
# so a later offload can restore them (pointer assignment, no anon RAM).
# Safetensor blocks stay on GPU (original behavior).
for i in range(total - n, total):
blk = swl._modules[str(i)]
if _has_ggml_params(blk):
blk.to(offload_device)
_backup_ggml_refs(blk)
orig_fwd = diffusion_model.forward
@@ -270,7 +333,8 @@ def install_te_block_swap(cond_stage_model, compute_device, offload_device,
for i in range(total - n, total):
blk = swl._modules[str(i)]
if _has_ggml_params(blk):
blk.to(offload_device)
# Record mmap references; the block stays file-backed on CPU.
_backup_ggml_refs(blk)
else:
_free_to_meta(blk)
Binary file not shown.

After

Width:  |  Height:  |  Size: 605 KiB

+2 -2
View File
@@ -1,7 +1,7 @@
[project]
name = "uniblockswap"
description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or other large models"
version = "1.0.0"
description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax Klein9B or other large models"
version = "1.1.0"
license = {file = "LICENSE"}
[project.urls]