init
This commit is contained in:
@@ -0,0 +1,7 @@
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
.pytest_cache/
|
||||
|
||||
test_outputs/
|
||||
outputs/
|
||||
@@ -1,13 +1,23 @@
|
||||
# ComfyUI_UniBlockSwap
|
||||
A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or Bernini or other large models
|
||||
A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax or Klein9B or Bernini or other large models
|
||||
|
||||
# Update
|
||||
* Fix gguf loader cause high ram error,修复gguf加载时内存占用过大的bug,使用时注意避免推理过大分辨率或者时长过长,导致调用共享显存(如果调用了,就变慢了,不划算)
|
||||
* Make it for ' low Vram and normal Ram' users to esay running ComfyUI origin workflows.(Support allmot all of comfyUI origin workflows)
|
||||
* Support text encoder or diffusion models, is enable text encoder will need more Ram
|
||||
|
||||
# Installation
|
||||
----
|
||||
|
||||
In the ./ComfyUI/custom_nodes directory, run the following:
|
||||
```
|
||||
git clone https://github.com/smthemex/ComfyUI_UniBlockSwap
|
||||
```
|
||||
|
||||
# Example
|
||||
* run minimax H3 5min 0.4 just need 4.5G Vram (要降低Te的占用需要加te模块,或者用comfyUI自带的Vbar,OOM再加TE swap,避免内存占用)
|
||||

|
||||

|
||||
* run bernini int4 +loras ,512x384x120frames just need 9-10G Vram (if unpack node,notice batch size is wrong 注意官方模板解开后,batch size指向是错的,须改成1)
|
||||

|
||||
* run klein9B Q8 just need 4.8G Vram
|
||||
|
||||
+78
-14
@@ -35,6 +35,49 @@ def _has_ggml_params(module):
|
||||
return False
|
||||
|
||||
|
||||
def _backup_ggml_refs(module):
|
||||
"""Preserve the ORIGINAL mmap-backed GGMLTensor objects for every GGML
|
||||
parameter in `module`.
|
||||
|
||||
Why a full-reference backup (not just .data): tensor_type / tensor_shape /
|
||||
patches live on the GGMLTensor *object*, and a .to(...) round trip creates
|
||||
a fresh tensor that loses the mmap mapping. We must keep the original object
|
||||
alive so we can point the parameter back at it later.
|
||||
"""
|
||||
if getattr(module, "_ggml_mmap_backup", None) is not None:
|
||||
return
|
||||
backup = {}
|
||||
for name, param in module.named_parameters(recurse=True):
|
||||
t = param.data
|
||||
if hasattr(t, "tensor_type"): # a GGMLTensor
|
||||
backup[name] = t # keep the object alive, mmap intact
|
||||
module._ggml_mmap_backup = backup
|
||||
|
||||
|
||||
def _restore_ggml_refs(module):
|
||||
"""Point params back at the original mmap GGMLTensors and drop any GPU
|
||||
copies. This is a *pointer assignment* (p.data = orig), so NO anonymous
|
||||
heap allocation happens -- unlike module.to(offload_device), which would
|
||||
reallocate the dequantized weights as non-reclaimable RAM.
|
||||
|
||||
If a block was never GPU-loaded (no backup), fall back to .to(cpu) which is
|
||||
a no-op for an already-mmap'd CPU tensor.
|
||||
"""
|
||||
backup = getattr(module, "_ggml_mmap_backup", None)
|
||||
if not backup:
|
||||
module.to(module.offload_device if hasattr(module, "offload_device") else "cpu")
|
||||
return
|
||||
params = dict(module.named_parameters(recurse=True))
|
||||
for name, orig in backup.items():
|
||||
p = params.get(name)
|
||||
if p is not None:
|
||||
p.data = orig
|
||||
# free the GPU copy of the now-unreferenced tensor
|
||||
if torch.cuda.is_available():
|
||||
gc.collect()
|
||||
torch.cuda.empty_cache()
|
||||
|
||||
|
||||
def _free_to_meta(module):
|
||||
"""Free param data to meta tensor - NO CPU copy created.
|
||||
The module structure is preserved. next load() restores from backup."""
|
||||
@@ -60,14 +103,19 @@ class SwappableModuleList(nn.ModuleList):
|
||||
if self._loaded_swap_idx >= 0:
|
||||
prev = self._loaded_swap_idx + self.non_swap_count
|
||||
try:
|
||||
prev_mod = self._modules[str(prev)]
|
||||
# FREE previous block GPU memory
|
||||
if _has_ggml_params(self._modules[str(prev)]):
|
||||
# GGUF: move quantized data to CPU (preserves GGMLTensor attributes)
|
||||
self._modules[str(prev)].to(self.offload_device)
|
||||
if _has_ggml_params(prev_mod):
|
||||
# GGUF: restore the original mmap-backed GGMLTensor by
|
||||
# pointer assignment. This drops the GPU copy WITHOUT
|
||||
# reallocating the weights as anonymous CPU RAM (which
|
||||
# .to(offload_device) would do after a .to(cuda) round
|
||||
# trip, blowing RAM from 40G to 60G).
|
||||
_restore_ggml_refs(prev_mod)
|
||||
else:
|
||||
# Safetensor: set to meta (vbar restores automatically)
|
||||
_free_to_meta(self._modules[str(prev)])
|
||||
for m in self._modules[str(prev)].modules():
|
||||
_free_to_meta(prev_mod)
|
||||
for m in prev_mod.modules():
|
||||
for attr in ('_v', '_prefetch', '_v_signature'):
|
||||
if hasattr(m, attr):
|
||||
try:
|
||||
@@ -77,19 +125,32 @@ class SwappableModuleList(nn.ModuleList):
|
||||
except Exception:
|
||||
pass
|
||||
# LOAD current block if GGUF
|
||||
if _has_ggml_params(self._modules[str(idx)]):
|
||||
self._modules[str(idx)].to(self.compute_device)
|
||||
cur_mod = self._modules[str(idx)]
|
||||
if _has_ggml_params(cur_mod):
|
||||
# Snapshot the mmap reference so we can later restore it. We do NOT
|
||||
# call cur_mod.to(compute_device) here: GGUF weights are dequantized
|
||||
# per-layer on demand inside GGMLLayer.cast_bias_weight() when each
|
||||
# op runs (self.weight.to(input.device)). Pre-moving the whole block
|
||||
# to GPU would force a full dequantization of every layer at once,
|
||||
# spiking VRAM and -- on the next swap -- a GPU->"CPU" round trip,
|
||||
# both of which defeat the mmap model's whole point.
|
||||
_backup_ggml_refs(cur_mod)
|
||||
# else: safetensor - vbar handles restoration
|
||||
self._loaded_swap_idx = local_idx
|
||||
|
||||
def offload_swap_blocks(self):
|
||||
for i in range(self.non_swap_count, self.total_count):
|
||||
try:
|
||||
if _has_ggml_params(self._modules[str(i)]):
|
||||
self._modules[str(i)].to(self.offload_device)
|
||||
blk = self._modules[str(i)]
|
||||
if _has_ggml_params(blk):
|
||||
# Restore the original mmap-backed GGMLTensor (pointer
|
||||
# assignment, no anonymous RAM). If a block was never
|
||||
# GPU-loaded the backup is empty and the helper safely
|
||||
# falls back to a no-op .to(cpu).
|
||||
_restore_ggml_refs(blk)
|
||||
else:
|
||||
_free_to_meta(self._modules[str(i)])
|
||||
for m in self._modules[str(i)].modules():
|
||||
_free_to_meta(blk)
|
||||
for m in blk.modules():
|
||||
for attr in ('_v', '_prefetch', '_v_signature'):
|
||||
if hasattr(m, attr):
|
||||
try:
|
||||
@@ -177,12 +238,14 @@ def install_block_swap(diffusion_model, compute_device, offload_device,
|
||||
logger.info("UniBlockSwap: '%s' = %d blocks, swapping %d",
|
||||
name, total, n)
|
||||
|
||||
# For GGUF: offload swap blocks to CPU immediately.
|
||||
# For GGUF: the swap blocks already live in the mmap file-backed mapping
|
||||
# on CPU. No copy is needed now; we just record the original references
|
||||
# so a later offload can restore them (pointer assignment, no anon RAM).
|
||||
# Safetensor blocks stay on GPU (original behavior).
|
||||
for i in range(total - n, total):
|
||||
blk = swl._modules[str(i)]
|
||||
if _has_ggml_params(blk):
|
||||
blk.to(offload_device)
|
||||
_backup_ggml_refs(blk)
|
||||
|
||||
orig_fwd = diffusion_model.forward
|
||||
|
||||
@@ -270,7 +333,8 @@ def install_te_block_swap(cond_stage_model, compute_device, offload_device,
|
||||
for i in range(total - n, total):
|
||||
blk = swl._modules[str(i)]
|
||||
if _has_ggml_params(blk):
|
||||
blk.to(offload_device)
|
||||
# Record mmap references; the block stays file-backed on CPU.
|
||||
_backup_ggml_refs(blk)
|
||||
else:
|
||||
_free_to_meta(blk)
|
||||
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 605 KiB |
+2
-2
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "uniblockswap"
|
||||
description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or other large models"
|
||||
version = "1.0.0"
|
||||
description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax Klein9B or other large models"
|
||||
version = "1.1.0"
|
||||
license = {file = "LICENSE"}
|
||||
|
||||
[project.urls]
|
||||
|
||||
Reference in New Issue
Block a user