diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..20007dd --- /dev/null +++ b/.gitignore @@ -0,0 +1,7 @@ +__pycache__/ +*.pyc + +.pytest_cache/ + +test_outputs/ +outputs/ \ No newline at end of file diff --git a/README.md b/README.md index 61ace34..c4260cf 100644 --- a/README.md +++ b/README.md @@ -1,13 +1,23 @@ # ComfyUI_UniBlockSwap -A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or Bernini or other large models +A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax or Klein9B or Bernini or other large models # Update +* Fix gguf loader cause high ram error,修复gguf加载时内存占用过大的bug,使用时注意避免推理过大分辨率或者时长过长,导致调用共享显存(如果调用了,就变慢了,不划算) * Make it for ' low Vram and normal Ram' users to esay running ComfyUI origin workflows.(Support allmot all of comfyUI origin workflows) * Support text encoder or diffusion models, is enable text encoder will need more Ram +# Installation +---- + +In the ./ComfyUI/custom_nodes directory, run the following: +``` +git clone https://github.com/smthemex/ComfyUI_UniBlockSwap +``` + # Example * run minimax H3 5min 0.4 just need 4.5G Vram (要降低Te的占用需要加te模块,或者用comfyUI自带的Vbar,OOM再加TE swap,避免内存占用) ![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/example_minimax.png) +![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/minimax.png) * run bernini int4 +loras ,512x384x120frames just need 9-10G Vram (if unpack node,notice batch size is wrong 注意官方模板解开后,batch size指向是错的,须改成1) ![](https://github.com/smthemex/ComfyUI_UniBlockSwap/blob/main/example_workflows/bernini.png) * run klein9B Q8 just need 4.8G Vram diff --git a/block_swap.py b/block_swap.py index 7d561a2..31a9b09 100644 --- a/block_swap.py +++ b/block_swap.py @@ -35,6 +35,49 @@ def _has_ggml_params(module): return False +def _backup_ggml_refs(module): + """Preserve the ORIGINAL mmap-backed GGMLTensor objects for every GGML + parameter in `module`. + + Why a full-reference backup (not just .data): tensor_type / tensor_shape / + patches live on the GGMLTensor *object*, and a .to(...) round trip creates + a fresh tensor that loses the mmap mapping. We must keep the original object + alive so we can point the parameter back at it later. + """ + if getattr(module, "_ggml_mmap_backup", None) is not None: + return + backup = {} + for name, param in module.named_parameters(recurse=True): + t = param.data + if hasattr(t, "tensor_type"): # a GGMLTensor + backup[name] = t # keep the object alive, mmap intact + module._ggml_mmap_backup = backup + + +def _restore_ggml_refs(module): + """Point params back at the original mmap GGMLTensors and drop any GPU + copies. This is a *pointer assignment* (p.data = orig), so NO anonymous + heap allocation happens -- unlike module.to(offload_device), which would + reallocate the dequantized weights as non-reclaimable RAM. + + If a block was never GPU-loaded (no backup), fall back to .to(cpu) which is + a no-op for an already-mmap'd CPU tensor. + """ + backup = getattr(module, "_ggml_mmap_backup", None) + if not backup: + module.to(module.offload_device if hasattr(module, "offload_device") else "cpu") + return + params = dict(module.named_parameters(recurse=True)) + for name, orig in backup.items(): + p = params.get(name) + if p is not None: + p.data = orig + # free the GPU copy of the now-unreferenced tensor + if torch.cuda.is_available(): + gc.collect() + torch.cuda.empty_cache() + + def _free_to_meta(module): """Free param data to meta tensor - NO CPU copy created. The module structure is preserved. next load() restores from backup.""" @@ -60,14 +103,19 @@ class SwappableModuleList(nn.ModuleList): if self._loaded_swap_idx >= 0: prev = self._loaded_swap_idx + self.non_swap_count try: + prev_mod = self._modules[str(prev)] # FREE previous block GPU memory - if _has_ggml_params(self._modules[str(prev)]): - # GGUF: move quantized data to CPU (preserves GGMLTensor attributes) - self._modules[str(prev)].to(self.offload_device) + if _has_ggml_params(prev_mod): + # GGUF: restore the original mmap-backed GGMLTensor by + # pointer assignment. This drops the GPU copy WITHOUT + # reallocating the weights as anonymous CPU RAM (which + # .to(offload_device) would do after a .to(cuda) round + # trip, blowing RAM from 40G to 60G). + _restore_ggml_refs(prev_mod) else: # Safetensor: set to meta (vbar restores automatically) - _free_to_meta(self._modules[str(prev)]) - for m in self._modules[str(prev)].modules(): + _free_to_meta(prev_mod) + for m in prev_mod.modules(): for attr in ('_v', '_prefetch', '_v_signature'): if hasattr(m, attr): try: @@ -77,19 +125,32 @@ class SwappableModuleList(nn.ModuleList): except Exception: pass # LOAD current block if GGUF - if _has_ggml_params(self._modules[str(idx)]): - self._modules[str(idx)].to(self.compute_device) + cur_mod = self._modules[str(idx)] + if _has_ggml_params(cur_mod): + # Snapshot the mmap reference so we can later restore it. We do NOT + # call cur_mod.to(compute_device) here: GGUF weights are dequantized + # per-layer on demand inside GGMLLayer.cast_bias_weight() when each + # op runs (self.weight.to(input.device)). Pre-moving the whole block + # to GPU would force a full dequantization of every layer at once, + # spiking VRAM and -- on the next swap -- a GPU->"CPU" round trip, + # both of which defeat the mmap model's whole point. + _backup_ggml_refs(cur_mod) # else: safetensor - vbar handles restoration self._loaded_swap_idx = local_idx def offload_swap_blocks(self): for i in range(self.non_swap_count, self.total_count): try: - if _has_ggml_params(self._modules[str(i)]): - self._modules[str(i)].to(self.offload_device) + blk = self._modules[str(i)] + if _has_ggml_params(blk): + # Restore the original mmap-backed GGMLTensor (pointer + # assignment, no anonymous RAM). If a block was never + # GPU-loaded the backup is empty and the helper safely + # falls back to a no-op .to(cpu). + _restore_ggml_refs(blk) else: - _free_to_meta(self._modules[str(i)]) - for m in self._modules[str(i)].modules(): + _free_to_meta(blk) + for m in blk.modules(): for attr in ('_v', '_prefetch', '_v_signature'): if hasattr(m, attr): try: @@ -177,12 +238,14 @@ def install_block_swap(diffusion_model, compute_device, offload_device, logger.info("UniBlockSwap: '%s' = %d blocks, swapping %d", name, total, n) - # For GGUF: offload swap blocks to CPU immediately. + # For GGUF: the swap blocks already live in the mmap file-backed mapping + # on CPU. No copy is needed now; we just record the original references + # so a later offload can restore them (pointer assignment, no anon RAM). # Safetensor blocks stay on GPU (original behavior). for i in range(total - n, total): blk = swl._modules[str(i)] if _has_ggml_params(blk): - blk.to(offload_device) + _backup_ggml_refs(blk) orig_fwd = diffusion_model.forward @@ -270,7 +333,8 @@ def install_te_block_swap(cond_stage_model, compute_device, offload_device, for i in range(total - n, total): blk = swl._modules[str(i)] if _has_ggml_params(blk): - blk.to(offload_device) + # Record mmap references; the block stays file-backed on CPU. + _backup_ggml_refs(blk) else: _free_to_meta(blk) diff --git a/example_workflows/minimax.png b/example_workflows/minimax.png new file mode 100644 index 0000000..73a501c Binary files /dev/null and b/example_workflows/minimax.png differ diff --git a/pyproject.toml b/pyproject.toml index 8ef15a7..e38307d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "uniblockswap" -description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Klein9B or other large models" -version = "1.0.0" +description = "A universal swap node that supports ComfyUI native workflow, allowing 4_6G users to experience Minimax Klein9B or other large models" +version = "1.1.0" license = {file = "LICENSE"} [project.urls]