preparing DisTorch for release to :main:
This commit is contained in:
@@ -6,6 +6,16 @@ This extension adds device selection capabilities to model loading nodes in Comf
|
||||
|
||||
*Note: This does not add parallelism. The workflow steps are still executed sequentially just with model components loaded on different GPUs or offloaded to the CPU where allowed. Any potential speedup comes from not having to constantly load and unload models from VRAM.*
|
||||
|
||||
# NEW: DisTorch - Advanced GGUF-Quantized Model Layer Distribution
|
||||
|
||||
DisTorch nodes are now available, allowing fine-grained control over model layer distribution across multiple devices for GGUF quantized models. Using a simple allocation string (e.g., "cuda:0,0.025;cuda:1,0.05;cpu,0.10"), you can precisely specify how much memory each device should contribute to hosting model layers. This enables sophisticated memory management strategies like:
|
||||
|
||||
- Splitting large models across multiple GPUs with different VRAM capacities
|
||||
- Utilizing CPU memory alongside GPU VRAM for handling memory-intensive models
|
||||
- Optimizing layer placement based on your specific hardware configuration
|
||||
|
||||
Check out the updated examples `hunyuan_gguf_distorch.json` and `flux1dev_gguf_distorch.json` to see DisTorch in action, demonstrating advanced layer distribution across multiple devices.
|
||||
|
||||
## Installation
|
||||
|
||||
Installation via [ComfyUI-Manager](https://github.com/ltdrdata/ComfyUI-Manager) is preferred. Simply search for `ComfyUI-MultiGPU` in the list of nodes and follow installation instructions.
|
||||
@@ -57,6 +67,14 @@ All MultiGPU nodes available for your install can be found in the "multigpu" cat
|
||||
|
||||
All workflows have been tested on a 2x 3090 linux setup, a 4070 win 11 setup, and a 3090/1070ti linux setup.
|
||||
|
||||
### Split GGUF-quantized UNet and CLIP models across multiple devices using DisTorch
|
||||
|
||||
- [examples/hunyuan_gguf_distorch.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/hunyuan_gguf_distorch.json)
|
||||
This workflow attaches a HunyuanVideo GGUF-quantized model on `cuda:0` for compute and distrubutes its UNet across itself, a secondary CUDA device, and the system's main memory (`cpu`) using a new DisTorch distributed-load methodology. The text encoder now attaches itself to `cuda:1` and splits iteself between `cuda:1` amd `cpu` layers. While the VAE is loaded on GPU 1 directly and use `cuda:1` for compute.
|
||||
|
||||
- [examples/flux1dev_gguf_distorch.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/flux1dev_gguf_distorch.json)
|
||||
This workflow loads a FLUX.1-dev model on `cuda:0` for compute and distrubutes its UNet across multiple CUDA devices using new DisTorch distributed-load methodology. While the text encoders and VAE are loaded on GPU 1 and use `cuda:1` for compute.
|
||||
|
||||
### Split Hunyuan Video UNet across two devices and use DiffSynth Just-in-Time loading
|
||||
|
||||
- [examples/hunyuanvideowrapper_diffsynth.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/hunyuanvideowrapper_diffsynth.json)
|
||||
|
||||
@@ -55,7 +55,7 @@
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"order": 9,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
@@ -209,7 +209,7 @@
|
||||
109.8011474609375
|
||||
],
|
||||
"flags": {},
|
||||
"order": 9,
|
||||
"order": 10,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
@@ -401,81 +401,6 @@
|
||||
"color": "#2a363b",
|
||||
"bgcolor": "#3f5159"
|
||||
},
|
||||
{
|
||||
"id": 88,
|
||||
"type": "UnetLoaderGGUFDisTorchMultiGPU",
|
||||
"pos": [
|
||||
2622.552734375,
|
||||
-105.02156829833984
|
||||
],
|
||||
"size": [
|
||||
358.005859375,
|
||||
128.5614013671875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"links": [
|
||||
141,
|
||||
142
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "UnetLoaderGGUFDisTorchMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"hunyuan-video-t2v-720p-Q3_K_S.gguf",
|
||||
"cuda:0",
|
||||
"cuda:0,0.025;cuda:1,0.05;cpu,0.10"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 100,
|
||||
"type": "DualCLIPLoaderGGUFDisTorchMultiGPU",
|
||||
"pos": [
|
||||
2623.7119140625,
|
||||
64.39619445800781
|
||||
],
|
||||
"size": [
|
||||
348.5331726074219,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "CLIP",
|
||||
"type": "CLIP",
|
||||
"links": [
|
||||
154
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DualCLIPLoaderGGUFDisTorchMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"clip_l.safetensors",
|
||||
"llava-llama-3-8B-v1_1-Q4_K_M.gguf",
|
||||
"hunyuan_video",
|
||||
"cuda:1",
|
||||
"cuda:0,0.025;cuda:1,0.05;cpu,0.10"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 70,
|
||||
"type": "VHS_VideoCombine",
|
||||
@@ -485,7 +410,7 @@
|
||||
],
|
||||
"size": [
|
||||
338.94024658203125,
|
||||
903.6536865234375
|
||||
902.6787109375
|
||||
],
|
||||
"flags": {},
|
||||
"order": 15,
|
||||
@@ -540,13 +465,13 @@
|
||||
"hidden": false,
|
||||
"paused": false,
|
||||
"params": {
|
||||
"filename": "HunyuanVideo_00314.mp4",
|
||||
"filename": "HunyuanVideo_00315.mp4",
|
||||
"subfolder": "",
|
||||
"type": "output",
|
||||
"format": "video/h264-mp4",
|
||||
"frame_rate": 24,
|
||||
"workflow": "HunyuanVideo_00314.png",
|
||||
"fullpath": "/home/johnj/ComfyUI/output/HunyuanVideo_00314.mp4"
|
||||
"workflow": "HunyuanVideo_00315.png",
|
||||
"fullpath": "/home/johnj/ComfyUI/output/HunyuanVideo_00315.mp4"
|
||||
},
|
||||
"muted": false
|
||||
}
|
||||
@@ -610,7 +535,7 @@
|
||||
528.065185546875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
@@ -634,7 +559,7 @@
|
||||
164.31304931640625
|
||||
],
|
||||
"flags": {},
|
||||
"order": 10,
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
@@ -662,6 +587,81 @@
|
||||
],
|
||||
"color": "#232",
|
||||
"bgcolor": "#353"
|
||||
},
|
||||
{
|
||||
"id": 100,
|
||||
"type": "DualCLIPLoaderGGUFDisTorchMultiGPU",
|
||||
"pos": [
|
||||
2623.7119140625,
|
||||
64.39619445800781
|
||||
],
|
||||
"size": [
|
||||
348.5331726074219,
|
||||
154
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "CLIP",
|
||||
"type": "CLIP",
|
||||
"links": [
|
||||
154
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "DualCLIPLoaderGGUFDisTorchMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"clip_l.safetensors",
|
||||
"llava-llama-3-8B-v1_1-Q4_K_M.gguf",
|
||||
"hunyuan_video",
|
||||
"cuda:1",
|
||||
"cuda:1,0.33;cpu,0.15"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
},
|
||||
{
|
||||
"id": 88,
|
||||
"type": "UnetLoaderGGUFDisTorchMultiGPU",
|
||||
"pos": [
|
||||
2622.552734375,
|
||||
-105.02156829833984
|
||||
],
|
||||
"size": [
|
||||
358.005859375,
|
||||
128.5614013671875
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"links": [
|
||||
141,
|
||||
142
|
||||
],
|
||||
"slot_index": 0
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "UnetLoaderGGUFDisTorchMultiGPU"
|
||||
},
|
||||
"widgets_values": [
|
||||
"flux1-dev-Q8_0.gguf",
|
||||
"cuda:0",
|
||||
"cuda:0,0.33;cuda:1,0.33;cpu,0.0825"
|
||||
],
|
||||
"color": "#233",
|
||||
"bgcolor": "#355"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
@@ -796,10 +796,10 @@
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.797202450000122,
|
||||
"scale": 1.283902517749697,
|
||||
"offset": [
|
||||
-1830.6457542655762,
|
||||
790.7863966854694
|
||||
-2107.057728768135,
|
||||
344.093734201536
|
||||
]
|
||||
},
|
||||
"ue_links": [],
|
||||
|
||||
+1
-1
@@ -1,7 +1,7 @@
|
||||
[project]
|
||||
name = "comfyui-multigpu"
|
||||
description = "This extension adds CUDA/CPU device selection to supported loader nodes in ComfyUI. By monkey-patching ComfyUI’s memory management, each model component (like UNet, Clip, or VAE) can be loaded on a specific GPU. Examples included are multi-GPU workflows for SDXL, FLUX, LTXVideo, and Hunyuan Video for both standard and GGUF loader nodes."
|
||||
version = "1.3.1"
|
||||
version = "1.4.0"
|
||||
license = {file = "LICENSE"}
|
||||
|
||||
[project.urls]
|
||||
|
||||
Reference in New Issue
Block a user