From e73c8bd5e3eaee168c3e46d73a6eeef8d4160239 Mon Sep 17 00:00:00 2001 From: John Pollock Date: Wed, 15 Jan 2025 06:53:59 -0600 Subject: [PATCH] Add experimental DiffSynth block-swapping support via new GPU offload device - Adds HyVideoModelLoaderDiffSynthMultiGPU node implementing DiffSynth block-swapping - Introduces offload_device selection for secondary GPU utilization - Updates documentation with known behaviors and expected OOM patterns - Adds example workflow demonstrating higher resolution/longer duration video generation - Maintains backwards compatibility with existing MultiGPU workflows --- README.md | 13 +- __init__.py | 96 +++- examples/hunyuanvideowrapper_diffsynth.json | 575 ++++++++++++++++++++ 3 files changed, 672 insertions(+), 12 deletions(-) create mode 100644 examples/hunyuanvideowrapper_diffsynth.json diff --git a/README.md b/README.md index ed4e21f..bf57817 100644 --- a/README.md +++ b/README.md @@ -19,6 +19,7 @@ Clone [this repository](https://github.com/pollockjj/ComfyUI-MultiGPU) inside `C The extension automatically creates MultiGPU versions of loader nodes. Each MultiGPU node has the same functionality as its original counterpart but adds a `device` parameter that allows you to specify the GPU to use. Currently supported nodes (automatically detected if available): + - Standard [ComfyUI](https://github.com/comfyanonymous/ComfyUI) model loaders: - CheckpointLoaderSimpleMultiGPU - CLIPLoaderMultiGPU @@ -44,10 +45,11 @@ Currently supported nodes (automatically detected if available): - CheckpointLoaderNF4MultiGPU - HunyuanVideoWrapper (requires [ComfyUI-HunyuanVideoWrapper](https://github.com/kijai/ComfyUI-HunyuanVideoWrapper)): - HyVideoModelLoaderMultiGPU + - HyVideoModelLoaderDiffSynthMultiGPU (**NEW** - MultiGPU-specific node for offloading to an `offload_device` using MultiGPU's device selectors) - HyVideoVAELoaderMultiGPU - DownloadAndLoadHyVideoTextEncoderMultiGPU - - Native to ComfyUI-MultiGPU - - DeviceSelectorMultiGPU (Allows user to link loaders together to use the same selected device) +- Native to ComfyUI-MultiGPU + - DeviceSelectorMultiGPU (Allows user to link loaders together to use the same selected device) All MultiGPU nodes available for your install can be found in the "multigpu" category in the node menu. @@ -55,6 +57,11 @@ All MultiGPU nodes available for your install can be found in the "multigpu" cat All workflows have been tested on a 2x 3090 linux setup, a 4070 win 11 setup, and a 3090/1070ti linux setup. +### Split Hunyuan Video UNet across two devices and use DiffSynth Just-in-Time loading + +- [examples/hunyuanvideowrapper_diffsynth.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/hunyuanvideowrapper_diffsynth.json) +This workflow demonstrates DiffSynth's memory optimization strategy enabled in kijai's `ComfyUI-HunyuanVideoWrapper` UNet loader, splitting the UNet model across two CUDA devices using block-swapping. The main device handles active computations while blocks are swapped to and from the offload device as needed. As written, the CLIP loads on cuda:0 and then offloads, and the VAE is loaded after the UNet model has been cleared from memory after generation. This approach enables processing of higher resolution or longer duration videos that would exceed a single GPU's memory capacity, though at the cost of additional processing time. Note that an initial OOM error is expected as the workflow calibrates its memory management strategy - simply run the generation again with the same parameters. + ### Split Hunyuan Video generation across multiple resources - [examples/hunyuanvideowrapper_native_vae.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/hunyuanvideowrapper_native_vae.json) @@ -131,4 +138,4 @@ If you encounter problems, please [open an issue](https://github.com/pollockjj/C Originally created by [Alexander Dzhoganov](https://github.com/AlexanderDzhoganov). Implementation improved by [City96](https://v100s.net/). -Currently maintained by [pollockjj](https://github.com/pollockjj). \ No newline at end of file +Currently maintained by [pollockjj](https://github.com/pollockjj). diff --git a/__init__.py b/__init__.py index 3c9e4e9..30116ea 100644 --- a/__init__.py +++ b/__init__.py @@ -12,7 +12,10 @@ logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %( logging.info("MultiGPU: Initialization started") current_device = comfy.model_management.get_torch_device() +current_offload_device = comfy.model_management.get_torch_device() + logging.info(f"MultiGPU: Initial device {current_device}") +logging.info(f"MultiGPU: Initial offload device {current_offload_device}") def get_torch_device_patched(): if ( @@ -25,6 +28,15 @@ def get_torch_device_patched(): comfy.model_management.get_torch_device = get_torch_device_patched +def unet_offload_device_patched(): + if (not torch.cuda.is_available() + or comfy.model_management.cpu_state == comfy.model_management.CPUState.CPU + or "cpu" in str(current_offload_device).lower()): + return torch.device("cpu") + return torch.device(current_offload_device) + +comfy.model_management.unet_offload_device = unet_offload_device_patched + def get_device_list(): import torch return ["cpu"] + [f"cuda:{i}" for i in range(torch.cuda.device_count())] @@ -70,6 +82,33 @@ def override_class(cls): return NodeOverride +def override_class_with_offload(cls): + class NodeOverrideDiffSynth(cls): + @classmethod + def INPUT_TYPES(s): + inputs = copy.deepcopy(cls.INPUT_TYPES()) + devices = get_device_list() + default_device = devices[1] if len(devices) > 1 else devices[0] + inputs["optional"] = inputs.get("optional", {}) + inputs["optional"]["device"] = (devices, {"default": default_device}) + inputs["optional"]["offload_device"] = (devices, {"default": "cpu"}) + return inputs + + CATEGORY = "multigpu" + FUNCTION = "override" + + def override(self, *args, device=None, offload_device=None, **kwargs): + global current_device + global current_offload_device + if device is not None: + current_device = device + if offload_device is not None: + current_offload_device = offload_device + fn = getattr(super(), cls.FUNCTION) + return fn(*args, **kwargs) + + return NodeOverrideDiffSynth + NODE_CLASS_MAPPINGS = { "DeviceSelectorMultiGPU": DeviceSelectorMultiGPU } @@ -617,19 +656,18 @@ def register_PulidEvaClipLoader(): logging.info(f"MultiGPU: Registered PulidEvaClipLoaderMultiGPU") def register_HyVideoModelLoader(): - global NODE_CLASS_MAPPINGS + # Keep original MultiGPU wrapper unchanged class HyVideoModelLoader: @classmethod def INPUT_TYPES(s): return { "required": { "model": (folder_paths.get_filename_list("diffusion_models"), {"tooltip": "These models are loaded from the 'ComfyUI/models/diffusion_models' -folder",}), - - "base_precision": (["fp32", "bf16"], {"default": "bf16"}), - "quantization": (['disabled', 'fp8_e4m3fn', 'fp8_e4m3fn_fast', 'fp8_scaled', 'torchao_fp8dq', "torchao_fp8dqrow", "torchao_int8dq", "torchao_fp6", "torchao_int4", "torchao_int8"], {"default": 'disabled', "tooltip": "optional quantization method"}), - "load_device": (["main_device"], {"default": "main_device"}), + "base_precision": (["fp32", "bf16"], {"default": "bf16"}), + "quantization": (['disabled', 'fp8_e4m3fn', 'fp8_e4m3fn_fast', 'fp8_scaled', 'torchao_fp8dq', "torchao_fp8dqrow", "torchao_int8dq", "torchao_fp6", "torchao_int4", "torchao_int8"], {"default": 'disabled', "tooltip": "optional quantization method"}), + "load_device": (["main_device"], {"default": "main_device"}), }, "optional": { "attention_mode": ([ @@ -637,7 +675,7 @@ def register_HyVideoModelLoader(): "flash_attn_varlen", "sageattn_varlen", "comfy", - ], {"default": "flash_attn"}), + ], {"default": "flash_attn"}), "compile_args": ("COMPILEARGS", ), "block_swap_args": ("BLOCKSWAPARGS", ), "lora": ("HYVIDLORA", {"default": None}), @@ -650,13 +688,53 @@ def register_HyVideoModelLoader(): FUNCTION = "loadmodel" CATEGORY = "HunyuanVideoWrapper" - def loadmodel(self, model, base_precision, load_device, quantization, compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None, auto_cpu_offload=False): + def loadmodel(self, model, base_precision, load_device, quantization, compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None, auto_cpu_offload=False): from nodes import NODE_CLASS_MAPPINGS original_loader = NODE_CLASS_MAPPINGS["HyVideoModelLoader"]() return original_loader.loadmodel(model, base_precision, load_device, quantization, compile_args, attention_mode, block_swap_args, lora, auto_cpu_offload) - + + # Add new DiffSynth-style node + class HyVideoModelLoaderDiffSynth: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": (folder_paths.get_filename_list("diffusion_models"), {"tooltip": "These models are loaded from the 'ComfyUI/models/diffusion_models' -folder",}), + "base_precision": (["fp32", "bf16"], {"default": "bf16"}), + "quantization": (['disabled', 'fp8_e4m3fn', 'fp8_e4m3fn_fast', 'fp8_scaled', 'torchao_fp8dq', "torchao_fp8dqrow", "torchao_int8dq", "torchao_fp6", "torchao_int4", "torchao_int8"], + {"default": 'disabled', "tooltip": "optional quantization method"}), + }, + "optional": { + "attention_mode": ([ + "sdpa", + "flash_attn_varlen", + "sageattn_varlen", + "comfy", + ], {"default": "flash_attn"}), + "compile_args": ("COMPILEARGS", ), + "block_swap_args": ("BLOCKSWAPARGS", ), + "lora": ("HYVIDLORA", {"default": None}), + } + } + + RETURN_TYPES = ("HYVIDEOMODEL",) + RETURN_NAMES = ("model", ) + FUNCTION = "loadmodel" + CATEGORY = "HunyuanVideoWrapper" + + def loadmodel(self, model, base_precision, quantization, compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None): + from nodes import NODE_CLASS_MAPPINGS + original_loader = NODE_CLASS_MAPPINGS["HyVideoModelLoader"]() + # Use DiffSynth's auto offloading approach + return original_loader.loadmodel(model, base_precision, "main_device", quantization, + compile_args, attention_mode, block_swap_args, lora, + auto_cpu_offload=True) + + # Register both with MultiGPU wrapper NODE_CLASS_MAPPINGS["HyVideoModelLoaderMultiGPU"] = override_class(HyVideoModelLoader) - logging.info(f"MultiGPU: Registered HyVideoModelLoaderMultiGPU") + NODE_CLASS_MAPPINGS["HyVideoModelLoaderDiffSynthMultiGPU"] = override_class_with_offload(HyVideoModelLoaderDiffSynth) + + logging.info(f"MultiGPU: Registered HyVideoModelLoader nodes") def register_HyVideoVAELoader(): diff --git a/examples/hunyuanvideowrapper_diffsynth.json b/examples/hunyuanvideowrapper_diffsynth.json new file mode 100644 index 0000000..d2e531c --- /dev/null +++ b/examples/hunyuanvideowrapper_diffsynth.json @@ -0,0 +1,575 @@ +{ + "last_node_id": 57, + "last_link_id": 75, + "nodes": [ + { + "id": 34, + "type": "VHS_VideoCombine", + "pos": [ + 847.0758666992188, + -415.1882629394531 + ], + "size": [ + 512.8807373046875, + 1128.39453125 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 69 + }, + { + "name": "audio", + "type": "AUDIO", + "link": null, + "shape": 7 + }, + { + "name": "meta_batch", + "type": "VHS_BatchManager", + "link": null, + "shape": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 24, + "loop_count": 0, + "filename_prefix": "HunyuanVideo", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "HunyuanVideo_00182.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 24, + "workflow": "HunyuanVideo_00182.png", + "fullpath": "/home/johnj/ComfyUI/output/HunyuanVideo_00182.mp4" + }, + "muted": false + } + } + }, + { + "id": 45, + "type": "VAEDecodeTiled", + "pos": [ + 604.5225830078125, + -401.70220947265625 + ], + "size": [ + 210, + 150 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 70 + }, + { + "name": "vae", + "type": "VAE", + "link": 56 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 69 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecodeTiled" + }, + "widgets_values": [ + 256, + 64, + 64, + 8 + ] + }, + { + "id": 44, + "type": "VAELoaderMultiGPU", + "pos": [ + 208.21218872070312, + -586.37841796875 + ], + "size": [ + 353.3572692871094, + 84.02234649658203 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 56 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAELoaderMultiGPU" + }, + "widgets_values": [ + "hunyuan_video_vae_bf16.safetensors", + "cuda:1" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 53, + "type": "HyVideoTextImageEncode", + "pos": [ + -163.92579650878906, + -51.38969421386719 + ], + "size": [ + 367.4625244140625, + 528.5474853515625 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "text_encoders", + "type": "HYVIDTEXTENCODER", + "link": 74 + }, + { + "name": "custom_prompt_template", + "type": "PROMPT_TEMPLATE", + "link": null, + "shape": 7 + }, + { + "name": "clip_l", + "type": "CLIP", + "link": null, + "shape": 7 + }, + { + "name": "image1", + "type": "IMAGE", + "link": 73, + "shape": 7 + }, + { + "name": "image2", + "type": "IMAGE", + "link": null, + "shape": 7 + }, + { + "name": "hyvid_cfg", + "type": "HYVID_CFG", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "hyvid_embeds", + "type": "HYVIDEMBEDS", + "links": [ + 75 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "HyVideoTextImageEncode" + }, + "widgets_values": [ + "The animal shown in appears within its own natural setting, moving calmly or resting in place as a soft light casts delicate shadows across its form. Over the course of five seconds, it makes subtle shifts in posture or position, revealing small details of its features, such as the texture of its skin or fur, and the quiet rhythm of its breathing.", + "::3", + true, + "video", + "" + ] + }, + { + "id": 54, + "type": "LoadImage", + "pos": [ + -526.0797119140625, + 233.78311157226562 + ], + "size": [ + 315, + 314 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 73 + ], + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "pasted/image (165).png", + "image" + ] + }, + { + "id": 49, + "type": "DownloadAndLoadHyVideoTextEncoderMultiGPU", + "pos": [ + -591.7539672851562, + -37.14900588989258 + ], + "size": [ + 380.8912353515625, + 202 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "hyvid_text_encoder", + "type": "HYVIDTEXTENCODER", + "links": [ + 74 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "DownloadAndLoadHyVideoTextEncoderMultiGPU" + }, + "widgets_values": [ + "xtuner/llava-llama-3-8b-v1_1-transformers", + "openai/clip-vit-large-patch14", + "bf16", + false, + 2, + "disabled", + "cuda:0" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 3, + "type": "HyVideoSampler", + "pos": [ + 255.96482849121094, + -403.58502197265625 + ], + "size": [ + 315, + 630 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "HYVIDEOMODEL", + "link": 71 + }, + { + "name": "hyvid_embeds", + "type": "HYVIDEMBEDS", + "link": 75 + }, + { + "name": "samples", + "type": "LATENT", + "link": null, + "shape": 7 + }, + { + "name": "stg_args", + "type": "STGARGS", + "link": null, + "shape": 7 + }, + { + "name": "context_options", + "type": "HYVIDCONTEXT", + "link": null, + "shape": 7 + }, + { + "name": "feta_args", + "type": "FETAARGS", + "link": null, + "shape": 7 + }, + { + "name": "teacache_args", + "type": "TEACACHEARGS", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "links": [ + 70 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "HyVideoSampler" + }, + "widgets_values": [ + 768, + 1216, + 101, + 30, + 7.5, + 7.5, + 5770521, + "fixed", + true, + 1, + "FlowMatchDiscreteScheduler" + ] + }, + { + "id": 52, + "type": "HyVideoModelLoaderDiffSynthMultiGPU", + "pos": [ + -217.43194580078125, + -401.6243896484375 + ], + "size": [ + 441, + 218 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "type": "COMPILEARGS", + "link": null, + "shape": 7 + }, + { + "name": "block_swap_args", + "type": "BLOCKSWAPARGS", + "link": null, + "shape": 7 + }, + { + "name": "lora", + "type": "HYVIDLORA", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "model", + "type": "HYVIDEOMODEL", + "links": [ + 71 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "HyVideoModelLoaderDiffSynthMultiGPU" + }, + "widgets_values": [ + "hunyuan_video_FastVideo_720_fp8_e4m3fn.safetensors", + "bf16", + "fp8_e4m3fn_fast", + "sageattn_varlen", + "cuda:0", + "cuda:1" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 57, + "type": "Note", + "pos": [ + 256.481689453125, + 293.84710693359375 + ], + "size": [ + 389.5359191894531, + 349.2815856933594 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "Explanation of Memory Constraints and OOM Behavior", + "properties": {}, + "widgets_values": [ + "The settings in this workflow deliberately push memory usage beyond what a single 24GB RTX 3090 can handle. As a result, UNet blocks must be offloaded to an offload_device, which is assumed to be a second 24GB RTX 3090. The chosen resolution and number of frames illustrate that one-megapixel image size videos over four seconds long can be generated—even though it exceeds the primary GPU's native memory capacity using the \"just-in-time\" DiffSynth block-swapping approach. The DiffSynth JiT algorithm is straightforward, but not optimized for speed.\n\n**Expected OOM Behavior**\n\n* One Initial \"Out of Memory\"\nSeeing a single OOM error at the start is normal for this workflow. It happens because kijai's HunyuanVideo Sampler node initially attempts a full load on \"device\", even if it will surpass available VRAM. A second attempt using the same generation parameters will correctly trigger the offload process to \"offload_device\".\n\n* Repeated OOM Errors \nIf you keep getting OOM errors with the same settings (beyond the first one), then the requirements of your generation parameters still exceed what your system—even with offloading—can manage. You may need to lower resolution, shorten video length, or upgrade hardware resources." + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 51, + "type": "Note", + "pos": [ + -793.3005981445312, + -466.9315490722656 + ], + "size": [ + 564.3460693359375, + 355.50164794921875 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [], + "title": "****FIRST ATTEMPT LOADS ONLY ON PRIMARY DEVICE, POSSIBLE OOM****", + "properties": {}, + "widgets_values": [ + "The MultiGPU implementation for HunyuanVideo now offers two approaches:\n1. The original workflow (examples/hunyuanvideowrapper_native_vae.json)\n2. A new experimental DiffSynth-enabled workflow that enables more efficient multi-GPU usage (examples/hunyuanvideowrapper_DiffSynth.json)\n\nNew DiffSynth-Based Workflow (Experimental - THIS WORKFLOW):\n* Uses the \"HyVideoModelLoaderDiffSynthMultiGPU\" node which implements DiffSynth's memory management approach with TWO device selections\n* Adds a second device selector \"offload_device\" specifically for model offloading\n* Automatically enables efficient block swapping between GPUs via kijai's adaptation of the DiffSynth methodology\n* Provides better memory utilization by keeping actively computed blocks in VRAM\n* Assumes a 2-device solution and sets \"force_offload\" to \"true\" to avoid VAE loader OOM issues. A third 3090, for instance could be used and thus the VAE would stay in memory as well\n\nKnown Behaviors:\n* Initial OOM errors are expected and consistent with the underlying implementation, occurring in the attention mechanism of the first double block\n* Second attempts typically succeed regardless of offload device selection (CPU or GPU)\n* When running, the secondary device shows active utilization, confirming proper block swapping\n\n***** Will require a second attempt if first generation fails with OOM****\n\nNOTE: The original workflow remains the recommended stable approach. The DiffSynth-based version is provided as an experimental alternative for users who want to explore more aggressive memory optimization strategies to extend either resolution or duration options at the expense of speed." + ], + "color": "#233", + "bgcolor": "#355" + } + ], + "links": [ + [ + 56, + 44, + 0, + 45, + 1, + "VAE" + ], + [ + 69, + 45, + 0, + 34, + 0, + "IMAGE" + ], + [ + 70, + 3, + 0, + 45, + 0, + "LATENT" + ], + [ + 71, + 52, + 0, + 3, + 0, + "HYVIDEOMODEL" + ], + [ + 73, + 54, + 0, + 53, + 3, + "IMAGE" + ], + [ + 74, + 49, + 0, + 53, + 0, + "HYVIDTEXTENCODER" + ], + [ + 75, + 53, + 0, + 3, + 1, + "HYVIDEMBEDS" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.797202450000112, + "offset": [ + 1219.7255813472686, + 777.2824223551936 + ] + }, + "ue_links": [], + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0 + }, + "version": 0.4 +} \ No newline at end of file