diff --git a/README.md b/README.md index 27f667f..0bc5495 100644 --- a/README.md +++ b/README.md @@ -122,9 +122,12 @@ Currently supported nodes (automatically detected if available): - CheckpointLoaderNF4MultiGPU - HunyuanVideoWrapper (requires [ComfyUI-HunyuanVideoWrapper](https://github.com/kijai/ComfyUI-HunyuanVideoWrapper)): - HyVideoModelLoaderMultiGPU - - HyVideoModelLoaderDiffSynthMultiGPU (**NEW** - MultiGPU-specific node for offloading to an `offload_device` using MultiGPU's device selectors) - HyVideoVAELoaderMultiGPU - DownloadAndLoadHyVideoTextEncoderMultiGPU +- WanVideoWrapper (requires [ComfyUI-WanVideoWrapper](https://github.com/kijai/ComfyUI-WanVideoWrapper)): + - WanVideoModelLoader + - WanVideoVAELoader + - LoadWanVideoT5TextEncoder - **Native to ComfyUI-MultiGPU** - DeviceSelectorMultiGPU - Allows user to link loaders together to use the same selected device - HunyuanVideoEmbeddingsAdapter - Allows Kijai's excellent IP2V CLIP for HunyuanVideo to be used with Comfy Core sampler. @@ -146,6 +149,11 @@ This workflow attaches a HunyuanVideo GGUF-quantized model on `cuda:0` for compu - [examples/flux1dev_gguf_distorch.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/flux1dev_gguf_distorch.json) This workflow loads a FLUX.1-dev model on `cuda:0` for compute and distrubutes its UNet across multiple CUDA devices using new DisTorch distributed-load methodology. While the text encoders and VAE are loaded on GPU 1 and use `cuda:1` for compute. +### Split Wan Video generation across multiple resources + +- [examples/hunyuanvideowrapper_native_vae.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/wanvideo_T2V_example_MultiGPU.json) +This workflow is a simple extension of [kijai's T2V example](https://github.com/kijai/ComfyUI-WanVideoWrapper/blob/main/example_workflows/wanvideo_T2V_example_02.json) from his custom_node. + ### Split Hunyuan Video generation across multiple resources - [examples/hunyuanvideowrapper_native_vae.json](https://github.com/pollockjj/ComfyUI-MultiGPU/blob/main/examples/hunyuanvideowrapper_native_vae.json) diff --git a/__init__.py b/__init__.py index 0c5aff7..e63e507 100644 --- a/__init__.py +++ b/__init__.py @@ -27,7 +27,8 @@ from .nodes import ( LoadFluxControlNet, MMAudioModelLoader, MMAudioFeatureUtilsLoader, MMAudioSampler, PulidModelLoader, PulidInsightFaceLoader, PulidEvaClipLoader, - HyVideoModelLoader, HyVideoVAELoader, DownloadAndLoadHyVideoTextEncoder + HyVideoModelLoader, HyVideoVAELoader, DownloadAndLoadHyVideoTextEncoder, + WanVideoModelLoader, WanVideoVAELoader, LoadWanVideoT5TextEncoder ) current_device = mm.get_torch_device() @@ -735,4 +736,10 @@ if check_module_exists("ComfyUI-HunyuanVideoWrapper") or check_module_exists("co NODE_CLASS_MAPPINGS["HyVideoVAELoaderMultiGPU"] = override_class(HyVideoVAELoader) NODE_CLASS_MAPPINGS["DownloadAndLoadHyVideoTextEncoderMultiGPU"] = override_class(DownloadAndLoadHyVideoTextEncoder) +if check_module_exists("ComfyUI-WanVideoWrapper") or check_module_exists("comfyui-wanvideowrapper"): + NODE_CLASS_MAPPINGS["WanVideoModelLoaderMultiGPU"] = override_class(WanVideoModelLoader) + NODE_CLASS_MAPPINGS["WanVideoVAELoaderMultiGPU"] = override_class(WanVideoVAELoader) + NODE_CLASS_MAPPINGS["LoadWanVideoT5TextEncoderMultiGPU"] = override_class(LoadWanVideoT5TextEncoder) + + logging.info(f"MultiGPU: Registration complete. Final mappings: {', '.join(NODE_CLASS_MAPPINGS.keys())}") diff --git a/examples/wanvideo_T2V_example_MultiGPU.json b/examples/wanvideo_T2V_example_MultiGPU.json new file mode 100755 index 0000000..abe7618 --- /dev/null +++ b/examples/wanvideo_T2V_example_MultiGPU.json @@ -0,0 +1,1211 @@ +{ + "id": "c6e410bc-5e2c-460b-ae81-c91b6094fbb1", + "revision": 0, + "last_node_id": 59, + "last_link_id": 61, + "nodes": [ + { + "id": 37, + "type": "WanVideoEmptyEmbeds", + "pos": [ + 1305.26708984375, + -571.7843627929688 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "links": [ + 42 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoEmptyEmbeds" + }, + "widgets_values": [ + 832, + 480, + 81 + ] + }, + { + "id": 28, + "type": "WanVideoDecode", + "pos": [ + 1692.973876953125, + -404.8614501953125 + ], + "size": [ + 315, + 174 + ], + "flags": {}, + "order": 25, + "mode": 0, + "inputs": [ + { + "name": "vae", + "type": "WANVAE", + "link": 58 + }, + { + "name": "samples", + "type": "LATENT", + "link": 33 + } + ], + "outputs": [ + { + "name": "images", + "type": "IMAGE", + "slot_index": 0, + "links": [ + 36 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoDecode" + }, + "widgets_values": [ + true, + 272, + 272, + 144, + 128 + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 36, + "type": "Note", + "pos": [ + 723.7317504882812, + -597.3093872070312 + ], + "size": [ + 374.3061828613281, + 171.9547576904297 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "fp8_fast seems to cause huge quality degradation\n\nfp_16_fast enables \"Full FP16 Accmumulation in FP16 GEMMs\" feature available in the very latest pytorch nightly, this is around 20% speed boost. \n\nSageattn if you have it installed can be used for almost double inference speed" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 42, + "type": "Note", + "pos": [ + -165.44613647460938, + -344.9282531738281 + ], + "size": [ + 314.96246337890625, + 152.77333068847656 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Adjust the blocks to swap based on your VRAM, this is a tradeoff between speed and memory usage.\n\nAlternatively there's option to use VRAM management introduced in DiffSynt-Studios. This is usually slower, but saves even more VRAM compared to BlockSwap" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 50, + "type": "CLIPTextEncode", + "pos": [ + 630.8994750976562, + 1154.7454833984375 + ], + "size": [ + 400, + 200 + ], + "flags": {}, + "order": 20, + "mode": 2, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 53 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "slot_index": 0, + "links": [ + 55 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 48, + "type": "CLIPLoader", + "pos": [ + 270.8995361328125, + 904.7449340820312 + ], + "size": [ + 315, + 98.00003051757812 + ], + "flags": {}, + "order": 3, + "mode": 2, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "slot_index": 0, + "links": [ + 52, + 53 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPLoader" + }, + "widgets_values": [ + "umt5_xxl_fp16.safetensors", + "wan", + "default" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 51, + "type": "Note", + "pos": [ + 300.89947509765625, + 734.7444458007812 + ], + "size": [ + 253.16725158691406, + 88 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "You can also use native ComfyUI text encoding with these nodes instead of the original, the models are node specific and can't otherwise be mixed." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 49, + "type": "CLIPTextEncode", + "pos": [ + 630.8994750976562, + 904.7449340820312 + ], + "size": [ + 400, + 200 + ], + "flags": {}, + "order": 19, + "mode": 2, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 52 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "slot_index": 0, + "links": [ + 54 + ] + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "high quality nature video featuring a red panda balancing on a bamboo stem while a bird lands on it's head, on the background there is a waterfall" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 45, + "type": "WanVideoVRAMManagement", + "pos": [ + -158.19737243652344, + -136.97467041015625 + ], + "size": [ + 315, + 58 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "vram_management_args", + "type": "VRAM_MANAGEMENTARGS", + "links": [] + } + ], + "properties": { + "Node name for S&R": "WanVideoVRAMManagement" + }, + "widgets_values": [ + 1 + ] + }, + { + "id": 33, + "type": "Note", + "pos": [ + -153.7365264892578, + -16.124788284301758 + ], + "size": [ + 359.0753479003906, + 88 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Models:\nhttps://huggingface.co/Kijai/WanVideo_comfy/tree/main" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 53, + "type": "Note", + "pos": [ + 531.5562133789062, + -1014.3677978515625 + ], + "size": [ + 324.64129638671875, + 159.47401428222656 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "TeaCache could be considered to be sort of an automated step skipper \n\nThe relative l1 threshold -value determines how aggressive this is, higher values are faster but quality suffers more. Very first steps should NEVER be skipped with this model or it kills the motion. When using the pre-calculated coefficients, the treshold value should be much higher than with the default coefficients." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 52, + "type": "WanVideoTeaCache", + "pos": [ + 870.7489013671875, + -1000.0360717773438 + ], + "size": [ + 315, + 154 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "teacache_args", + "type": "TEACACHEARGS", + "links": [ + 56 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoTeaCache" + }, + "widgets_values": [ + 0.25, + 1, + -1, + "offload_device", + "true" + ] + }, + { + "id": 55, + "type": "WanVideoEnhanceAVideo", + "pos": [ + 1282.9122314453125, + -994.9732666015625 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "feta_args", + "type": "FETAARGS", + "links": [ + 57 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoEnhanceAVideo" + }, + "widgets_values": [ + 2, + 0, + 1 + ] + }, + { + "id": 54, + "type": "Note", + "pos": [ + 1278.7947998046875, + -1137.541748046875 + ], + "size": [ + 327.61932373046875, + 88 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Enhance-a-video can increase the fidelity of the results, too high values lead to noisy results." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 35, + "type": "WanVideoTorchCompileSettings", + "pos": [ + 222.5817413330078, + -677.6240844726562 + ], + "size": [ + 390.5999755859375, + 178 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "torch_compile_args", + "type": "WANCOMPILEARGS", + "slot_index": 0, + "links": [] + } + ], + "properties": { + "Node name for S&R": "WanVideoTorchCompileSettings" + }, + "widgets_values": [ + "inductor", + false, + "default", + false, + 64, + true + ] + }, + { + "id": 44, + "type": "Note", + "pos": [ + -98.58364868164062, + -675.3411254882812 + ], + "size": [ + 303.0501403808594, + 88 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "If you have Triton installed, connect this for ~30% speed increase" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 30, + "type": "VHS_VideoCombine", + "pos": [ + 2068.651611328125, + -582.5413818359375 + ], + "size": [ + 1245.8460693359375, + 1056.6922607421875 + ], + "flags": {}, + "order": 26, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 36 + }, + { + "name": "audio", + "shape": 7, + "type": "AUDIO", + "link": null + }, + { + "name": "meta_batch", + "shape": 7, + "type": "VHS_BatchManager", + "link": null + }, + { + "name": "vae", + "shape": 7, + "type": "VAE", + "link": null + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 16, + "loop_count": 0, + "filename_prefix": "WanVideo2_1_T2V", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "trim_to_audio": false, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "WanVideo2_1_T2V_00002.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 16, + "workflow": "WanVideo2_1_T2V_00002.png", + "fullpath": "/home/johnj/ComfyUI/output/WanVideo2_1_T2V_00002.mp4" + } + } + } + }, + { + "id": 16, + "type": "WanVideoTextEncode", + "pos": [ + 675.8850708007812, + -36.032100677490234 + ], + "size": [ + 420.30511474609375, + 261.5306701660156 + ], + "flags": {}, + "order": 21, + "mode": 0, + "inputs": [ + { + "name": "t5", + "type": "WANTEXTENCODER", + "link": 60 + }, + { + "name": "model_to_offload", + "shape": 7, + "type": "WANVIDEOMODEL", + "link": null + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "slot_index": 0, + "links": [ + 30 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoTextEncode" + }, + "widgets_values": [ + "high quality nature video featuring a red panda balancing on a bamboo stem while a bird lands on it's head, on the background there is a waterfall", + "色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走", + true + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 46, + "type": "WanVideoTextEmbedBridge", + "pos": [ + 1080.9002685546875, + 894.7449340820312 + ], + "size": [ + 315, + 46 + ], + "flags": {}, + "order": 23, + "mode": 2, + "inputs": [ + { + "name": "positive", + "type": "CONDITIONING", + "link": 54 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 55 + } + ], + "outputs": [ + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "links": null + } + ], + "properties": { + "Node name for S&R": "WanVideoTextEmbedBridge" + }, + "widgets_values": [] + }, + { + "id": 27, + "type": "WanVideoSampler", + "pos": [ + 1315.2401123046875, + -401.48028564453125 + ], + "size": [ + 315, + 574.1923217773438 + ], + "flags": {}, + "order": 24, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "link": 59 + }, + { + "name": "text_embeds", + "type": "WANVIDEOTEXTEMBEDS", + "link": 30 + }, + { + "name": "image_embeds", + "type": "WANVIDIMAGE_EMBEDS", + "link": 42 + }, + { + "name": "samples", + "shape": 7, + "type": "LATENT", + "link": null + }, + { + "name": "feta_args", + "shape": 7, + "type": "FETAARGS", + "link": 57 + }, + { + "name": "context_options", + "shape": 7, + "type": "WANVIDCONTEXT", + "link": null + }, + { + "name": "teacache_args", + "shape": 7, + "type": "TEACACHEARGS", + "link": 56 + }, + { + "name": "flowedit_args", + "shape": 7, + "type": "FLOWEDITARGS", + "link": null + }, + { + "name": "slg_args", + "shape": 7, + "type": "SLGARGS", + "link": null + }, + { + "name": "loop_args", + "shape": 7, + "type": "LOOPARGS", + "link": null + } + ], + "outputs": [ + { + "name": "samples", + "type": "LATENT", + "slot_index": 0, + "links": [ + 33 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoSampler" + }, + "widgets_values": [ + 25, + 6, + 5, + 1057359483639288, + "fixed", + true, + "unipc", + 0, + 1, + false, + "comfy" + ] + }, + { + "id": 38, + "type": "WanVideoVAELoader", + "pos": [ + 2040.6475830078125, + -974.1658325195312 + ], + "size": [ + 315, + 82 + ], + "flags": {}, + "order": 13, + "mode": 4, + "inputs": [], + "outputs": [ + { + "name": "vae", + "type": "WANVAE", + "slot_index": 0, + "links": [] + } + ], + "properties": { + "Node name for S&R": "WanVideoVAELoader" + }, + "widgets_values": [ + "wan_2.1_vae.safetensors", + "bf16" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 56, + "type": "WanVideoVAELoaderMultiGPU", + "pos": [ + 1705.1253662109375, + -595.9674682617188 + ], + "size": [ + 287.0846252441406, + 106 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "vae", + "type": "WANVAE", + "links": [ + 58 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoVAELoaderMultiGPU" + }, + "widgets_values": [ + "wan_2.1_vae.safetensors", + "bf16", + "cuda:1" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 59, + "type": "LoadWanVideoT5TextEncoderMultiGPU", + "pos": [ + 228.23208618164062, + -21.56218719482422 + ], + "size": [ + 415.8000183105469, + 154 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "wan_t5_model", + "type": "WANTEXTENCODER", + "links": [ + 60 + ] + } + ], + "properties": { + "Node name for S&R": "LoadWanVideoT5TextEncoderMultiGPU" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "main_device", + "disabled", + "cuda:0" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 39, + "type": "WanVideoBlockSwap", + "pos": [ + 253.16395568847656, + -343.3807678222656 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "block_swap_args", + "type": "BLOCKSWAPARGS", + "slot_index": 0, + "links": [ + 61 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoBlockSwap" + }, + "widgets_values": [ + 20, + false, + false, + true + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 58, + "type": "WanVideoModelLoaderMultiGPU", + "pos": [ + 723.1156616210938, + -363.45001220703125 + ], + "size": [ + 340.20001220703125, + 238 + ], + "flags": {}, + "order": 22, + "mode": 0, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": 61 + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "vram_management_args", + "shape": 7, + "type": "VRAM_MANAGEMENTARGS", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "links": [ + 59 + ] + } + ], + "properties": { + "Node name for S&R": "WanVideoModelLoaderMultiGPU" + }, + "widgets_values": [ + "wan2.1_t2v_1.3B_bf16.safetensors", + "bf16", + "disabled", + "main_device", + "sdpa", + "cuda:0" + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 11, + "type": "LoadWanVideoT5TextEncoder", + "pos": [ + 232.7933807373047, + 328.55560302734375 + ], + "size": [ + 377.1661376953125, + 130 + ], + "flags": {}, + "order": 17, + "mode": 4, + "inputs": [], + "outputs": [ + { + "name": "wan_t5_model", + "type": "WANTEXTENCODER", + "slot_index": 0, + "links": [] + } + ], + "properties": { + "Node name for S&R": "LoadWanVideoT5TextEncoder" + }, + "widgets_values": [ + "umt5-xxl-enc-bf16.safetensors", + "bf16", + "main_device", + "disabled" + ], + "color": "#332922", + "bgcolor": "#593930" + }, + { + "id": 22, + "type": "WanVideoModelLoader", + "pos": [ + 692.9950561523438, + 285.87725830078125 + ], + "size": [ + 477.4410095214844, + 226.43276977539062 + ], + "flags": {}, + "order": 18, + "mode": 4, + "inputs": [ + { + "name": "compile_args", + "shape": 7, + "type": "WANCOMPILEARGS", + "link": null + }, + { + "name": "block_swap_args", + "shape": 7, + "type": "BLOCKSWAPARGS", + "link": null + }, + { + "name": "lora", + "shape": 7, + "type": "WANVIDLORA", + "link": null + }, + { + "name": "vram_management_args", + "shape": 7, + "type": "VRAM_MANAGEMENTARGS", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "WANVIDEOMODEL", + "slot_index": 0, + "links": [] + } + ], + "properties": { + "Node name for S&R": "WanVideoModelLoader" + }, + "widgets_values": [ + "wan2.1_t2v_14B_bf16.safetensors", + "fp16", + "disabled", + "offload_device", + "sdpa" + ], + "color": "#223", + "bgcolor": "#335" + } + ], + "links": [ + [ + 30, + 16, + 0, + 27, + 1, + "WANVIDEOTEXTEMBEDS" + ], + [ + 33, + 27, + 0, + 28, + 1, + "LATENT" + ], + [ + 36, + 28, + 0, + 30, + 0, + "IMAGE" + ], + [ + 42, + 37, + 0, + 27, + 2, + "WANVIDIMAGE_EMBEDS" + ], + [ + 52, + 48, + 0, + 49, + 0, + "CLIP" + ], + [ + 53, + 48, + 0, + 50, + 0, + "CLIP" + ], + [ + 54, + 49, + 0, + 46, + 0, + "CONDITIONING" + ], + [ + 55, + 50, + 0, + 46, + 1, + "CONDITIONING" + ], + [ + 56, + 52, + 0, + 27, + 6, + "TEACACHEARGS" + ], + [ + 57, + 55, + 0, + 27, + 4, + "FETAARGS" + ], + [ + 58, + 56, + 0, + 28, + 0, + "WANVAE" + ], + [ + 59, + 58, + 0, + 27, + 0, + "WANVIDEOMODEL" + ], + [ + 60, + 59, + 0, + 16, + 0, + "WANTEXTENCODER" + ], + [ + 61, + 39, + 0, + 58, + 1, + "BLOCKSWAPARGS" + ] + ], + "groups": [ + { + "id": 1, + "title": "ComfyUI text encoding alternative", + "bounding": [ + 208.0836944580078, + 590.8111572265625, + 1210.621337890625, + 805.9080810546875 + ], + "color": "#3f789e", + "font_size": 24, + "flags": {} + } + ], + "config": {}, + "extra": { + "ds": { + "scale": 0.6209213230591559, + "offset": [ + 539.8142868513826, + 1149.0788831028588 + ] + }, + "node_versions": { + "ComfyUI-WanVideoWrapper": "5a2383621a05825d0d0437781afcb8552d9590fd", + "comfy-core": "0.3.26", + "ComfyUI-VideoHelperSuite": "0a75c7958fe320efcb052f1d9f8451fd20c730a8" + }, + "VHS_latentpreview": false, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file diff --git a/nodes.py b/nodes.py index 6c5bdeb..b36a794 100644 --- a/nodes.py +++ b/nodes.py @@ -473,4 +473,99 @@ class DownloadAndLoadHyVideoTextEncoder: def loadmodel(self, llm_model, clip_model, precision, apply_final_norm=False, hidden_state_skip_layer=2, quantization="disabled"): from nodes import NODE_CLASS_MAPPINGS original_loader = NODE_CLASS_MAPPINGS["DownloadAndLoadHyVideoTextEncoder"]() - return original_loader.loadmodel(llm_model, clip_model, precision, apply_final_norm, hidden_state_skip_layer, quantization) \ No newline at end of file + return original_loader.loadmodel(llm_model, clip_model, precision, apply_final_norm, hidden_state_skip_layer, quantization) +class WanVideoModelLoader: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": (folder_paths.get_filename_list("diffusion_models"), + {"tooltip": "These models are loaded from the 'ComfyUI/models/diffusion_models' folder",}), + "base_precision": (["fp32", "bf16", "fp16", "fp16_fast"], {"default": "bf16"}), + "quantization": ( + ['disabled', 'fp8_e4m3fn', 'fp8_e4m3fn_fast', 'fp8_e5m2', 'fp8_scaled', + 'torchao_fp8dq', "torchao_fp8dqrow", "torchao_int8dq", "torchao_fp6", + "torchao_int4", "torchao_int8"], + {"default": 'disabled', "tooltip": "optional quantization method"} + ), + "load_device": (["main_device"], {"default": "main_device"}), + }, + "optional": { + "attention_mode": ([ + "sdpa", + "flash_attn_2", + "flash_attn_3", + "sageattn", + ], {"default": "sdpa"}), + "compile_args": ("WANCOMPILEARGS", ), + "block_swap_args": ("BLOCKSWAPARGS", ), + "lora": ("WANVIDLORA", {"default": None}), + "vram_management_args": ("VRAM_MANAGEMENTARGS", + {"default": None, "tooltip": "Alternative offloading method"}), + } + } + + RETURN_TYPES = ("WANVIDEOMODEL",) + RETURN_NAMES = ("model", ) + FUNCTION = "loadmodel" + CATEGORY = "WanVideoWrapper" + + def loadmodel(self, model, base_precision, load_device, quantization, + compile_args=None, attention_mode="sdpa", block_swap_args=None, lora=None, vram_management_args=None): + from nodes import NODE_CLASS_MAPPINGS + original_loader = NODE_CLASS_MAPPINGS["WanVideoModelLoader"]() + return original_loader.loadmodel(model, base_precision, load_device, quantization, + compile_args, attention_mode, block_swap_args, lora, vram_management_args) + + +class WanVideoVAELoader: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model_name": (folder_paths.get_filename_list("vae"), + {"tooltip": "These models are loaded from 'ComfyUI/models/vae'"}), + }, + "optional": { + "precision": (["fp16", "fp32", "bf16"], {"default": "bf16"}), + } + } + + RETURN_TYPES = ("WANVAE",) + RETURN_NAMES = ("vae", ) + FUNCTION = "loadmodel" + CATEGORY = "WanVideoWrapper" + DESCRIPTION = "Loads Wan VAE model from 'ComfyUI/models/vae'" + + def loadmodel(self, model_name, precision): + from nodes import NODE_CLASS_MAPPINGS + original_loader = NODE_CLASS_MAPPINGS["WanVideoVAELoader"]() + return original_loader.loadmodel(model_name, precision) + + +class LoadWanVideoT5TextEncoder: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model_name": (folder_paths.get_filename_list("text_encoders"), + {"tooltip": "These models are loaded from 'ComfyUI/models/text_encoders'"}), + "precision": (["fp16", "fp32", "bf16"], {"default": "bf16"}), + }, + "optional": { + "load_device": (["main_device"], {"default": "main_device"}), + "quantization": (['disabled', 'fp8_e4m3fn'], + {"default": 'disabled', "tooltip": "optional quantization method"}), + } + } + + RETURN_TYPES = ("WANTEXTENCODER",) + RETURN_NAMES = ("wan_t5_model", ) + FUNCTION = "loadmodel" + CATEGORY = "WanVideoWrapper" + DESCRIPTION = "Loads Wan text_encoder model from 'ComfyUI/models/LLM'" + + def loadmodel(self, model_name, precision, load_device="offload_device", quantization="disabled"): + from nodes import NODE_CLASS_MAPPINGS + original_loader = NODE_CLASS_MAPPINGS["LoadWanVideoT5TextEncoder"]() + return original_loader.loadmodel(model_name, precision, load_device, quantization) diff --git a/pyproject.toml b/pyproject.toml index e8bab38..88aafc8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "comfyui-multigpu" -description = "This custom_node for ComfyUI adds one-click 'Virtual VRAM' for any GGUF UNet and CLIP loader, managing the offload of layers to DRAM or VRAM to maximize the latent space of your card. Also includes nodes for directly loading entire components (UNet, CLIP, VAE) onto the device you choose. Includes 16 examples covering common use cases." -version = "1.6.2" +description = "This custom_node for ComfyUI adds one-click 'Virtual VRAM' for any GGUF UNet and CLIP loader, managing the offload of layers to DRAM or VRAM to maximize the latent space of your card. Also includes nodes for directly loading entire components (UNet, CLIP, VAE) onto the device you choose. Includes 16 examples covering common use cases. Includes support for kijai's ComfyUI-WanVideoWrapper and ComfyUI-HunyuanVideoWrapper, among other popular loaders." +version = "1.7.0" license = {file = "LICENSE"} [project.urls]