diff --git a/README.md b/README.md index e0da359..a1e9634 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,9 @@ Timestep Embedding Aware Cache ([TeaCache](https://github.com/ali-vilab/TeaCache TeaCache has now been integrated into ComfyUI and is compatible with the ComfyUI native nodes. ComfyUI-TeaCache is easy to use, simply connect the TeaCache node with the ComfyUI native nodes for seamless usage. ## Updates +- Jan 14 2025: ComfyUI-TeaCache supports Compile Model and fixs a bug that TeaCache keeps forever even if we remove/bypass the node: + - Support Compile Model, now it can bring a faster inference when you add Compile Model node! + - Fixs a bug related to usability, now we can go back to the workflow state without TeaCache if we remove/bypass TeaCache node. - Jan 10 2025: ComfyUI-TeaCache supports LTX-Video: - It can achieve a 1.4x lossless speedup and a 1.7x speedup without much visual quality degradation. - Support Text to Video and Image to Video! @@ -22,16 +25,23 @@ Installation via ComfyUI-Manager is preferred. Simply search for ComfyUI-TeaCach 1. Go to comfyUI custom_nodes folder, `ComfyUI/custom_nodes/` 2. git clone https://github.com/welltop-cn/ComfyUI-TeaCache.git -## Recommended settings -The following table gives the recommended rel_l1_thresh ​for different models: +## Usage +### TeaCache +To use TeaCache node, simply add `TeaCache For Img Gen` or `TeaCache For Vid Gen` node to your workflow after `Load Diffusion Model` node or `Load LoRA` node (if you need LoRA). The following table gives the recommended rel_l1_thresh ​for different models: | | FLUX | HunyuanVideo | LTX-Video | |:---------------------:|:----------------------------:|:---------------------:|:---------------------:| | rel_l1_thresh | 0.4 | 0.15 | 0.06 | | speedup | ~2x | ~2x | ~1.7x | -## Usage -The demo workflows are placed in examples folder. +The demo workflows ([teacache_flux](./examples/teacache_flux.json), [teacache_hunyuanvideo](./examples/teacache_hunyuanvideo.json), [teacache_ltx_video](./examples/teacache_ltx_video.json)) are placed in examples folder. + +### Compile Model +To use Compile Model node, simply add `Compile Model` node to your workflow after TeaCache node. Compile Model uses `torch.compile` to enhance the model performance by compiling model into more efficient intermediate representations (IRs). This compilation process leverages backend compilers to generate optimized code, which can significantly speed up inference. The compilation may take long time when you run the workflow at first, but once it is compiled, inference is extremely fast. The usage is shown below: +![](./assets/compile.png) + +The demo workflows ([teacache_compile_flux](./examples/teacache_compile_flux.json), [teacache_compile_hunyuanvideo](./examples/teacache_compile_hunyuanvideo.json), [teacache_compile_ltx_video](./examples/teacache_compile_ltx_video.json)) are also placed in examples folder. + ## Demo -

FLUX

diff --git a/assets/compile.png b/assets/compile.png new file mode 100644 index 0000000..14349d9 Binary files /dev/null and b/assets/compile.png differ diff --git a/assets/teacache_flux.png b/assets/teacache_flux.png deleted file mode 100644 index d6af57e..0000000 Binary files a/assets/teacache_flux.png and /dev/null differ diff --git a/assets/teacache_hunyuanvideo.png b/assets/teacache_hunyuanvideo.png deleted file mode 100644 index 6fab93d..0000000 Binary files a/assets/teacache_hunyuanvideo.png and /dev/null differ diff --git a/assets/teacache_ltx_video.png b/assets/teacache_ltx_video.png deleted file mode 100644 index 553d3e6..0000000 Binary files a/assets/teacache_ltx_video.png and /dev/null differ diff --git a/examples/teacache_compile_flux.json b/examples/teacache_compile_flux.json new file mode 100644 index 0000000..8cbadfd --- /dev/null +++ b/examples/teacache_compile_flux.json @@ -0,0 +1,983 @@ +{ + "last_node_id": 41, + "last_link_id": 124, + "nodes": [ + { + "id": 11, + "type": "DualCLIPLoader", + "pos": [ + 48, + 288 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 10 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "DualCLIPLoader" + }, + "widgets_values": [ + "t5xxl_fp16.safetensors", + "clip_l.safetensors", + "flux" + ] + }, + { + "id": 26, + "type": "FluxGuidance", + "pos": [ + 480, + 144 + ], + "size": [ + 317.4000244140625, + 58 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 41 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 42 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "FluxGuidance" + }, + "widgets_values": [ + 3.5 + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 37, + "type": "Note", + "pos": [ + 480, + 1344 + ], + "size": [ + 314.99755859375, + 117.98363494873047 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "The reference sampling implementation auto adjusts the shift value based on the resolution, if you don't want this you can just bypass (CTRL-B) this ModelSamplingFlux node.\n" + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 10, + "type": "VAELoader", + "pos": [ + 48, + 432 + ], + "size": [ + 311.81634521484375, + 60.429901123046875 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 12 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "ae.safetensors" + ] + }, + { + "id": 28, + "type": "Note", + "pos": [ + 48, + 576 + ], + "size": [ + 336, + 288 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": { + "text": "" + }, + "widgets_values": [ + "If you get an error in any of the nodes above make sure the files are in the correct directories.\n\nSee the top of the examples page for the links : https://comfyanonymous.github.io/ComfyUI_examples/flux/\n\nflux1-dev.safetensors goes in: ComfyUI/models/unet/\n\nt5xxl_fp16.safetensors and clip_l.safetensors go in: ComfyUI/models/clip/\n\nae.safetensors goes in: ComfyUI/models/vae/\n\n\nTip: You can set the weight_dtype above to one of the fp8 types if you have memory issues." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 34, + "type": "PrimitiveNode", + "pos": [ + 432, + 480 + ], + "size": [ + 210, + 82 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 112, + 115 + ], + "slot_index": 0, + "widget": { + "name": "width" + } + } + ], + "title": "width", + "properties": { + "Run widget replace on values": false + }, + "widgets_values": [ + 1024, + "fixed" + ], + "color": "#323", + "bgcolor": "#535" + }, + { + "id": 35, + "type": "PrimitiveNode", + "pos": [ + 672, + 480 + ], + "size": [ + 210, + 82 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 113, + 114 + ], + "slot_index": 0, + "widget": { + "name": "height" + } + } + ], + "title": "height", + "properties": { + "Run widget replace on values": false + }, + "widgets_values": [ + 1024, + "fixed" + ], + "color": "#323", + "bgcolor": "#535" + }, + { + "id": 30, + "type": "ModelSamplingFlux", + "pos": [ + 480, + 1152 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 124, + "slot_index": 0 + }, + { + "name": "width", + "type": "INT", + "link": 115, + "slot_index": 1, + "widget": { + "name": "width" + } + }, + { + "name": "height", + "type": "INT", + "link": 114, + "slot_index": 2, + "widget": { + "name": "height" + } + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 54, + 55 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "ModelSamplingFlux" + }, + "widgets_values": [ + 1.15, + 0.5, + 1024, + 1024 + ] + }, + { + "id": 22, + "type": "BasicGuider", + "pos": [ + 576, + 48 + ], + "size": [ + 222.3482666015625, + 46 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 54, + "slot_index": 0 + }, + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 42, + "slot_index": 1 + } + ], + "outputs": [ + { + "name": "GUIDER", + "type": "GUIDER", + "links": [ + 30 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "BasicGuider" + }, + "widgets_values": [] + }, + { + "id": 13, + "type": "SamplerCustomAdvanced", + "pos": [ + 864, + 192 + ], + "size": [ + 272.3617858886719, + 124.53733825683594 + ], + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "noise", + "type": "NOISE", + "link": 37, + "slot_index": 0 + }, + { + "name": "guider", + "type": "GUIDER", + "link": 30, + "slot_index": 1 + }, + { + "name": "sampler", + "type": "SAMPLER", + "link": 19, + "slot_index": 2 + }, + { + "name": "sigmas", + "type": "SIGMAS", + "link": 20, + "slot_index": 3 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 116, + "slot_index": 4 + } + ], + "outputs": [ + { + "name": "output", + "type": "LATENT", + "links": [ + 24 + ], + "slot_index": 0, + "shape": 3 + }, + { + "name": "denoised_output", + "type": "LATENT", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "SamplerCustomAdvanced" + }, + "widgets_values": [] + }, + { + "id": 16, + "type": "KSamplerSelect", + "pos": [ + 480, + 912 + ], + "size": [ + 315, + 58 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "SAMPLER", + "type": "SAMPLER", + "links": [ + 19 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "KSamplerSelect" + }, + "widgets_values": [ + "euler" + ] + }, + { + "id": 27, + "type": "EmptySD3LatentImage", + "pos": [ + 480, + 624 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "width", + "type": "INT", + "link": 112, + "widget": { + "name": "width" + } + }, + { + "name": "height", + "type": "INT", + "link": 113, + "widget": { + "name": "height" + } + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 116 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "EmptySD3LatentImage" + }, + "widgets_values": [ + 1024, + 1024, + 1 + ] + }, + { + "id": 6, + "type": "CLIPTextEncode", + "pos": [ + 384, + 240 + ], + "size": [ + 422.84503173828125, + 164.31304931640625 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 10 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 41 + ], + "slot_index": 0 + } + ], + "title": "CLIP Text Encode (Positive Prompt)", + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "cute anime girl with massive fluffy fennec ears and a big fluffy tail blonde messy long hair blue eyes wearing a maid outfit with a long black gold leaf pattern dress and a white apron mouth open holding a fancy black forest cake with candles on top in the kitchen of an old dark Victorian mansion lit by candlelight with a bright window to the foggy forest and very expensive stuff everywhere", + true + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 866, + 367 + ], + "size": [ + 210, + 46 + ], + "flags": {}, + "order": 18, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 24 + }, + { + "name": "vae", + "type": "VAE", + "link": 12 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 9 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 17, + "type": "BasicScheduler", + "pos": [ + 480, + 1008 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 16, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 55, + "slot_index": 0 + } + ], + "outputs": [ + { + "name": "SIGMAS", + "type": "SIGMAS", + "links": [ + 20 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "BasicScheduler" + }, + "widgets_values": [ + "simple", + 30, + 1 + ] + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1171.8719482421875, + 197.6873321533203 + ], + "size": [ + 688.3538818359375, + 574.4689331054688 + ], + "flags": {}, + "order": 19, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 9 + } + ], + "outputs": [], + "properties": { + "Node name for S&R": "SaveImage" + }, + "widgets_values": [ + "ComfyUI" + ] + }, + { + "id": 25, + "type": "RandomNoise", + "pos": [ + 480, + 768 + ], + "size": [ + 315, + 82 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "NOISE", + "type": "NOISE", + "links": [ + 37 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "RandomNoise" + }, + "widgets_values": [ + 725991276578445, + "randomize" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 12, + "type": "UNETLoader", + "pos": [ + 48, + 144 + ], + "size": [ + 315, + 82 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 122 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "UNETLoader" + }, + "widgets_values": [ + "flux1-dev.safetensors", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 40, + "type": "TeaCacheForImgGen", + "pos": [ + 99.87027740478516, + 990.72705078125 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 122 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 123 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TeaCacheForImgGen" + }, + "widgets_values": [ + "flux", + 0.4, + 30 + ] + }, + { + "id": 41, + "type": "CompileModel", + "pos": [ + 103.08354187011719, + 1156.1611328125 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 123 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 124 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CompileModel" + }, + "widgets_values": [ + "default", + "inductor", + false, + false + ] + } + ], + "links": [ + [ + 9, + 8, + 0, + 9, + 0, + "IMAGE" + ], + [ + 10, + 11, + 0, + 6, + 0, + "CLIP" + ], + [ + 12, + 10, + 0, + 8, + 1, + "VAE" + ], + [ + 19, + 16, + 0, + 13, + 2, + "SAMPLER" + ], + [ + 20, + 17, + 0, + 13, + 3, + "SIGMAS" + ], + [ + 24, + 13, + 0, + 8, + 0, + "LATENT" + ], + [ + 30, + 22, + 0, + 13, + 1, + "GUIDER" + ], + [ + 37, + 25, + 0, + 13, + 0, + "NOISE" + ], + [ + 41, + 6, + 0, + 26, + 0, + "CONDITIONING" + ], + [ + 42, + 26, + 0, + 22, + 1, + "CONDITIONING" + ], + [ + 54, + 30, + 0, + 22, + 0, + "MODEL" + ], + [ + 55, + 30, + 0, + 17, + 0, + "MODEL" + ], + [ + 112, + 34, + 0, + 27, + 0, + "INT" + ], + [ + 113, + 35, + 0, + 27, + 1, + "INT" + ], + [ + 114, + 35, + 0, + 30, + 2, + "INT" + ], + [ + 115, + 34, + 0, + 30, + 1, + "INT" + ], + [ + 116, + 27, + 0, + 13, + 4, + "LATENT" + ], + [ + 122, + 12, + 0, + 40, + 0, + "MODEL" + ], + [ + 123, + 40, + 0, + 41, + 0, + "MODEL" + ], + [ + 124, + 41, + 0, + 30, + 0, + "MODEL" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.7513148009015778, + "offset": [ + 610.4698954060623, + -362.9900162421666 + ] + }, + "groupNodes": {} + }, + "version": 0.4 +} \ No newline at end of file diff --git a/examples/teacache_compile_hunyuanvideo.json b/examples/teacache_compile_hunyuanvideo.json new file mode 100644 index 0000000..2cac575 --- /dev/null +++ b/examples/teacache_compile_hunyuanvideo.json @@ -0,0 +1,936 @@ +{ + "last_node_id": 82, + "last_link_id": 227, + "nodes": [ + { + "id": 16, + "type": "KSamplerSelect", + "pos": [ + 484, + 751 + ], + "size": [ + 315, + 58 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "SAMPLER", + "type": "SAMPLER", + "links": [ + 19 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "KSamplerSelect" + }, + "widgets_values": [ + "euler" + ] + }, + { + "id": 26, + "type": "FluxGuidance", + "pos": [ + 520, + 100 + ], + "size": [ + 317.4000244140625, + 58 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 175 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 129 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "FluxGuidance" + }, + "widgets_values": [ + 6 + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 22, + "type": "BasicGuider", + "pos": [ + 600, + 0 + ], + "size": [ + 222.3482666015625, + 46 + ], + "flags": {}, + "order": 14, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 195, + "slot_index": 0 + }, + { + "name": "conditioning", + "type": "CONDITIONING", + "link": 129, + "slot_index": 1 + } + ], + "outputs": [ + { + "name": "GUIDER", + "type": "GUIDER", + "links": [ + 30 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "BasicGuider" + }, + "widgets_values": [] + }, + { + "id": 10, + "type": "VAELoader", + "pos": [ + 0, + 420 + ], + "size": [ + 350, + 60 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 206, + 211 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "hunyuan_video_vae_bf16.safetensors" + ] + }, + { + "id": 11, + "type": "DualCLIPLoader", + "pos": [ + 0, + 270 + ], + "size": [ + 350, + 106 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 205 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "DualCLIPLoader" + }, + "widgets_values": [ + "clip_l.safetensors", + "llava_llama3_fp8_scaled.safetensors", + "hunyuan_video" + ] + }, + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 1150, + 90 + ], + "size": [ + 210, + 46 + ], + "flags": {}, + "order": 16, + "mode": 2, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 181 + }, + { + "name": "vae", + "type": "VAE", + "link": 206 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 74, + "type": "Note", + "pos": [ + 1150, + 360 + ], + "size": [ + 210, + 170 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Use the tiled decode node by default because most people will need it.\n\nLower the tile_size and overlap if you run out of memory." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 13, + "type": "SamplerCustomAdvanced", + "pos": [ + 860, + 200 + ], + "size": [ + 272.3617858886719, + 124.53733825683594 + ], + "flags": {}, + "order": 15, + "mode": 0, + "inputs": [ + { + "name": "noise", + "type": "NOISE", + "link": 37, + "slot_index": 0 + }, + { + "name": "guider", + "type": "GUIDER", + "link": 30, + "slot_index": 1 + }, + { + "name": "sampler", + "type": "SAMPLER", + "link": 19, + "slot_index": 2 + }, + { + "name": "sigmas", + "type": "SIGMAS", + "link": 20, + "slot_index": 3 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 180, + "slot_index": 4 + } + ], + "outputs": [ + { + "name": "output", + "type": "LATENT", + "links": [ + 181, + 210 + ], + "slot_index": 0, + "shape": 3 + }, + { + "name": "denoised_output", + "type": "LATENT", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "SamplerCustomAdvanced" + }, + "widgets_values": [] + }, + { + "id": 44, + "type": "CLIPTextEncode", + "pos": [ + 420, + 200 + ], + "size": [ + 422.84503173828125, + 164.31304931640625 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 205 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 175 + ], + "slot_index": 0 + } + ], + "title": "CLIP Text Encode (Positive Prompt)", + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "anime style anime girl with massive fennec ears and one big fluffy tail, she has blonde hair long hair blue eyes wearing a pink sweater and a long blue skirt walking in a beautiful outdoor scenery with snow mountains in the background", + true + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 17, + "type": "BasicScheduler", + "pos": [ + 478, + 860 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 227, + "slot_index": 0 + } + ], + "outputs": [ + { + "name": "SIGMAS", + "type": "SIGMAS", + "links": [ + 20 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "BasicScheduler" + }, + "widgets_values": [ + "simple", + 25, + 1 + ] + }, + { + "id": 45, + "type": "EmptyHunyuanLatentVideo", + "pos": [ + 475.540771484375, + 432.673583984375 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 180 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyHunyuanLatentVideo" + }, + "widgets_values": [ + 848, + 848, + 73, + 1 + ] + }, + { + "id": 80, + "type": "VHS_VideoCombine", + "pos": [ + 1456.383544921875, + 199.92672729492188 + ], + "size": [ + 292.3946838378906, + 596.3946533203125 + ], + "flags": {}, + "order": 18, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 219 + }, + { + "name": "audio", + "type": "AUDIO", + "link": null, + "shape": 7 + }, + { + "name": "meta_batch", + "type": "VHS_BatchManager", + "link": null, + "shape": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 24, + "loop_count": 0, + "filename_prefix": "teacache", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "teacache_00011.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 24 + }, + "muted": false + } + } + }, + { + "id": 67, + "type": "ModelSamplingSD3", + "pos": [ + 361.5338439941406, + -5.368382453918457 + ], + "size": [ + 210, + 58 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 226 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 195 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "ModelSamplingSD3" + }, + "widgets_values": [ + 7 + ] + }, + { + "id": 73, + "type": "VAEDecodeTiled", + "pos": [ + 1150, + 200 + ], + "size": [ + 210, + 150 + ], + "flags": {}, + "order": 17, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 210 + }, + { + "name": "vae", + "type": "VAE", + "link": 211 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 219 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecodeTiled" + }, + "widgets_values": [ + 256, + 64, + 64, + 8 + ] + }, + { + "id": 25, + "type": "RandomNoise", + "pos": [ + 479, + 618 + ], + "size": [ + 315, + 82 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "NOISE", + "type": "NOISE", + "links": [ + 37 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "RandomNoise" + }, + "widgets_values": [ + 1, + "fixed" + ], + "color": "#2a363b", + "bgcolor": "#3f5159" + }, + { + "id": 12, + "type": "UNETLoader", + "pos": [ + -358.96661376953125, + 132.7143096923828 + ], + "size": [ + 350, + 82 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 222 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "UNETLoader" + }, + "widgets_values": [ + "hunyuan_video_t2v_720p_bf16.safetensors", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 77, + "type": "Note", + "pos": [ + -404.2021789550781, + -47.617000579833984 + ], + "size": [ + 350, + 110 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "Select a fp8 weight_dtype if you are running out of memory." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 81, + "type": "TeaCacheForVidGen", + "pos": [ + 22.06330108642578, + -82.81591033935547 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 222 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 225 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TeaCacheForVidGen" + }, + "widgets_values": [ + "hunyuan_video", + 0.15, + 25 + ] + }, + { + "id": 82, + "type": "CompileModel", + "pos": [ + 20.815486907958984, + 79.58741760253906 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 225 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 226, + 227 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CompileModel" + }, + "widgets_values": [ + "default", + "inductor", + false, + false + ] + } + ], + "links": [ + [ + 19, + 16, + 0, + 13, + 2, + "SAMPLER" + ], + [ + 20, + 17, + 0, + 13, + 3, + "SIGMAS" + ], + [ + 30, + 22, + 0, + 13, + 1, + "GUIDER" + ], + [ + 37, + 25, + 0, + 13, + 0, + "NOISE" + ], + [ + 129, + 26, + 0, + 22, + 1, + "CONDITIONING" + ], + [ + 175, + 44, + 0, + 26, + 0, + "CONDITIONING" + ], + [ + 180, + 45, + 0, + 13, + 4, + "LATENT" + ], + [ + 181, + 13, + 0, + 8, + 0, + "LATENT" + ], + [ + 195, + 67, + 0, + 22, + 0, + "MODEL" + ], + [ + 205, + 11, + 0, + 44, + 0, + "CLIP" + ], + [ + 206, + 10, + 0, + 8, + 1, + "VAE" + ], + [ + 210, + 13, + 0, + 73, + 0, + "LATENT" + ], + [ + 211, + 10, + 0, + 73, + 1, + "VAE" + ], + [ + 219, + 73, + 0, + 80, + 0, + "IMAGE" + ], + [ + 222, + 12, + 0, + 81, + 0, + "MODEL" + ], + [ + 225, + 81, + 0, + 82, + 0, + "MODEL" + ], + [ + 226, + 82, + 0, + 67, + 0, + "MODEL" + ], + [ + 227, + 82, + 0, + 17, + 0, + "MODEL" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.8264462809917363, + "offset": [ + 636.6668001785367, + 211.1442443052729 + ] + }, + "groupNodes": {}, + "workspace_info": { + "id": "hSBfPb8KVVUoyCyWT3oHM" + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/examples/teacache_compile_ltx_video.json b/examples/teacache_compile_ltx_video.json new file mode 100644 index 0000000..84eadfe --- /dev/null +++ b/examples/teacache_compile_ltx_video.json @@ -0,0 +1,761 @@ +{ + "last_node_id": 90, + "last_link_id": 190, + "nodes": [ + { + "id": 71, + "type": "LTXVScheduler", + "pos": [ + 856, + 531 + ], + "size": [ + 315, + 154 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "latent", + "type": "LATENT", + "link": 168, + "shape": 7 + } + ], + "outputs": [ + { + "name": "SIGMAS", + "type": "SIGMAS", + "links": [ + 182 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LTXVScheduler" + }, + "widgets_values": [ + 30, + 2.05, + 0.95, + true, + 0.1 + ] + }, + { + "id": 6, + "type": "CLIPTextEncode", + "pos": [ + 420, + 190 + ], + "size": [ + 422.84503173828125, + 164.31304931640625 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 74 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 169 + ], + "slot_index": 0 + } + ], + "title": "CLIP Text Encode (Positive Prompt)", + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "A woman with long brown hair and light skin smiles at another woman with long blonde hair. The woman with brown hair wears a black jacket and has a small, barely noticeable mole on her right cheek. The camera angle is a close-up, focused on the woman with brown hair's face. The lighting is warm and natural, likely from the setting sun, casting a soft glow on the scene. The scene appears to be real-life footage.", + true + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 7, + "type": "CLIPTextEncode", + "pos": [ + 420, + 390 + ], + "size": [ + 425.27801513671875, + 180.6060791015625 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 75 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 170 + ], + "slot_index": 0 + } + ], + "title": "CLIP Text Encode (Negative Prompt)", + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "low quality, worst quality, deformed, distorted, disfigured, motion smear, motion artifacts, fused fingers, bad anatomy, weird hand, ugly", + true + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 73, + "type": "KSamplerSelect", + "pos": [ + 860, + 420 + ], + "size": [ + 315, + 58 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "SAMPLER", + "type": "SAMPLER", + "links": [ + 172 + ] + } + ], + "properties": { + "Node name for S&R": "KSamplerSelect" + }, + "widgets_values": [ + "euler" + ] + }, + { + "id": 76, + "type": "Note", + "pos": [ + 40, + 350 + ], + "size": [ + 360, + 200 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [], + "properties": {}, + "widgets_values": [ + "This model needs long descriptive prompts, if the prompt is too short the quality will suffer greatly." + ], + "color": "#432", + "bgcolor": "#653" + }, + { + "id": 70, + "type": "EmptyLTXVLatentVideo", + "pos": [ + 860, + 240 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 168, + 175 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyLTXVLatentVideo" + }, + "widgets_values": [ + 768, + 768, + 97, + 1 + ] + }, + { + "id": 69, + "type": "LTXVConditioning", + "pos": [ + 920, + 60 + ], + "size": [ + 223.8660125732422, + 78 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "positive", + "type": "CONDITIONING", + "link": 169 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 170 + } + ], + "outputs": [ + { + "name": "positive", + "type": "CONDITIONING", + "links": [ + 166 + ], + "slot_index": 0 + }, + { + "name": "negative", + "type": "CONDITIONING", + "links": [ + 167 + ], + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "LTXVConditioning" + }, + "widgets_values": [ + 25 + ] + }, + { + "id": 38, + "type": "CLIPLoader", + "pos": [ + 60, + 190 + ], + "size": [ + 315, + 82 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 74, + 75 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPLoader" + }, + "widgets_values": [ + "t5xxl_fp16.safetensors", + "ltxv" + ] + }, + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 1600, + 30 + ], + "size": [ + 210, + 46 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 171 + }, + { + "name": "vae", + "type": "VAE", + "link": 87 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 185 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 86, + "type": "VHS_VideoCombine", + "pos": [ + 1890.7164306640625, + 30.7105770111084 + ], + "size": [ + 312.7515869140625, + 616.7515869140625 + ], + "flags": {}, + "order": 13, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 185 + }, + { + "name": "audio", + "type": "AUDIO", + "link": null, + "shape": 7 + }, + { + "name": "meta_batch", + "type": "VHS_BatchManager", + "link": null, + "shape": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": null, + "shape": 7 + } + ], + "outputs": [ + { + "name": "Filenames", + "type": "VHS_FILENAMES", + "links": null + } + ], + "properties": { + "Node name for S&R": "VHS_VideoCombine" + }, + "widgets_values": { + "frame_rate": 24, + "loop_count": 0, + "filename_prefix": "ltxv", + "format": "video/h264-mp4", + "pix_fmt": "yuv420p", + "crf": 19, + "save_metadata": true, + "pingpong": false, + "save_output": true, + "videopreview": { + "hidden": false, + "paused": false, + "params": { + "filename": "ltxv_00032.mp4", + "subfolder": "", + "type": "output", + "format": "video/h264-mp4", + "frame_rate": 24 + }, + "muted": false + } + } + }, + { + "id": 72, + "type": "SamplerCustom", + "pos": [ + 1206.866943359375, + 26.604873657226562 + ], + "size": [ + 355.20001220703125, + 230 + ], + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 190 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 166 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 167 + }, + { + "name": "sampler", + "type": "SAMPLER", + "link": 172 + }, + { + "name": "sigmas", + "type": "SIGMAS", + "link": 182 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 175 + } + ], + "outputs": [ + { + "name": "output", + "type": "LATENT", + "links": [ + 171 + ], + "slot_index": 0 + }, + { + "name": "denoised_output", + "type": "LATENT", + "links": null + } + ], + "properties": { + "Node name for S&R": "SamplerCustom" + }, + "widgets_values": [ + true, + 11905454606274, + "fixed", + 3 + ] + }, + { + "id": 44, + "type": "CheckpointLoaderSimple", + "pos": [ + 520.5762329101562, + 17.9000244140625 + ], + "size": [ + 315, + 98 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 187 + ], + "slot_index": 0 + }, + { + "name": "CLIP", + "type": "CLIP", + "links": null + }, + { + "name": "VAE", + "type": "VAE", + "links": [ + 87 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "ltx-video-2b-v0.9.1.safetensors" + ] + }, + { + "id": 89, + "type": "TeaCacheForVidGen", + "pos": [ + 883.0584716796875, + -323.0739440917969 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 187 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 189 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TeaCacheForVidGen" + }, + "widgets_values": [ + "ltxv", + 0.06, + 30 + ] + }, + { + "id": 90, + "type": "CompileModel", + "pos": [ + 889.2291870117188, + -153.8597412109375 + ], + "size": [ + 315, + 130 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 189 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 190 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CompileModel" + }, + "widgets_values": [ + "default", + "inductor", + false, + false + ] + } + ], + "links": [ + [ + 74, + 38, + 0, + 6, + 0, + "CLIP" + ], + [ + 75, + 38, + 0, + 7, + 0, + "CLIP" + ], + [ + 87, + 44, + 2, + 8, + 1, + "VAE" + ], + [ + 166, + 69, + 0, + 72, + 1, + "CONDITIONING" + ], + [ + 167, + 69, + 1, + 72, + 2, + "CONDITIONING" + ], + [ + 168, + 70, + 0, + 71, + 0, + "LATENT" + ], + [ + 169, + 6, + 0, + 69, + 0, + "CONDITIONING" + ], + [ + 170, + 7, + 0, + 69, + 1, + "CONDITIONING" + ], + [ + 171, + 72, + 0, + 8, + 0, + "LATENT" + ], + [ + 172, + 73, + 0, + 72, + 3, + "SAMPLER" + ], + [ + 175, + 70, + 0, + 72, + 5, + "LATENT" + ], + [ + 182, + 71, + 0, + 72, + 4, + "SIGMAS" + ], + [ + 185, + 8, + 0, + 86, + 0, + "IMAGE" + ], + [ + 187, + 44, + 0, + 89, + 0, + "MODEL" + ], + [ + 189, + 89, + 0, + 90, + 0, + "MODEL" + ], + [ + 190, + 90, + 0, + 72, + 0, + "MODEL" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.6830134553650712, + "offset": [ + 80.98795348793925, + 512.9846969483236 + ] + } + }, + "version": 0.4 +} \ No newline at end of file diff --git a/examples/teacache_flux.json b/examples/teacache_flux.json index fe8eaed..6aaa302 100644 --- a/examples/teacache_flux.json +++ b/examples/teacache_flux.json @@ -1,6 +1,6 @@ { - "last_node_id": 39, - "last_link_id": 119, + "last_node_id": 40, + "last_link_id": 122, "nodes": [ { "id": 11, @@ -259,7 +259,7 @@ { "name": "model", "type": "MODEL", - "link": 118, + "link": 121, "slot_index": 0 }, { @@ -541,42 +541,6 @@ "color": "#232", "bgcolor": "#353" }, - { - "id": 12, - "type": "UNETLoader", - "pos": [ - 48, - 144 - ], - "size": [ - 315, - 82 - ], - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": [ - 117 - ], - "slot_index": 0, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "UNETLoader" - }, - "widgets_values": [ - "flux1-dev.safetensors", - "default" - ], - "color": "#223", - "bgcolor": "#335" - }, { "id": 8, "type": "VAEDecode", @@ -700,7 +664,7 @@ 82 ], "flags": {}, - "order": 8, + "order": 7, "mode": 0, "inputs": [], "outputs": [ @@ -724,15 +688,51 @@ "bgcolor": "#3f5159" }, { - "id": 38, - "type": "TeaCacheForImgGen", + "id": 12, + "type": "UNETLoader", "pos": [ - 98.35289001464844, - 1152.2481689453125 + 48, + 144 ], "size": [ 315, - 130 + 82 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 122 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "UNETLoader" + }, + "widgets_values": [ + "flux1-dev.safetensors", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 40, + "type": "TeaCacheForImgGen", + "pos": [ + 77.93648529052734, + 1151.0115966796875 + ], + "size": [ + 315, + 106 ], "flags": {}, "order": 11, @@ -741,7 +741,7 @@ { "name": "model", "type": "MODEL", - "link": 117 + "link": 122 } ], "outputs": [ @@ -749,7 +749,7 @@ "name": "MODEL", "type": "MODEL", "links": [ - 118 + 121 ], "slot_index": 0 } @@ -758,7 +758,6 @@ "Node name for S&R": "TeaCacheForImgGen" }, "widgets_values": [ - true, "flux", 0.4, 30 @@ -903,18 +902,18 @@ "LATENT" ], [ - 117, - 12, + 121, + 40, 0, - 38, + 30, 0, "MODEL" ], [ - 118, - 38, + 122, + 12, 0, - 30, + 40, 0, "MODEL" ] @@ -923,10 +922,10 @@ "config": {}, "extra": { "ds": { - "scale": 0.6830134553650711, + "scale": 0.5644739300537778, "offset": [ - 674.5191353378493, - -473.9958452791898 + 789.503519235993, + -441.9800029782946 ] }, "groupNodes": {} diff --git a/examples/teacache_hunyuanvideo.json b/examples/teacache_hunyuanvideo.json index ac71b47..92505cc 100644 --- a/examples/teacache_hunyuanvideo.json +++ b/examples/teacache_hunyuanvideo.json @@ -1,6 +1,6 @@ { - "last_node_id": 80, - "last_link_id": 219, + "last_node_id": 81, + "last_link_id": 224, "nodes": [ { "id": 16, @@ -384,42 +384,6 @@ "color": "#432", "bgcolor": "#653" }, - { - "id": 12, - "type": "UNETLoader", - "pos": [ - -358.96661376953125, - 132.7143096923828 - ], - "size": [ - 350, - 82 - ], - "flags": {}, - "order": 5, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": [ - 216 - ], - "slot_index": 0, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "UNETLoader" - }, - "widgets_values": [ - "hunyuan_video_t2v_720p_bf16.safetensors", - "default" - ], - "color": "#223", - "bgcolor": "#335" - }, { "id": 17, "type": "BasicScheduler", @@ -438,7 +402,7 @@ { "name": "model", "type": "MODEL", - "link": 218, + "link": 224, "slot_index": 0 } ], @@ -473,7 +437,7 @@ 130 ], "flags": {}, - "order": 6, + "order": 5, "mode": 0, "inputs": [], "outputs": [ @@ -569,48 +533,6 @@ } } }, - { - "id": 78, - "type": "TeaCacheForVidGen", - "pos": [ - 18.11713981628418, - 76.63819885253906 - ], - "size": [ - 315, - 130 - ], - "flags": {}, - "order": 9, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "MODEL", - "link": 216 - } - ], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": [ - 217, - 218 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "TeaCacheForVidGen" - }, - "widgets_values": [ - true, - "hunyuan_video", - 0.15, - 25 - ] - }, { "id": 67, "type": "ModelSamplingSD3", @@ -629,7 +551,7 @@ { "name": "model", "type": "MODEL", - "link": 217 + "link": 223 } ], "outputs": [ @@ -707,7 +629,7 @@ 82 ], "flags": {}, - "order": 7, + "order": 6, "mode": 0, "inputs": [], "outputs": [ @@ -729,6 +651,83 @@ ], "color": "#2a363b", "bgcolor": "#3f5159" + }, + { + "id": 12, + "type": "UNETLoader", + "pos": [ + -358.96661376953125, + 132.7143096923828 + ], + "size": [ + 350, + 82 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 222 + ], + "slot_index": 0, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "UNETLoader" + }, + "widgets_values": [ + "hunyuan_video_t2v_720p_bf16.safetensors", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 81, + "type": "TeaCacheForVidGen", + "pos": [ + 20.529560089111328, + 102.00926208496094 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 222 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 223, + 224 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TeaCacheForVidGen" + }, + "widgets_values": [ + "hunyuan_video", + 0.15, + 25 + ] } ], "links": [ @@ -836,30 +835,6 @@ 1, "VAE" ], - [ - 216, - 12, - 0, - 78, - 0, - "MODEL" - ], - [ - 217, - 78, - 0, - 67, - 0, - "MODEL" - ], - [ - 218, - 78, - 0, - 17, - 0, - "MODEL" - ], [ 219, 73, @@ -867,16 +842,40 @@ 80, 0, "IMAGE" + ], + [ + 222, + 12, + 0, + 81, + 0, + "MODEL" + ], + [ + 223, + 81, + 0, + 67, + 0, + "MODEL" + ], + [ + 224, + 81, + 0, + 17, + 0, + "MODEL" ] ], "groups": [], "config": {}, "extra": { "ds": { - "scale": 0.5644739300537773, + "scale": 0.6830134553650712, "offset": [ - 580.5372164842474, - 242.0392570515768 + 553.6828346433346, + 75.84624922798709 ] }, "groupNodes": {}, diff --git a/examples/teacache_ltx_video.json b/examples/teacache_ltx_video.json index 5ee7d68..720f491 100644 --- a/examples/teacache_ltx_video.json +++ b/examples/teacache_ltx_video.json @@ -1,6 +1,6 @@ { - "last_node_id": 88, - "last_link_id": 185, + "last_node_id": 89, + "last_link_id": 188, "nodes": [ { "id": 71, @@ -436,7 +436,7 @@ { "name": "model", "type": "MODEL", - "link": 184 + "link": 188 }, { "name": "positive", @@ -489,47 +489,6 @@ 3 ] }, - { - "id": 85, - "type": "TeaCacheForVidGen", - "pos": [ - 864.0098266601562, - -156.047119140625 - ], - "size": [ - 315, - 130 - ], - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "MODEL", - "link": 183 - } - ], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": [ - 184 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "TeaCacheForVidGen" - }, - "widgets_values": [ - true, - "ltxv", - 0.06, - 30 - ] - }, { "id": 44, "type": "CheckpointLoaderSimple", @@ -550,7 +509,7 @@ "name": "MODEL", "type": "MODEL", "links": [ - 183 + 187 ], "slot_index": 0 }, @@ -574,6 +533,46 @@ "widgets_values": [ "ltx-video-2b-v0.9.1.safetensors" ] + }, + { + "id": 89, + "type": "TeaCacheForVidGen", + "pos": [ + 862.1427612304688, + -115.31109619140625 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 187 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 188 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "TeaCacheForVidGen" + }, + "widgets_values": [ + "ltxv", + 0.06, + 30 + ] } ], "links": [ @@ -673,22 +672,6 @@ 4, "SIGMAS" ], - [ - 183, - 44, - 0, - 85, - 0, - "MODEL" - ], - [ - 184, - 85, - 0, - 72, - 0, - "MODEL" - ], [ 185, 8, @@ -696,16 +679,32 @@ 86, 0, "IMAGE" + ], + [ + 187, + 44, + 0, + 89, + 0, + "MODEL" + ], + [ + 188, + 89, + 0, + 72, + 0, + "MODEL" ] ], "groups": [], "config": {}, "extra": { "ds": { - "scale": 0.7513148009015777, + "scale": 0.6830134553650712, "offset": [ - -12.24569670436394, - 293.53734923538843 + 67.19056247475572, + 283.72050363030013 ] } }, diff --git a/nodes.py b/nodes.py index e2f7288..8152b35 100644 --- a/nodes.py +++ b/nodes.py @@ -3,6 +3,7 @@ import torch import numpy as np from torch import Tensor +from unittest.mock import patch from comfy.ldm.flux.model import Flux from comfy.ldm.flux.layers import timestep_embedding @@ -11,6 +12,12 @@ from comfy.ldm.lightricks.model import LTXVModel, precompute_freqs_cis from comfy.ldm.common_dit import rms_norm +def poly1d(coefficients, x): + result = torch.zeros_like(x) + for i, coeff in enumerate(coefficients): + result += coeff * (x ** (len(coefficients) - 1 - i)) + return result + def teacache_flux_forward( self, img: Tensor, @@ -56,8 +63,7 @@ def teacache_flux_forward( self.accumulated_rel_l1_distance = 0 else: coefficients = [4.98651651e+02, -2.83781631e+02, 5.58554382e+01, -3.82021401e+00, 2.64230861e-01] - rescale_func = np.poly1d(coefficients) - self.accumulated_rel_l1_distance += rescale_func(((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean()).cpu().item()) + self.accumulated_rel_l1_distance += poly1d(coefficients, ((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean())) if self.accumulated_rel_l1_distance < self.rel_l1_thresh: should_calc = False else: @@ -198,8 +204,7 @@ def teacache_hunyuanvideo_forward( self.accumulated_rel_l1_distance = 0 else: coefficients = [7.33226126e+02, -4.01131952e+02, 6.75869174e+01, -3.14987800e+00, 9.61237896e-02] - rescale_func = np.poly1d(coefficients) - self.accumulated_rel_l1_distance += rescale_func(((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean()).cpu().item()) + self.accumulated_rel_l1_distance += poly1d(coefficients, ((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean())) if self.accumulated_rel_l1_distance < self.rel_l1_thresh: should_calc = False else: @@ -364,8 +369,7 @@ def teacache_ltxvmodel_forward( self.accumulated_rel_l1_distance = 0 else: coefficients = [2.14700694e+01, -1.28016453e+01, 2.31279151e+00, 7.92487521e-01, 9.69274326e-03] - rescale_func = np.poly1d(coefficients) - self.accumulated_rel_l1_distance += rescale_func(((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean()).cpu().item()) + self.accumulated_rel_l1_distance += poly1d(coefficients, ((modulated_inp-self.previous_modulated_input).abs().mean() / self.previous_modulated_input.abs().mean())) if self.accumulated_rel_l1_distance < self.rel_l1_thresh: should_calc = False else: @@ -432,7 +436,6 @@ class TeaCacheForImgGen: return { "required": { "model": ("MODEL", {"tooltip": "The image diffusion model the TeaCache will be applied to."}), - "enable_teacache": ("BOOLEAN", {"default": True, "tooltip": "Enable teacache will speed up inference but may lose visual quality."}), "model_type": (["flux"],), "rel_l1_thresh": ("FLOAT", {"default": 0.4, "min": 0.0, "max": 10.0, "step": 0.01, "tooltip": "How strongly to cache the output of diffusion model. This value must be non-negative."}), "steps": ("INT", {"default": 25, "min": 1, "max": 10000, "step": 1}), @@ -444,28 +447,32 @@ class TeaCacheForImgGen: CATEGORY = "TeaCache" TITLE = "TeaCache For Img Gen" - def apply_teacache(self, model, enable_teacache: bool, model_type: str, rel_l1_thresh: float, steps: int): - if enable_teacache: - model.model.diffusion_model.__class__.cnt = 0 - model.model.diffusion_model.__class__.rel_l1_thresh = rel_l1_thresh - model.model.diffusion_model.__class__.steps = steps - if model_type == "flux": - model.model.diffusion_model.forward_orig = teacache_flux_forward.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - else: - raise ValueError(f"Unknown type {model_type}") - else: - if model_type == "flux": - model.model.diffusion_model.forward_orig = Flux.forward_orig.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - else: - raise ValueError(f"Unknown type {model_type}") + def apply_teacache(self, model, model_type: str, rel_l1_thresh: float, steps: int): + new_model = model.clone() + diffusion_model = new_model.get_model_object("diffusion_model") + diffusion_model.__class__.cnt = 0 + diffusion_model.__class__.rel_l1_thresh = rel_l1_thresh + diffusion_model.__class__.steps = steps - return (model,) + if model_type == "flux": + forward_name = "forward_orig" + replaced_forward_fn = teacache_flux_forward.__get__( + diffusion_model, + diffusion_model.__class__ + ) + else: + raise ValueError(f"Unknown type {model_type}") + + def unet_wrapper_function(model_function, kwargs): + input = kwargs["input"] + timestep = kwargs["timestep"] + c = kwargs["c"] + with patch.object(diffusion_model, forward_name, replaced_forward_fn): + return model_function(input, timestep, **c) + + new_model.set_model_unet_function_wrapper(unet_wrapper_function) + + return (new_model,) class TeaCacheForVidGen: @classmethod @@ -473,7 +480,6 @@ class TeaCacheForVidGen: return { "required": { "model": ("MODEL", {"tooltip": "The video diffusion model the TeaCache will be applied to."}), - "enable_teacache": ("BOOLEAN", {"default": True, "tooltip": "Enable teacache will speed up inference but may lose visual quality."}), "model_type": (["hunyuan_video", "ltxv"],), "rel_l1_thresh": ("FLOAT", {"default": 0.15, "min": 0.0, "max": 10.0, "step": 0.01, "tooltip": "How strongly to cache the output of diffusion model. This value must be non-negative."}), "steps": ("INT", {"default": 25, "min": 1, "max": 10000, "step": 1}), @@ -485,40 +491,75 @@ class TeaCacheForVidGen: CATEGORY = "TeaCache" TITLE = "TeaCache For Vid Gen" - def apply_teacache(self, model, enable_teacache: bool, model_type: str, rel_l1_thresh: float, steps: int): - if enable_teacache: - model.model.diffusion_model.__class__.cnt = 0 - model.model.diffusion_model.__class__.rel_l1_thresh = rel_l1_thresh - model.model.diffusion_model.__class__.steps = steps - if model_type == "hunyuan_video": - model.model.diffusion_model.forward_orig = teacache_hunyuanvideo_forward.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - elif model_type == "ltxv": - model.model.diffusion_model.forward = teacache_ltxvmodel_forward.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - else: - raise ValueError(f"Unknown type {model_type}") - else: - if model_type == "hunyuan_video": - model.model.diffusion_model.forward_orig = HunyuanVideo.forward_orig.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - elif model_type == "ltxv": - model.model.diffusion_model.forward = LTXVModel.forward.__get__( - model.model.diffusion_model, - model.model.diffusion_model.__class__ - ) - else: - raise ValueError(f"Unknown type {model_type}") + def apply_teacache(self, model, model_type: str, rel_l1_thresh: float, steps: int): + new_model = model.clone() + diffusion_model = new_model.get_model_object("diffusion_model") + diffusion_model.__class__.cnt = 0 + diffusion_model.__class__.rel_l1_thresh = rel_l1_thresh + diffusion_model.__class__.steps = steps - return (model,) + if model_type == "hunyuan_video": + forward_name = "forward_orig" + replaced_forward_fn = teacache_hunyuanvideo_forward.__get__( + diffusion_model, + diffusion_model.__class__ + ) + elif model_type == "ltxv": + forward_name = "forward" + replaced_forward_fn = teacache_ltxvmodel_forward.__get__( + diffusion_model, + diffusion_model.__class__ + ) + else: + raise ValueError(f"Unknown type {model_type}") + + def unet_wrapper_function(model_function, kwargs): + input = kwargs["input"] + timestep = kwargs["timestep"] + c = kwargs["c"] + with patch.object(diffusion_model, forward_name, replaced_forward_fn): + return model_function(input, timestep, **c) + + new_model.set_model_unet_function_wrapper(unet_wrapper_function) + + return (new_model,) + +class CompileModel: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", {"tooltip": "The diffusion model the torch.compile will be applied to."}), + "mode": (["default", "max-autotune", "max-autotune-no-cudagraphs", "reduce-overhead"], {"default": "default"}), + "backend": (["inductor","cudagraphs"], {"default": "inductor"}), + "fullgraph": ("BOOLEAN", {"default": False, "tooltip": "Enable full graph mode"}), + "dynamic": ("BOOLEAN", {"default": False, "tooltip": "Enable dynamic mode"}), + } + } + + RETURN_TYPES = ("MODEL",) + FUNCTION = "apply_compile" + CATEGORY = "TeaCache" + TITLE = "Compile Model" + + def apply_compile(self, model, mode: str, backend: str, fullgraph: bool, dynamic: bool): + new_model = model.clone() + new_model.add_object_patch( + "diffusion_model", + torch.compile( + new_model.get_model_object("diffusion_model"), + mode=mode, + backend=backend, + fullgraph=fullgraph, + dynamic=dynamic + ) + ) + + return (new_model,) + NODE_CLASS_MAPPINGS = { "TeaCacheForImgGen": TeaCacheForImgGen, "TeaCacheForVidGen": TeaCacheForVidGen, + "CompileModel": CompileModel } diff --git a/pyproject.toml b/pyproject.toml index d03b182..8eb34d0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [project] name = "teacache" description = "Unofficial implementation of [ali-vilab/TeaCache](https://github.com/ali-vilab/TeaCache) for ComfyUI" -version = "1.0.0" +version = "1.1.0" license = {file = "LICENSE"} [project.urls]