Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2d7912f9a5 | ||
|
|
93b90fad99 | ||
|
|
d82b218e1c | ||
|
|
a9238253e0 | ||
|
|
391ca9625f | ||
|
|
e7e8166e02 | ||
|
|
f564ee886b | ||
|
|
0168172781 | ||
|
|
be34cf3054 | ||
|
|
649d6ea465 | ||
|
|
c054966ae8 | ||
|
|
e513cf6778 | ||
|
|
32cf848e93 | ||
|
|
12c8035ffa | ||
|
|
9fe919558f | ||
|
|
6948b2c766 | ||
|
|
2771dd17d0 | ||
|
|
f145728829 | ||
|
|
1e146fa465 |
@@ -13,17 +13,29 @@
|
||||
|
||||
Universally applicable inpainting ability for every model. LanPaint sampler lets the model "think" through multiple iterations before denoising, enabling you to invest more computation time for superior inpainting quality.
|
||||
|
||||
This is the official implementation of ["LanPaint: Training-Free Diffusion Inpainting with Asymptotically Exact and Fast Conditional Sampling"](https://arxiv.org/abs/2502.03491), accepted by TMLR.
|
||||
## What LanPaint Enables
|
||||
|
||||
The repository is for ComfyUI extension.
|
||||
- Training-free image inpainting
|
||||
- Outpainting and generative fill
|
||||
- Mask-constrained local image editing
|
||||
- Object / region replacement with text guidance
|
||||
- Character-consistent local generation
|
||||
- Video inpainting and local video editing
|
||||
- Video + audio masked generation
|
||||
|
||||
Diffusers Support: [LanPaint-Diffusers](https://github.com/charrywhite/LanPaint-diffusers) by [@charrywhite](https://github.com/charrywhite/)
|
||||
## Research & Benchmark
|
||||
|
||||
Benchmark code for paper reproduce: [LanPaintBench](https://github.com/scraed/LanPaintBench).
|
||||
LanPaint is a training-free partial conditional sampler that enables mask-constrained inpainting and local editing with pretrained diffusion and rectified-flow models, without fine-tuning or backpropagation.
|
||||
|
||||
## Citation
|
||||
* 📄 **Paper:** [LanPaint: Training-Free Diffusion Inpainting with Asymptotically Exact and Fast Conditional Sampling](https://openreview.net/forum?id=JPC8JyOUSW) — TMLR 2025
|
||||
* 🧩 **ComfyUI Implementation:** This repository
|
||||
* 🐍 **Diffusers Implementation:** [LanPaint-Diffusers](https://github.com/charrywhite/LanPaint-diffusers) by [@charrywhite](https://github.com/charrywhite/)
|
||||
* 🧪 **Benchmark & Reproduction:** [LanPaintBench](https://github.com/scraed/LanPaintBench)
|
||||
* 🌐 **Project Website:** [LanPaint Page](https://scraed.github.io/scraedBlog/lanpaint/)
|
||||
|
||||
```
|
||||
### Citation
|
||||
|
||||
```bibtex id="2r5ioa"
|
||||
@article{
|
||||
zheng2025lanpaint,
|
||||
title={LanPaint: Training-Free Diffusion Inpainting with Asymptotically Exact and Fast Conditional Sampling},
|
||||
@@ -31,17 +43,27 @@ author={Candi Zheng and Yuan Lan and Yang Wang},
|
||||
journal={Transactions on Machine Learning Research},
|
||||
issn={2835-8856},
|
||||
year={2025},
|
||||
url={https://openreview.net/forum?id=JPC8JyOUSW},
|
||||
note={}
|
||||
url={https://openreview.net/forum?id=JPC8JyOUSW}
|
||||
}
|
||||
```
|
||||
|
||||
**🎉 NEW 2026: Join our discord!**
|
||||
|
||||
[Join our Discord](https://discord.gg/yN5wYDE6W4) to share experiences, discuss features, and explore future development.
|
||||
|
||||
`v1.5.0` fixes an important hidden bug that reduced performance and could blur images (especially with `z-image-base`) and also boosts overall LanPaint performance across other models.
|
||||
`v2.1.0` significantly accelerates LanPaint with a new schedule mechanism and fixes MiniMax H3 support on the latest ComfyUI.
|
||||
If your inpainting results have wierd (glowing / broken) mask boundary, check this [issue](https://github.com/scraed/LanPaint/issues/80).
|
||||
|
||||
**🎨 NEW: LanPaint now supports Qwen-Image 2.1 - transparency, and masked image editing!**
|
||||
|
||||

|
||||
|
||||
Qwen 2.1's **image edit** model now works under a LanPaint mask: tell it what to change, paint over the part you want it to touch, and only that part changes. Hand it a second picture to borrow from if you want one. Check our latest [Qwen Image 2.1 Image Edit Example](#example-qwen-image-21-image-edit-masked-inpaintlanpaint-k-sampler-5-steps-of-thinking).
|
||||
|
||||

|
||||
|
||||
And if your picture carries transparency, it gets inpainted too - the rebuilt part comes back with a new outline, not just new colours. Check our latest [Qwen Image 2.1 Example](#example-qwen-image-21-inpaint-with-transparencylanpaint-k-sampler-5-steps-of-thinking).
|
||||
|
||||
**🎬 NEW: LanPaint now supports MiniMax H3 video + audio inpainting!**
|
||||
|
||||
| Masked Input (paint in the editor) | Mask (visible overlay) | Inpainted Result |
|
||||
@@ -103,10 +125,11 @@ Check our latest [Krea2 Example](#example-krea2-inpaintlanpaint-k-sampler-3-step
|
||||
- [Features](#features)
|
||||
- [Quickstart](#quickstart)
|
||||
- [How to Use Examples](#how-to-use-examples)
|
||||
- [Video Examples (Beta)](#video-examples-beta)
|
||||
- [Video Examples](#video-examples)
|
||||
- [Wan 2.2 Video Inpainting](#wan-22-video-inpainting)
|
||||
- [Wan 2.2 5B Video Inpainting](#wan-22-5b-video-inpainting)
|
||||
- [Wan 2.2 Video Outpainting](#wan-22-video-outpainting)
|
||||
- [MiniMax H3 Video + Audio Inpainting](#minimax-h3-video--audio-inpainting-av-pipeline)
|
||||
- [Resource Consumption](#resource-consumption)
|
||||
- [Image Examples](#image-examples)
|
||||
- [Flux.2.Dev](#example-flux2dev-inpaintlanpaint-k-sampler-5-steps-of-thinking)
|
||||
@@ -121,6 +144,8 @@ Check our latest [Krea2 Example](#example-krea2-inpaintlanpaint-k-sampler-3-step
|
||||
- [Wan 2.2 T2I with reference](#example-wan22-partial-inpaintlanpaint-k-sampler-5-steps-of-thinking)
|
||||
- [Qwen Image Edit 2511 2509](#example-qwen-edit-2509-inpaint)
|
||||
- [Qwen Image Edit 2508](#example-qwen-edit-2508-inpaint)
|
||||
- [Qwen Image 2.1 Image Edit](#example-qwen-image-21-image-edit-masked-inpaintlanpaint-k-sampler-5-steps-of-thinking)
|
||||
- [Qwen Image 2.1](#example-qwen-image-21-inpaint-with-transparencylanpaint-k-sampler-5-steps-of-thinking)
|
||||
- [Qwen Image](#example-qwen-image-inpaintlanpaint-k-sampler-5-steps-of-thinking)
|
||||
- [HiDream](#example-hidream-inpaint-lanpaint-k-sampler-5-steps-of-thinking)
|
||||
- [SD 3.5](#example-sd-35-inpaintlanpaint-k-sampler-5-steps-of-thinking)
|
||||
@@ -138,7 +163,7 @@ Check our latest [Krea2 Example](#example-krea2-inpaintlanpaint-k-sampler-3-step
|
||||
|
||||
## Features
|
||||
|
||||
- **Universal Compatibility** – Works instantly with almost any model (**Ideogram4, Krea2, Z-image, Z-image-base, Hunyuan, Wan 2.2, Qwen Image/Edit, Anima, HiDream, SD 3.5, Flux-series, SDXL, SD 1.5 or custom LoRAs**) and ControlNet.
|
||||
- **Universal Compatibility** – Works instantly with almost any model (**Ideogram4, Krea2, Z-image, Z-image-base, Hunyuan, Wan 2.2, Qwen Image 2.1/Image/Edit, Anima, HiDream, SD 3.5, Flux-series, SDXL, SD 1.5 or custom LoRAs**) and ControlNet.
|
||||

|
||||
- **No Training Needed** – Works out of the box with your existing model.
|
||||
- **Easy to Use** – Same workflow as standard ComfyUI KSampler.
|
||||
@@ -177,7 +202,7 @@ Once installed, you'll find the LanPaint nodes under the "sampling" category in
|
||||
- **[VAE Encode for Inpainting](https://comfyanonymous.github.io/ComfyUI_examples/inpaint/)**
|
||||
- **[Set Latent Noise Mask](https://comfyui-wiki.com/en/tutorial/basic/how-to-inpaint-an-image-in-comfyui)**
|
||||
|
||||
## Video Examples (Beta)
|
||||
## Video Examples
|
||||
|
||||
LanPaint now supports video inpainting with Wan 2.2, enabling you to seamlessly inpaint masked regions across video frames while maintaining temporal consistency.
|
||||
|
||||
@@ -223,6 +248,8 @@ LanPaint's AV pipeline inpaints video **and audio** together with the MiniMax H3
|
||||
|:----------------------------------:|:----------------------:|:----------------:|
|
||||
|  |  |  |
|
||||
|
||||

|
||||
|
||||
[View Workflow & Masks](https://github.com/scraed/LanPaint/tree/master/examples/Example_29) · [Workflow JSON](https://github.com/scraed/LanPaint/blob/master/example_workflows/MiniMax_H3_AV_EncodeDecode_Inpaint.json)
|
||||
|
||||
**How it works:**
|
||||
@@ -433,6 +460,20 @@ Check [Mased Qwen Edit Workflow](https://github.com/scraed/LanPaint/tree/master/
|
||||
|
||||
|
||||
|
||||
### Example Qwen Image 2.1 Image Edit: Masked InPaint(LanPaint K Sampler, 5 steps of thinking)
|
||||
|
||||
Qwen-Image 2.1's image edit model now works under a LanPaint mask: write what you want changed, paint over the part it should touch, and only that part changes - everything else, transparency included, comes back exactly as it was. In this example a second picture supplies the material for the earcups, and the headband and stitching stay as they are. Workflow and images are in `examples/Example_32`; drag `InPainted_Drag_Me_to_ComfyUI.png` into ComfyUI to load it. Use your own pictures with the official [Qwen Image 2.1 Image Edit template](https://docs.comfy.org/tutorials/image/qwen/qwen-image-2-1).
|
||||
|
||||

|
||||
[View Workflow & Masks](https://github.com/scraed/LanPaint/tree/master/examples/Example_32) · [Workflow JSON](https://github.com/scraed/LanPaint/blob/master/example_workflows/Qwen_Image_2.1_Edit_Masked_Inpaint.json)
|
||||
|
||||
### Example Qwen Image 2.1: InPaint with Transparency(LanPaint K Sampler, 5 steps of thinking)
|
||||
|
||||
Qwen-Image 2.1 inpaints a picture's transparency along with its pixels, so the rebuilt part can come back with a new outline instead of merely new colours - here the boot's sole is replaced and the silhouette grows with it. Workflow and images are in `examples/Example_31`; drag `InPainted_Drag_Me_to_ComfyUI.png` into ComfyUI to load it.
|
||||
|
||||

|
||||
[View Workflow & Masks](https://github.com/scraed/LanPaint/tree/master/examples/Example_31) · [Workflow JSON](https://github.com/scraed/LanPaint/blob/master/example_workflows/Transparent_Edit_EncodeDecode_Inpaint.json)
|
||||
|
||||
### Example Qwen Image: InPaint(LanPaint K Sampler, 5 steps of thinking)
|
||||
|
||||

|
||||
@@ -617,6 +658,14 @@ Submit a PR to add your tutorial/video here, or open an [Issue](https://github.c
|
||||
[Working togather with crop&stitch](https://github.com/scraed/LanPaint/issues/46)
|
||||
|
||||
## Updates
|
||||
- 2026/09/28
|
||||
- Add Qwen-Image 2.1 image edit support: masked, instruction-driven editing (Example_32).
|
||||
- Add Qwen-Image 2.1 inpainting support with LanPaint KSampler (Example_31).
|
||||
- Inpainting a picture that carries transparency now works end to end: the 2.1 VAE is 4-in/4-out, so the alpha travels through the latent and is edited alongside the pixels. Keep the inpainting mask in its own greyscale file, since 2.1's alpha channel means image transparency.
|
||||
- `LanPaint_ImageDecode` now matches the decoded channel count to the source image, so an RGBA source comes back RGBA and an RGB source still comes back RGB.
|
||||
- 2026/08/12
|
||||
- `v2.1.0`: Significantly accelerated LanPaint using a new schedule mechanism.
|
||||
- Fix bugs for MiniMax H3 on the latest ComfyUI.
|
||||
- 2026/08/09
|
||||
- Add MiniMax H3 video + audio inpainting support (Example_29): paint per-frame video masks and audio intervals in one editor session, encode both streams into a nested AV latent, sample once, and decode back with the source fps and bit depth preserved.
|
||||
- The mask editor can export the masks into the video itself (mp4 metadata) - share a single video file and the masks travel with it.
|
||||
|
||||
@@ -59,7 +59,11 @@ def _install_lightweight_runtime_stubs() -> None:
|
||||
class DummyKSAMPLER: # noqa: N801 (match ComfyUI naming)
|
||||
pass
|
||||
|
||||
class KSampler: # noqa: N801 (match ComfyUI naming)
|
||||
SCHEDULERS = ["normal", "karras", "exponential", "sgm_uniform", "simple", "ddim_uniform", "beta", "linear_quadratic", "kl_optimal", "AYS"]
|
||||
|
||||
comfy_samplers_mod.KSAMPLER = DummyKSAMPLER
|
||||
comfy_samplers_mod.KSampler = KSampler
|
||||
|
||||
comfy_model_base_mod = types.ModuleType("comfy.model_base")
|
||||
|
||||
|
||||
|
After Width: | Height: | Size: 199 KiB |
|
After Width: | Height: | Size: 219 KiB |
@@ -0,0 +1,976 @@
|
||||
{
|
||||
"id": "c4e8a1f7-2b93-4d6e-8a50-1f7e9c3b6d28",
|
||||
"revision": 0,
|
||||
"last_node_id": 14,
|
||||
"last_link_id": 19,
|
||||
"nodes": [
|
||||
{
|
||||
"id": 1,
|
||||
"type": "UNETLoader",
|
||||
"pos": [
|
||||
-1180,
|
||||
40
|
||||
],
|
||||
"size": [
|
||||
390,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 1,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "MODEL",
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
1
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "UNETLoader",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"qwen_image_2.1_int8_convrot.safetensors",
|
||||
"default"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"type": "CLIPLoader",
|
||||
"pos": [
|
||||
-1180,
|
||||
170
|
||||
],
|
||||
"size": [
|
||||
390,
|
||||
106
|
||||
],
|
||||
"flags": {},
|
||||
"order": 2,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "CLIP",
|
||||
"name": "CLIP",
|
||||
"type": "CLIP",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
3,
|
||||
4
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CLIPLoader",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"qwen3vl_8b_int8_convrot.safetensors",
|
||||
"qwen_image",
|
||||
"default"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"type": "VAELoader",
|
||||
"pos": [
|
||||
-1180,
|
||||
320
|
||||
],
|
||||
"size": [
|
||||
390,
|
||||
58
|
||||
],
|
||||
"flags": {},
|
||||
"order": 3,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "VAE",
|
||||
"name": "VAE",
|
||||
"type": "VAE",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
5,
|
||||
6,
|
||||
7
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "VAELoader",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"qwen_image_2.1_vae_bf16.safetensors"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"type": "QwenImage21Cache",
|
||||
"pos": [
|
||||
-740,
|
||||
40
|
||||
],
|
||||
"size": [
|
||||
310,
|
||||
82
|
||||
],
|
||||
"flags": {},
|
||||
"order": 4,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "model",
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 1
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "MODEL",
|
||||
"name": "MODEL",
|
||||
"type": "MODEL",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
2
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "QwenImage21Cache",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"auto",
|
||||
"default"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 5,
|
||||
"type": "LoadImage",
|
||||
"pos": [
|
||||
-1180,
|
||||
420
|
||||
],
|
||||
"size": [
|
||||
390,
|
||||
440
|
||||
],
|
||||
"flags": {},
|
||||
"order": 5,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
8
|
||||
]
|
||||
},
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"slot_index": 1,
|
||||
"links": [
|
||||
9
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImage",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Image_Load_Me_in_Loader.png",
|
||||
"image"
|
||||
],
|
||||
"title": "The picture (RGBA: the alpha is the background)"
|
||||
},
|
||||
{
|
||||
"id": 6,
|
||||
"type": "JoinImageWithAlpha",
|
||||
"pos": [
|
||||
-740,
|
||||
420
|
||||
],
|
||||
"size": [
|
||||
330,
|
||||
60
|
||||
],
|
||||
"flags": {},
|
||||
"order": 6,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 8
|
||||
},
|
||||
{
|
||||
"localized_name": "alpha",
|
||||
"name": "alpha",
|
||||
"type": "MASK",
|
||||
"link": 9
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
10,
|
||||
11,
|
||||
12
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "JoinImageWithAlpha",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [],
|
||||
"title": "Re-attach the alpha as the 4th channel"
|
||||
},
|
||||
{
|
||||
"id": 7,
|
||||
"type": "LoadImageMask",
|
||||
"pos": [
|
||||
-1180,
|
||||
900
|
||||
],
|
||||
"size": [
|
||||
390,
|
||||
480
|
||||
],
|
||||
"flags": {},
|
||||
"order": 7,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "MASK",
|
||||
"name": "MASK",
|
||||
"type": "MASK",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
13,
|
||||
14
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LoadImageMask",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Mask_Load_Me_in_Loader.png",
|
||||
"red"
|
||||
],
|
||||
"title": "The mask, on its own (white = rebuild)"
|
||||
},
|
||||
{
|
||||
"id": 8,
|
||||
"type": "TextEncodeQwenImageEditPlus",
|
||||
"pos": [
|
||||
-740,
|
||||
520
|
||||
],
|
||||
"size": [
|
||||
440,
|
||||
250
|
||||
],
|
||||
"flags": {},
|
||||
"order": 8,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "clip",
|
||||
"name": "clip",
|
||||
"type": "CLIP",
|
||||
"link": 3
|
||||
},
|
||||
{
|
||||
"localized_name": "prompt",
|
||||
"name": "prompt",
|
||||
"type": "STRING",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "prompt"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "vae",
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": 5
|
||||
},
|
||||
{
|
||||
"localized_name": "image1",
|
||||
"name": "image1",
|
||||
"type": "IMAGE",
|
||||
"link": 10
|
||||
},
|
||||
{
|
||||
"localized_name": "image2",
|
||||
"name": "image2",
|
||||
"type": "IMAGE",
|
||||
"link": null
|
||||
},
|
||||
{
|
||||
"localized_name": "image3",
|
||||
"name": "image3",
|
||||
"type": "IMAGE",
|
||||
"link": null
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "CONDITIONING",
|
||||
"name": "CONDITIONING",
|
||||
"type": "CONDITIONING",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
15
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "TextEncodeQwenImageEditPlus",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"the boot's sole and forefoot replaced by a sleek futuristic mechanical unit: segmented brushed titanium plating with fine panel lines, glowing cyan energy strips along the flanks, a sculpted dark carbon-fibre heel block, precise engineered details and tiny bolts, the brown leather upper above stays exactly as it is, high-end sci-fi product photography, transparent background"
|
||||
],
|
||||
"title": "Text Encode Qwen Image Edit Plus (reference + prompt)",
|
||||
"color": "#232",
|
||||
"bgcolor": "#353"
|
||||
},
|
||||
{
|
||||
"id": 9,
|
||||
"type": "CLIPTextEncode",
|
||||
"pos": [
|
||||
-740,
|
||||
800
|
||||
],
|
||||
"size": [
|
||||
440,
|
||||
150
|
||||
],
|
||||
"flags": {},
|
||||
"order": 9,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "clip",
|
||||
"name": "clip",
|
||||
"type": "CLIP",
|
||||
"link": 4
|
||||
},
|
||||
{
|
||||
"localized_name": "text",
|
||||
"name": "text",
|
||||
"type": "STRING",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "text"
|
||||
}
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "CONDITIONING",
|
||||
"name": "CONDITIONING",
|
||||
"type": "CONDITIONING",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
16
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "CLIPTextEncode",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"低分辨率,低画质,肢体畸形,手指畸形,画面过饱和,蜡像感,人脸无细节,过度光滑,画面具有AI感,模糊,变形"
|
||||
],
|
||||
"title": "CLIP Text Encode (Negative Prompt)",
|
||||
"color": "#223",
|
||||
"bgcolor": "#335"
|
||||
},
|
||||
{
|
||||
"id": 10,
|
||||
"type": "LanPaint_ImageEncode",
|
||||
"pos": [
|
||||
-240,
|
||||
40
|
||||
],
|
||||
"size": [
|
||||
300,
|
||||
110
|
||||
],
|
||||
"flags": {},
|
||||
"order": 10,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 11
|
||||
},
|
||||
{
|
||||
"localized_name": "vae",
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": 6
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 13
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "LATENT",
|
||||
"name": "LATENT",
|
||||
"type": "LATENT",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
17
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LanPaint_ImageEncode",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": []
|
||||
},
|
||||
{
|
||||
"id": 11,
|
||||
"type": "LanPaint_KSampler",
|
||||
"pos": [
|
||||
-240,
|
||||
190
|
||||
],
|
||||
"size": [
|
||||
330,
|
||||
320
|
||||
],
|
||||
"flags": {},
|
||||
"order": 11,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "model",
|
||||
"name": "model",
|
||||
"type": "MODEL",
|
||||
"link": 2
|
||||
},
|
||||
{
|
||||
"localized_name": "positive",
|
||||
"name": "positive",
|
||||
"type": "CONDITIONING",
|
||||
"link": 15
|
||||
},
|
||||
{
|
||||
"localized_name": "negative",
|
||||
"name": "negative",
|
||||
"type": "CONDITIONING",
|
||||
"link": 16
|
||||
},
|
||||
{
|
||||
"localized_name": "latent_image",
|
||||
"name": "latent_image",
|
||||
"type": "LATENT",
|
||||
"link": 17
|
||||
},
|
||||
{
|
||||
"localized_name": "seed",
|
||||
"name": "seed",
|
||||
"type": "INT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "seed"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "steps",
|
||||
"name": "steps",
|
||||
"type": "INT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "steps"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "cfg",
|
||||
"name": "cfg",
|
||||
"type": "FLOAT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "cfg"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "sampler_name",
|
||||
"name": "sampler_name",
|
||||
"type": "COMBO",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "sampler_name"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "scheduler",
|
||||
"name": "scheduler",
|
||||
"type": "COMBO",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "scheduler"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "denoise",
|
||||
"name": "denoise",
|
||||
"type": "FLOAT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "denoise"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "LanPaint_NumSteps",
|
||||
"name": "LanPaint_NumSteps",
|
||||
"type": "INT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "LanPaint_NumSteps"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "LanPaint_PromptMode",
|
||||
"name": "LanPaint_PromptMode",
|
||||
"type": "COMBO",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "LanPaint_PromptMode"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "LanPaint_Info",
|
||||
"name": "LanPaint_Info",
|
||||
"type": "STRING",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "LanPaint_Info"
|
||||
}
|
||||
},
|
||||
{
|
||||
"localized_name": "Inpainting_mode",
|
||||
"name": "Inpainting_mode",
|
||||
"type": "COMBO",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "Inpainting_mode"
|
||||
}
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "LATENT",
|
||||
"name": "LATENT",
|
||||
"type": "LATENT",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
18
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LanPaint_KSampler",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
777,
|
||||
"fixed",
|
||||
20,
|
||||
4.0,
|
||||
"euler",
|
||||
"simple",
|
||||
1.0,
|
||||
5,
|
||||
"Image First",
|
||||
"LanPaint KSampler. For more info, visit https://github.com/scraed/LanPaint. If you find it useful, please give a star ⭐!",
|
||||
"🖼️ Image Inpainting",
|
||||
"lanpaint_star_button"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 12,
|
||||
"type": "LanPaint_ImageDecode",
|
||||
"pos": [
|
||||
180,
|
||||
40
|
||||
],
|
||||
"size": [
|
||||
300,
|
||||
120
|
||||
],
|
||||
"flags": {},
|
||||
"order": 12,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "samples",
|
||||
"name": "samples",
|
||||
"type": "LATENT",
|
||||
"link": 18
|
||||
},
|
||||
{
|
||||
"localized_name": "vae",
|
||||
"name": "vae",
|
||||
"type": "VAE",
|
||||
"link": 7
|
||||
},
|
||||
{
|
||||
"localized_name": "image",
|
||||
"name": "image",
|
||||
"type": "IMAGE",
|
||||
"link": 12
|
||||
},
|
||||
{
|
||||
"localized_name": "mask",
|
||||
"name": "mask",
|
||||
"type": "MASK",
|
||||
"link": 14
|
||||
},
|
||||
{
|
||||
"localized_name": "blend_overlap",
|
||||
"name": "blend_overlap",
|
||||
"type": "INT",
|
||||
"link": null,
|
||||
"widget": {
|
||||
"name": "blend_overlap"
|
||||
}
|
||||
}
|
||||
],
|
||||
"outputs": [
|
||||
{
|
||||
"localized_name": "IMAGE",
|
||||
"name": "IMAGE",
|
||||
"type": "IMAGE",
|
||||
"slot_index": 0,
|
||||
"links": [
|
||||
19
|
||||
]
|
||||
}
|
||||
],
|
||||
"properties": {
|
||||
"Node name for S&R": "LanPaint_ImageDecode",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
9
|
||||
],
|
||||
"title": "Decode and merge (keeps RGBA)"
|
||||
},
|
||||
{
|
||||
"id": 13,
|
||||
"type": "SaveImage",
|
||||
"pos": [
|
||||
180,
|
||||
220
|
||||
],
|
||||
"size": [
|
||||
400,
|
||||
420
|
||||
],
|
||||
"flags": {},
|
||||
"order": 13,
|
||||
"mode": 0,
|
||||
"inputs": [
|
||||
{
|
||||
"localized_name": "images",
|
||||
"name": "images",
|
||||
"type": "IMAGE",
|
||||
"link": 19
|
||||
}
|
||||
],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "SaveImage",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"Qwen2.1_Transparent_Edit"
|
||||
]
|
||||
},
|
||||
{
|
||||
"id": 14,
|
||||
"type": "MarkdownNote",
|
||||
"pos": [
|
||||
-1180,
|
||||
1420
|
||||
],
|
||||
"size": [
|
||||
900,
|
||||
640
|
||||
],
|
||||
"flags": {},
|
||||
"order": 14,
|
||||
"mode": 0,
|
||||
"inputs": [],
|
||||
"outputs": [],
|
||||
"properties": {
|
||||
"Node name for S&R": "MarkdownNote",
|
||||
"cnr_id": "comfy-core"
|
||||
},
|
||||
"widgets_values": [
|
||||
"## Qwen-Image 2.1 + LanPaint, inpainting a picture with transparency\n\nThe source is RGBA and the edit is allowed to change the alpha, so the rebuilt sole gets\na new silhouette instead of just new pixels.\n\n`Join Image With Alpha` re-attaches the picture's alpha - `Load Image` hands it out on its\n`MASK` output rather than keeping it on the image. That mask is `1 - alpha` (1 =\ntransparent), which is what `Join Image With Alpha` wants, so it connects with no invert.\n\nThe mask is a separate greyscale file, white = repaint, and it is kept modest here (about\n16% of the frame). LanPaint anchors everything outside the mask to the original on every\nstep, so a mask covering most of the picture leaves the model without context."
|
||||
],
|
||||
"title": "How this works (and the mask size trap)",
|
||||
"color": "#432",
|
||||
"bgcolor": "#653"
|
||||
}
|
||||
],
|
||||
"links": [
|
||||
[
|
||||
1,
|
||||
1,
|
||||
0,
|
||||
4,
|
||||
0,
|
||||
"MODEL"
|
||||
],
|
||||
[
|
||||
2,
|
||||
4,
|
||||
0,
|
||||
11,
|
||||
0,
|
||||
"MODEL"
|
||||
],
|
||||
[
|
||||
3,
|
||||
2,
|
||||
0,
|
||||
8,
|
||||
0,
|
||||
"CLIP"
|
||||
],
|
||||
[
|
||||
4,
|
||||
2,
|
||||
0,
|
||||
9,
|
||||
0,
|
||||
"CLIP"
|
||||
],
|
||||
[
|
||||
5,
|
||||
3,
|
||||
0,
|
||||
8,
|
||||
2,
|
||||
"VAE"
|
||||
],
|
||||
[
|
||||
6,
|
||||
3,
|
||||
0,
|
||||
10,
|
||||
1,
|
||||
"VAE"
|
||||
],
|
||||
[
|
||||
7,
|
||||
3,
|
||||
0,
|
||||
12,
|
||||
1,
|
||||
"VAE"
|
||||
],
|
||||
[
|
||||
8,
|
||||
5,
|
||||
0,
|
||||
6,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
9,
|
||||
5,
|
||||
1,
|
||||
6,
|
||||
1,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
10,
|
||||
6,
|
||||
0,
|
||||
8,
|
||||
3,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
11,
|
||||
6,
|
||||
0,
|
||||
10,
|
||||
0,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
12,
|
||||
6,
|
||||
0,
|
||||
12,
|
||||
2,
|
||||
"IMAGE"
|
||||
],
|
||||
[
|
||||
13,
|
||||
7,
|
||||
0,
|
||||
10,
|
||||
2,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
14,
|
||||
7,
|
||||
0,
|
||||
12,
|
||||
3,
|
||||
"MASK"
|
||||
],
|
||||
[
|
||||
15,
|
||||
8,
|
||||
0,
|
||||
11,
|
||||
1,
|
||||
"CONDITIONING"
|
||||
],
|
||||
[
|
||||
16,
|
||||
9,
|
||||
0,
|
||||
11,
|
||||
2,
|
||||
"CONDITIONING"
|
||||
],
|
||||
[
|
||||
17,
|
||||
10,
|
||||
0,
|
||||
11,
|
||||
3,
|
||||
"LATENT"
|
||||
],
|
||||
[
|
||||
18,
|
||||
11,
|
||||
0,
|
||||
12,
|
||||
0,
|
||||
"LATENT"
|
||||
],
|
||||
[
|
||||
19,
|
||||
12,
|
||||
0,
|
||||
13,
|
||||
0,
|
||||
"IMAGE"
|
||||
]
|
||||
],
|
||||
"groups": [
|
||||
{
|
||||
"id": 1,
|
||||
"title": "Step 1 - Load models",
|
||||
"bounding": [
|
||||
-1190,
|
||||
0,
|
||||
420,
|
||||
400
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 2,
|
||||
"title": "Step 2 - Picture + its alpha + the mask",
|
||||
"bounding": [
|
||||
-1190,
|
||||
380,
|
||||
880,
|
||||
1020
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 3,
|
||||
"title": "Step 3 - Texts",
|
||||
"bounding": [
|
||||
-750,
|
||||
490,
|
||||
460,
|
||||
480
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
},
|
||||
{
|
||||
"id": 4,
|
||||
"title": "Step 4 - LanPaint: encode, sample, decode",
|
||||
"bounding": [
|
||||
-250,
|
||||
0,
|
||||
840,
|
||||
520
|
||||
],
|
||||
"color": "#3f789e",
|
||||
"font_size": 24,
|
||||
"flags": {}
|
||||
}
|
||||
],
|
||||
"config": {},
|
||||
"extra": {
|
||||
"ds": {
|
||||
"scale": 0.55,
|
||||
"offset": [
|
||||
1210,
|
||||
180
|
||||
]
|
||||
},
|
||||
"workflowRendererVersion": "LG"
|
||||
},
|
||||
"version": 0.4
|
||||
}
|
||||
|
Before Width: | Height: | Size: 3.8 MiB After Width: | Height: | Size: 4.3 MiB |
|
After Width: | Height: | Size: 793 KiB |
|
After Width: | Height: | Size: 1.2 MiB |
|
After Width: | Height: | Size: 1.0 MiB |
|
After Width: | Height: | Size: 2.6 KiB |
|
After Width: | Height: | Size: 1.6 MiB |
|
After Width: | Height: | Size: 965 KiB |
|
After Width: | Height: | Size: 1018 KiB |
|
After Width: | Height: | Size: 5.0 KiB |
|
After Width: | Height: | Size: 2.8 MiB |
|
After Width: | Height: | Size: 797 KiB |
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
||||
|
||||
[project]
|
||||
name = "LanPaint"
|
||||
version = "2.0.0"
|
||||
version = "2.1.0"
|
||||
description = "Achieve seamless inpainting results without needing a specialized inpainting model."
|
||||
authors = [
|
||||
{name = "LanPaint", email = "czhengac@connect.ust.hk"}
|
||||
|
||||
@@ -5,7 +5,7 @@ from .earlystop import LanPaintEarlyStopper
|
||||
from .types import LangevinState
|
||||
|
||||
class LanPaint():
|
||||
def __init__(self, Model, NSteps, Friction, Lambda, Beta, StepSize, IS_FLUX = False, IS_FLOW = False, EarlyStopThreshold = 0.0, EarlyStopPatience = 1, EarlyStopHook = None):
|
||||
def __init__(self, Model, NSteps, Friction, Lambda, Beta, StepSize, IS_FLUX = False, IS_FLOW = False, EarlyStopThreshold = 0.0, EarlyStopPatience = 1, EarlyStopHook = None, MinStepFrac = 0.0):
|
||||
self.n_steps = NSteps
|
||||
self.chara_lamb = Lambda
|
||||
self.IS_FLUX = IS_FLUX
|
||||
@@ -14,6 +14,7 @@ class LanPaint():
|
||||
self.inner_model = Model
|
||||
self.friction = Friction
|
||||
self.chara_beta = Beta
|
||||
self.min_step_frac = MinStepFrac
|
||||
self.img_dim_size = None
|
||||
self.early_stop_threshold = EarlyStopThreshold
|
||||
self.early_stop_patience = EarlyStopPatience
|
||||
@@ -72,7 +73,12 @@ class LanPaint():
|
||||
replace_sigma = sigma * (1 - ai) + Flow_a * ai
|
||||
current_times = (VE_Sigma, abt, Flow_t)
|
||||
|
||||
step_size = self.step_size * (1 - abt)
|
||||
# Above MinStepFrac the step size scales with the remaining noise
|
||||
# fraction (1 - abt); below it the step size is pinned at
|
||||
# StepSize*MinStepFrac and the inner-step count ramps down instead
|
||||
# (see KSamplerX0Inpaint.__call__). 0.0 disables the pin (the step
|
||||
# size keeps shrinking to zero as before).
|
||||
step_size = self.step_size * (1 - abt).clamp(min=self.min_step_frac)
|
||||
step_size = self.add_none_dims(step_size)
|
||||
# self.inner_model.inner_model.scale_latent_inpaint returns variance exploding x_t values
|
||||
# This is the replace step
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import json
|
||||
import os
|
||||
from contextlib import contextmanager
|
||||
from contextlib import contextmanager, nullcontext
|
||||
import math
|
||||
# import nodes.py
|
||||
import comfy
|
||||
@@ -10,6 +10,11 @@ import torch
|
||||
from comfy.utils import repeat_to_batch_size
|
||||
from comfy.samplers import *
|
||||
from comfy.model_base import ModelType
|
||||
|
||||
# MiniMaxH3 was ModelType.FLOW before ComfyUI 0.31.0 and ModelType.FLOW_AV
|
||||
# afterwards (the FLOW_AV enum + ModelSamplingAV audio carriage, commit
|
||||
# bdcb886a). Both are rectified-flow schedules; treat them alike.
|
||||
FLOW_MODEL_TYPES = (ModelType.FLOW, getattr(ModelType, "FLOW_AV", None))
|
||||
from .lanpaint import LanPaint
|
||||
from comfy.model_base import WAN22
|
||||
import comfyui_version
|
||||
@@ -51,6 +56,39 @@ def _version_tuple(value):
|
||||
|
||||
COMFYUI_VERSION_060_OR_NEWER = _version_tuple(comfyui_version.__version__) >= (0, 6, 0)
|
||||
|
||||
# ComfyUI >= 0.34 runs MiniMax H3 with per-token denoise-mask row timesteps
|
||||
# (commit ff6c8a8a). LanPaint keeps its 0.33 single-schedule contract, so this
|
||||
# gates hiding the mask from the model during the paint loop.
|
||||
COMFYUI_H3_DENOISE_MASK_CONTRACT = _version_tuple(getattr(comfyui_version, "__version__", "0.0.0")) >= (0, 34, 0)
|
||||
|
||||
@contextmanager
|
||||
def _hide_h3_denoise_mask(model):
|
||||
"""ComfyUI >= 0.34 hands MiniMax H3 a per-token denoise mask, which
|
||||
switches the DiT to mask-driven row timesteps (commit ff6c8a8a). LanPaint
|
||||
implements the 0.33 single-schedule contract in its own replace step and
|
||||
inner dynamics, so during the paint loop the mask is hidden from the
|
||||
model's extra_conds and the DiT keeps the uniform row timesteps that
|
||||
LanPaint's math expects."""
|
||||
had_instance_attr = "extra_conds" in getattr(model, "__dict__", {})
|
||||
original_extra_conds = model.extra_conds
|
||||
|
||||
def _extra_conds_without_mask(*args, **kwargs):
|
||||
kwargs.pop("denoise_mask", None)
|
||||
return original_extra_conds(*args, **kwargs)
|
||||
|
||||
model.extra_conds = _extra_conds_without_mask
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
if had_instance_attr:
|
||||
model.extra_conds = original_extra_conds
|
||||
else:
|
||||
try:
|
||||
del model.extra_conds
|
||||
except AttributeError:
|
||||
model.extra_conds = original_extra_conds
|
||||
|
||||
|
||||
def reshape_mask(input_mask, output_shape,video_inpainting=False):
|
||||
dims = len(output_shape) - 2
|
||||
scale_mode = "nearest-exact"
|
||||
@@ -67,9 +105,9 @@ def reshape_mask(input_mask, output_shape,video_inpainting=False):
|
||||
elif input_mask.ndim == 2:
|
||||
input_mask = input_mask.unsqueeze(0).unsqueeze(0).unsqueeze(0)
|
||||
elif input_mask.ndim == 1 and len(output_shape) == 4:
|
||||
# audio mask [F] at video frame rate -> the audio latent's token grid:
|
||||
# nearest-exact along time, then expanded to the (ch, tokens) layout
|
||||
f, t = input_mask.shape[0], output_shape[-1]
|
||||
# audio mask [F] at video frame rate -> the audio latent's time axis:
|
||||
# nearest-exact along time, then expanded to the (ch, T) layout
|
||||
t = output_shape[-1]
|
||||
input_mask = torch.nn.functional.interpolate(
|
||||
input_mask.float().unsqueeze(0).unsqueeze(0),
|
||||
size=(t,),
|
||||
@@ -79,7 +117,7 @@ def reshape_mask(input_mask, output_shape,video_inpainting=False):
|
||||
elif input_mask.ndim == 4 and len(output_shape) == 4 and input_mask.shape[1] == 1 and input_mask.shape[3] == 1:
|
||||
# audio mask [F, 1] that SetLatentNoiseMask reshaped to [1, 1, F, 1]:
|
||||
# nearest F -> T, then expanded to (1, 1, ch, T)
|
||||
f, t = input_mask.shape[2], output_shape[-1]
|
||||
t = output_shape[-1]
|
||||
input_mask = torch.nn.functional.interpolate(input_mask, size=(t, 1), mode="nearest-exact")
|
||||
input_mask = input_mask.permute(0, 1, 3, 2).expand(1, 1, output_shape[-2], t)
|
||||
elif input_mask.ndim == 2:
|
||||
@@ -94,35 +132,22 @@ def reshape_mask(input_mask, output_shape,video_inpainting=False):
|
||||
|
||||
if video_inpainting: # Video case: (batch, channels, frames, height, width)
|
||||
|
||||
# Temporal union: a latent token covers ~4 video frames (the VAE's
|
||||
# nominal temporal stride), and a token-level mask is all-or-nothing
|
||||
# (binarized at 0.5 downstream). Averaging the frames into a token
|
||||
# (trilinear) can erase sparse temporal strokes entirely; instead the
|
||||
# token takes the UNION (max) of its ~4 frames - it regenerates iff
|
||||
# any of them is painted. The last window is partial; short sequences
|
||||
# (1-4 frames) collapse to a single window holding the max, which is
|
||||
# exactly the union.
|
||||
f = input_mask.shape[2]
|
||||
n_win = (f + 3) // 4
|
||||
pooled = torch.zeros(
|
||||
(input_mask.shape[0], input_mask.shape[1], n_win, input_mask.shape[3], input_mask.shape[4]),
|
||||
dtype=input_mask.dtype, device=input_mask.device)
|
||||
for i in range(n_win):
|
||||
pooled[:, :, i] = input_mask[:, :, i * 4 : (i + 1) * 4].amax(dim=2)
|
||||
input_mask = pooled
|
||||
|
||||
target_frames = output_shape[2]
|
||||
target_height, target_width = output_shape[-2:]
|
||||
|
||||
# 3D nearest-exact interpolation: (batch, channels, frames, height, width) -> (batch, channels, target_frames, target_height, target_width)
|
||||
temp_mask = torch.nn.functional.interpolate(
|
||||
# EXPERIMENTAL: nearest-exact first, then the slice-level union.
|
||||
# The resample snaps each latent slice to a single picked frame
|
||||
# (src = floor((t+0.5)*F/T), ~1 in 4 frames), then the sliding max
|
||||
# over ~5 slices spreads the value to neighboring slices. NOTE: this
|
||||
# drops strokes painted on frames that no slice picks.
|
||||
input_mask = torch.nn.functional.interpolate(
|
||||
input_mask,
|
||||
size=(target_frames, target_height, target_width),
|
||||
mode=scale_mode,
|
||||
)
|
||||
|
||||
# temp_mask is already 5D: (batch, channels, target_frames, target_height, target_width)
|
||||
mask = temp_mask
|
||||
mask = torch.nn.functional.max_pool3d(
|
||||
input_mask, kernel_size=(5, 1, 1), stride=(1, 1, 1), padding=(2, 0, 0))
|
||||
# Handle channel dimension expansion if needed
|
||||
if mask.shape[1] < output_shape[1]:
|
||||
mask = mask.repeat(1, output_shape[1], 1, 1, 1)[:, :output_shape[1]]
|
||||
@@ -139,6 +164,31 @@ def reshape_mask(input_mask, output_shape,video_inpainting=False):
|
||||
|
||||
|
||||
return mask
|
||||
def min_step_frac_effective_steps(n_steps, frac, min_frac):
|
||||
"""Inner-step count under the MinStepFrac tail ramp.
|
||||
|
||||
While the remaining-noise fraction stays above min_frac (or with
|
||||
min_frac == 0, the feature disabled) the count is unchanged and the step
|
||||
size keeps scaling with the noise inside LanPaint. Below the fraction the
|
||||
step size is pinned and the count ramps down linearly,
|
||||
NSteps * frac / min_frac, to zero."""
|
||||
if min_frac <= 0 or frac >= min_frac or n_steps <= 0:
|
||||
return n_steps
|
||||
return max(0, round(n_steps * frac / min_frac))
|
||||
|
||||
def _sanitize_param(value, default, allowed=None):
|
||||
"""Coerce a widget value to a sane default.
|
||||
|
||||
Old workflows and hand-edited JSON can supply values that no longer fit
|
||||
the widget (wrong type, or a value removed from a combo list). Fail back
|
||||
to the default instead of crashing the node.
|
||||
"""
|
||||
if allowed is not None:
|
||||
return value if value in allowed else default
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
return default
|
||||
return value
|
||||
|
||||
def prepare_mask(noise_mask, shape, device,video_inpainting=False):
|
||||
return reshape_mask(noise_mask, shape,video_inpainting).to(device)
|
||||
def sampling_function_LanPaint(model, x, timestep, uncond, cond, cond_scale, cond_scale_BIG, model_options={}, seed=None):
|
||||
@@ -173,6 +223,13 @@ class CFGGuider_LanPaint:
|
||||
self.minimax_h3_audio = _detect_minimax_h3_audio(
|
||||
self.model_patcher, self.model_options, kwargs.get("latent_shapes", None))
|
||||
|
||||
# ComfyUI >= 0.34 switches MiniMax H3 to per-token row timesteps driven
|
||||
# by the denoise mask; LanPaint's replace step and inner dynamics follow
|
||||
# the 0.33 single-schedule contract, so hide the mask from the model for
|
||||
# the paint loop and keep the row timesteps uniform.
|
||||
h3_hide_mask = self.minimax_h3_audio is not None and COMFYUI_H3_DENOISE_MASK_CONTRACT
|
||||
hide_mask_ctx = _hide_h3_denoise_mask(self.inner_model) if h3_hide_mask else nullcontext()
|
||||
|
||||
if denoise_mask is not None:
|
||||
video_inpainting = self.model_options.get("video_inpainting", False)
|
||||
if tuple(denoise_mask.shape) != tuple(noise.shape):
|
||||
@@ -187,7 +244,8 @@ class CFGGuider_LanPaint:
|
||||
|
||||
try:
|
||||
self.model_patcher.pre_run()
|
||||
output = self.inner_sample(noise, latent_image, device, sampler, sigmas, denoise_mask, callback, disable_pbar, seed, **kwargs)
|
||||
with hide_mask_ctx:
|
||||
output = self.inner_sample(noise, latent_image, device, sampler, sigmas, denoise_mask, callback, disable_pbar, seed, **kwargs)
|
||||
finally:
|
||||
self.model_patcher.cleanup()
|
||||
|
||||
@@ -217,7 +275,7 @@ class KSamplerX0Inpaint:
|
||||
# x is rectified flow x_t = sigma * noise + (1.0 - sigma) * x_0
|
||||
|
||||
IS_FLUX = self.inner_model.inner_model.model_type == ModelType.FLUX
|
||||
IS_FLOW = self.inner_model.inner_model.model_type == ModelType.FLOW
|
||||
IS_FLOW = self.inner_model.inner_model.model_type in FLOW_MODEL_TYPES
|
||||
#print("model class", type(self.inner_model.inner_model))
|
||||
#print("model type", self.inner_model.inner_model.model_type, "IS_FLUX", IS_FLUX, "IS_FLOW", IS_FLOW)
|
||||
#print("sigma", torch.mean(sigma).item(), torch.min(sigma).item(), torch.max(sigma).item())
|
||||
@@ -269,10 +327,18 @@ class KSamplerX0Inpaint:
|
||||
current_step = torch.argmin( torch.abs( self.sigmas - torch.mean(sigma) ) )
|
||||
total_steps = len(self.sigmas)-1
|
||||
|
||||
# Only one knob changes at a time: while the remaining noise
|
||||
# fraction (1 - abt) stays above LanPaint_MinStepFrac the inner
|
||||
# step size scales with it (LanPaint pins it below the fraction)
|
||||
# and the count is constant. Below the fraction the count ramps
|
||||
# down linearly, NSteps * (1 - abt) / MinStepFrac, down to 0.
|
||||
n_eff = self.PaintMethod.n_steps
|
||||
if total_steps - current_step <= self.LanPaint_early_stop:
|
||||
out = self.PaintMethod(x, self.latent_image, self.noise, sigma, latent_mask, current_times, model_options, seed, n_steps=0, current_times_audio=current_times_audio, audio_indicator=self.audio_indicator, audio_correction=audio_correction)
|
||||
n_eff = 0
|
||||
else:
|
||||
out = self.PaintMethod(x, self.latent_image, self.noise, sigma, latent_mask, current_times, model_options, seed, current_times_audio=current_times_audio, audio_indicator=self.audio_indicator, audio_correction=audio_correction)
|
||||
n_eff = min_step_frac_effective_steps(
|
||||
n_eff, float((1.0 - abt).mean()), getattr(self, "LanPaint_min_step_frac", 1.0))
|
||||
out = self.PaintMethod(x, self.latent_image, self.noise, sigma, latent_mask, current_times, model_options, seed, n_steps=n_eff, current_times_audio=current_times_audio, audio_indicator=self.audio_indicator, audio_correction=audio_correction)
|
||||
else:
|
||||
out, _ = self.inner_model(x, sigma, model_options=model_options, seed=seed)
|
||||
|
||||
@@ -304,7 +370,7 @@ class KSAMPLER(comfy.samplers.KSAMPLER):
|
||||
model_k.noise = noise
|
||||
|
||||
IS_FLUX = model_wrap.inner_model.model_type == ModelType.FLUX
|
||||
IS_FLOW = model_wrap.inner_model.model_type == ModelType.FLOW
|
||||
IS_FLOW = model_wrap.inner_model.model_type in FLOW_MODEL_TYPES
|
||||
# unify the notations into variance exploding diffusion model
|
||||
if IS_FLUX:
|
||||
model_wrap.cfg_BIG = 1.0
|
||||
@@ -333,8 +399,10 @@ class KSAMPLER(comfy.samplers.KSAMPLER):
|
||||
IS_FLOW = IS_FLOW,
|
||||
EarlyStopThreshold = getattr(model_wrap.model_patcher, "LanPaint_InnerThreshold", 0.0),
|
||||
EarlyStopPatience = getattr(model_wrap.model_patcher, "LanPaint_InnerPatience", 1),
|
||||
EarlyStopHook = extra_args.get("model_options", {}).get("lanpaint_semantic_hook", None))
|
||||
EarlyStopHook = extra_args.get("model_options", {}).get("lanpaint_semantic_hook", None),
|
||||
MinStepFrac = getattr(model_wrap.model_patcher, "LanPaint_MinStepFrac", 1.0))
|
||||
model_k.LanPaint_early_stop = model_wrap.model_patcher.LanPaint_EarlyStop
|
||||
model_k.LanPaint_min_step_frac = getattr(model_wrap.model_patcher, "LanPaint_MinStepFrac", 1.0)
|
||||
#if not inpainting, after noise_scaling, noise = noise * sigma, which is the noise added to the clean latent image in the variance exploding diffusion model notation.
|
||||
#if inpainting, after noise_scaling, noise = latent_image + noise * sigma, which is x_t in the variance exploding diffusion model notation for the known region.
|
||||
k_callback = None
|
||||
@@ -441,7 +509,13 @@ class LanPaint_KSampler():
|
||||
"LanPaint_PromptMode": (["Image First", "Prompt First"], {"tooltip": "Image First: emphasis image quality, Prompt First: emphasis prompt following"}),
|
||||
"LanPaint_Info": ("STRING", {"default": "LanPaint KSampler.", "tooltip": "For more info, visit https://github.com/scraed/LanPaint. If you find it useful, please give a star ⭐️!"}),
|
||||
"Inpainting_mode": (["🖼️ Image Inpainting", "🎬 Video Inpainting"], {"default": "🖼️ Image Inpainting", "tooltip": "Choose Image mode for photos or Video mode for video frames with temporal consistency"}),
|
||||
}
|
||||
},
|
||||
"hidden": {
|
||||
# Retired hyperparameters. Old prompts (saved before these
|
||||
# were removed) still pass them; accept and ignore the
|
||||
# values -- the fixed defaults below are always used.
|
||||
"LanPaint_MinStepFrac": "DEFAULT",
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("LATENT",)
|
||||
@@ -451,12 +525,16 @@ class LanPaint_KSampler():
|
||||
CATEGORY = "sampling"
|
||||
DESCRIPTION = "Uses the provided model, positive and negative conditioning to denoise the latent image."
|
||||
|
||||
def sample(self, model, seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, denoise=1.0, LanPaint_NumSteps=5, LanPaint_PromptMode="Image First", LanPaint_Info="",Inpainting_mode="🖼️ Image Inpainting"):
|
||||
def sample(self, model, seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, denoise=1.0, LanPaint_NumSteps=5, LanPaint_PromptMode="Image First", LanPaint_Info="",Inpainting_mode="🖼️ Image Inpainting", **kwargs):
|
||||
LanPaint_NumSteps = _sanitize_param(LanPaint_NumSteps, 5)
|
||||
LanPaint_PromptMode = _sanitize_param(LanPaint_PromptMode, "Image First", allowed=("Image First", "Prompt First"))
|
||||
Inpainting_mode = _sanitize_param(Inpainting_mode, "🖼️ Image Inpainting", allowed=("🖼️ Image Inpainting", "🎬 Video Inpainting"))
|
||||
|
||||
model.LanPaint_StepSize = 0.2
|
||||
model.LanPaint_Lambda = 5.0
|
||||
model.LanPaint_Beta = 1.
|
||||
model.LanPaint_NumSteps = LanPaint_NumSteps
|
||||
model.LanPaint_MinStepFrac = 1.0
|
||||
model.LanPaint_Friction = 15.
|
||||
model.LanPaint_EarlyStop = 1
|
||||
model.LanPaint_InnerThreshold = 0.0
|
||||
@@ -494,15 +572,21 @@ class LanPaint_KSamplerAdvanced:
|
||||
"LanPaint_NumSteps": ("INT", {"default": 5, "min": 0, "max": 100, "tooltip": "The number of steps for the Langevin dynamics, representing the turns of thinking per step."}),
|
||||
"LanPaint_Lambda": ("FLOAT", {"default": 5.0, "min": 0.1, "max": 50.0, "step": 0.1, "round": 0.1, "tooltip": "The bidirectional guidance scale. Higher values align with known regions more closely, but may result in instability."}),
|
||||
"LanPaint_StepSize": ("FLOAT", {"default": 0.2, "min": 0.0001, "max": 1., "step": 0.01, "round": 0.001, "tooltip": "The step size for the Langevin dynamics. Higher values result in faster convergence but may be unstable."}),
|
||||
"LanPaint_Beta": ("FLOAT", {"default": 1., "min": 0.0001, "max": 5, "step": 0.1, "round": 0.1, "tooltip": "The step size ratio between masked / unmasked regions. Lower value can compensate high values of LanPaint_Lambda."}),
|
||||
"LanPaint_Friction": ("FLOAT", {"default": 15, "min": 0., "max": 50.0, "step": 0.1, "round": 0.1, "tooltip": "The friction parameter for fast langevin, lower values result in faster convergence but may be unstable."}),
|
||||
"LanPaint_PromptMode": (["Image First", "Prompt First"], {"tooltip": "Image First: emphasis image quality, Prompt First: emphasis prompt following"}),
|
||||
"LanPaint_EarlyStop": ("INT", {"default": 1, "min": 0, "max": 10000, "tooltip": "The number of steps to stop the LanPaint early, useful for preventing the image from irregular patterns."}),
|
||||
"LanPaint_Info": ("STRING", {"default": "LanPaint KSampler Adv.", "tooltip": "For more info, visit https://github.com/scraed/LanPaint. If you find it useful, please give a star ⭐️!"}),
|
||||
"Inpainting_mode": (["🖼️ Image Inpainting", "🎬 Video Inpainting"], {"default": "🖼️ Image Inpainting", "tooltip": "Choose Image mode for photos or Video mode for video frames with temporal consistency"}),
|
||||
"LanPaint_InnerThreshold": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.0001, "round": 0.0001, "tooltip": "Early stop threshold for Langevin iterations based on semantic distance. 0.0 to disable. (Contributed by godnight10061)"}),
|
||||
"LanPaint_InnerPatience": ("INT", {"default": 1, "min": 1, "max": 100, "tooltip": "Number of consecutive steps below threshold required to stop. (Contributed by godnight10061)"}),
|
||||
},
|
||||
"hidden": {
|
||||
# Retired hyperparameters. Old prompts (saved before these
|
||||
# were removed) still pass them; accept and ignore the
|
||||
# values -- the fixed defaults below are always used.
|
||||
"LanPaint_Beta": "DEFAULT",
|
||||
"LanPaint_Friction": "DEFAULT",
|
||||
"LanPaint_EarlyStop": "DEFAULT",
|
||||
"LanPaint_InnerThreshold": "DEFAULT",
|
||||
"LanPaint_InnerPatience": "DEFAULT",
|
||||
"LanPaint_MinStepFrac": "DEFAULT",
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("LATENT",)
|
||||
@@ -510,21 +594,27 @@ class LanPaint_KSamplerAdvanced:
|
||||
|
||||
CATEGORY = "sampling"
|
||||
|
||||
def sample(self, model, add_noise, noise_seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, start_at_step, end_at_step, return_with_leftover_noise, LanPaint_NumSteps=5, LanPaint_Lambda=5.0, LanPaint_StepSize=0.2, LanPaint_Beta=1.0, LanPaint_Friction=15.0, LanPaint_PromptMode="Image First", LanPaint_EarlyStop=1, LanPaint_Info="", Inpainting_mode="🖼️ Image Inpainting", LanPaint_InnerThreshold=0.0, LanPaint_InnerPatience=1):
|
||||
def sample(self, model, add_noise, noise_seed, steps, cfg, sampler_name, scheduler, positive, negative, latent_image, start_at_step, end_at_step, return_with_leftover_noise, LanPaint_NumSteps=5, LanPaint_Lambda=5.0, LanPaint_StepSize=0.2, LanPaint_PromptMode="Image First", LanPaint_Info="", Inpainting_mode="🖼️ Image Inpainting", **kwargs):
|
||||
force_full_denoise = True
|
||||
if return_with_leftover_noise == "enable":
|
||||
force_full_denoise = False
|
||||
disable_noise = False
|
||||
if add_noise == "disable":
|
||||
disable_noise = True
|
||||
LanPaint_NumSteps = _sanitize_param(LanPaint_NumSteps, 5)
|
||||
LanPaint_Lambda = _sanitize_param(LanPaint_Lambda, 5.0)
|
||||
LanPaint_StepSize = _sanitize_param(LanPaint_StepSize, 0.2)
|
||||
LanPaint_PromptMode = _sanitize_param(LanPaint_PromptMode, "Image First", allowed=("Image First", "Prompt First"))
|
||||
Inpainting_mode = _sanitize_param(Inpainting_mode, "🖼️ Image Inpainting", allowed=("🖼️ Image Inpainting", "🎬 Video Inpainting"))
|
||||
model.LanPaint_StepSize = LanPaint_StepSize
|
||||
model.LanPaint_Lambda = LanPaint_Lambda
|
||||
model.LanPaint_Beta = LanPaint_Beta
|
||||
model.LanPaint_Beta = 1.0
|
||||
model.LanPaint_NumSteps = LanPaint_NumSteps
|
||||
model.LanPaint_Friction = LanPaint_Friction
|
||||
model.LanPaint_EarlyStop = LanPaint_EarlyStop
|
||||
model.LanPaint_InnerThreshold = LanPaint_InnerThreshold
|
||||
model.LanPaint_InnerPatience = LanPaint_InnerPatience
|
||||
model.LanPaint_MinStepFrac = 1.0
|
||||
model.LanPaint_Friction = 15.0
|
||||
model.LanPaint_EarlyStop = 1
|
||||
model.LanPaint_InnerThreshold = 0.0
|
||||
model.LanPaint_InnerPatience = 1
|
||||
if LanPaint_PromptMode == "Image First":
|
||||
model.LanPaint_cfg_BIG = cfg
|
||||
else:
|
||||
@@ -632,10 +722,13 @@ class LanPaint_SamplerCustom:
|
||||
CATEGORY = "sampling/custom_sampling"
|
||||
|
||||
def sample(self, model, sampler, sigmas, add_noise, noise_seed, cfg, positive, negative, latent_image, LanPaint_NumSteps, LanPaint_PromptMode, LanPaint_Info=""):
|
||||
LanPaint_NumSteps = _sanitize_param(LanPaint_NumSteps, 5)
|
||||
LanPaint_PromptMode = _sanitize_param(LanPaint_PromptMode, "Image First", allowed=("Image First", "Prompt First"))
|
||||
model.LanPaint_StepSize = 0.2
|
||||
model.LanPaint_Lambda = 5.0
|
||||
model.LanPaint_Beta = 1.
|
||||
model.LanPaint_NumSteps = LanPaint_NumSteps
|
||||
model.LanPaint_MinStepFrac = 1.0
|
||||
model.LanPaint_Friction = 15.
|
||||
model.LanPaint_EarlyStop = 1
|
||||
model.LanPaint_InnerThreshold = 0.0
|
||||
@@ -686,14 +779,20 @@ class LanPaint_SamplerCustomAdvanced:
|
||||
"LanPaint_NumSteps": ("INT", {"default": 5, "min": 0, "max": 100, "tooltip": "Number of steps for Langevin dynamics, representing turns of thinking per step."}),
|
||||
"LanPaint_Lambda": ("FLOAT", {"default": 5.0, "min": 0.1, "max": 50.0, "step": 0.1, "tooltip": "Bidirectional guidance scale. Higher values align with known regions but may cause instability."}),
|
||||
"LanPaint_StepSize": ("FLOAT", {"default": 0.2, "min": 0.0001, "max": 1.0, "step": 0.01, "tooltip": "Step size for Langevin dynamics. Higher values speed convergence but may be unstable."}),
|
||||
"LanPaint_Beta": ("FLOAT", {"default": 1.0, "min": 0.0001, "max": 5.0, "step": 0.1, "tooltip": "Step size ratio between masked/unmasked regions. Lower values balance high Lambda."}),
|
||||
"LanPaint_Friction": ("FLOAT", {"default": 15.0, "min": 0.0, "max": 50.0, "step": 0.1, "tooltip": "Friction parameter for fast Langevin. Lower values speed convergence but may be unstable."}),
|
||||
"LanPaint_PromptMode": (["Image First", "Prompt First"], {"tooltip": "Image First: prioritizes image quality; Prompt First: prioritizes prompt adherence."}),
|
||||
"LanPaint_EarlyStop": ("INT", {"default": 1, "min": 0, "max": 10000, "tooltip": "Steps to stop LanPaint early, preventing irregular patterns."}),
|
||||
"LanPaint_Info": ("STRING", {"default": "LanPaint Custom Sampler Adv.", "tooltip": "For more info, visit https://github.com/scraed/LanPaint. If you find it useful, please give a star ⭐️!"}),
|
||||
"LanPaint_InnerThreshold": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.0001, "round": 0.0001, "tooltip": "Early stop threshold for Langevin iterations based on semantic distance. 0.0 to disable. (Contributed by godnight10061)"}),
|
||||
"LanPaint_InnerPatience": ("INT", {"default": 1, "min": 1, "max": 100, "tooltip": "Number of consecutive steps below threshold required to stop. (Contributed by godnight10061)"}),
|
||||
}
|
||||
},
|
||||
"hidden": {
|
||||
# Retired hyperparameters. Old prompts (saved before these
|
||||
# were removed) still pass them; accept and ignore the
|
||||
# values -- the fixed defaults below are always used.
|
||||
"LanPaint_Beta": "DEFAULT",
|
||||
"LanPaint_Friction": "DEFAULT",
|
||||
"LanPaint_EarlyStop": "DEFAULT",
|
||||
"LanPaint_InnerThreshold": "DEFAULT",
|
||||
"LanPaint_InnerPatience": "DEFAULT",
|
||||
"LanPaint_MinStepFrac": "DEFAULT",
|
||||
},
|
||||
}
|
||||
|
||||
RETURN_TYPES = ("LATENT","LATENT")
|
||||
@@ -703,16 +802,21 @@ class LanPaint_SamplerCustomAdvanced:
|
||||
|
||||
CATEGORY = "sampling/custom_sampling"
|
||||
|
||||
def sample(self, noise, guider, sampler, sigmas, latent_image, LanPaint_NumSteps, LanPaint_Lambda, LanPaint_StepSize, LanPaint_Beta, LanPaint_Friction, LanPaint_PromptMode, LanPaint_EarlyStop, LanPaint_Info="", LanPaint_InnerThreshold=0.0, LanPaint_InnerPatience=1):
|
||||
def sample(self, noise, guider, sampler, sigmas, latent_image, LanPaint_NumSteps, LanPaint_Lambda, LanPaint_StepSize, LanPaint_PromptMode, LanPaint_Info="", **kwargs):
|
||||
LanPaint_NumSteps = _sanitize_param(LanPaint_NumSteps, 5)
|
||||
LanPaint_Lambda = _sanitize_param(LanPaint_Lambda, 5.0)
|
||||
LanPaint_StepSize = _sanitize_param(LanPaint_StepSize, 0.2)
|
||||
LanPaint_PromptMode = _sanitize_param(LanPaint_PromptMode, "Image First", allowed=("Image First", "Prompt First"))
|
||||
model = guider.model_patcher
|
||||
model.LanPaint_StepSize = LanPaint_StepSize
|
||||
model.LanPaint_Lambda = LanPaint_Lambda
|
||||
model.LanPaint_Beta = LanPaint_Beta
|
||||
model.LanPaint_Beta = 1.0
|
||||
model.LanPaint_NumSteps = LanPaint_NumSteps
|
||||
model.LanPaint_Friction = LanPaint_Friction
|
||||
model.LanPaint_EarlyStop = LanPaint_EarlyStop
|
||||
model.LanPaint_InnerThreshold = LanPaint_InnerThreshold
|
||||
model.LanPaint_InnerPatience = LanPaint_InnerPatience
|
||||
model.LanPaint_Friction = 15.0
|
||||
model.LanPaint_EarlyStop = 1
|
||||
model.LanPaint_InnerThreshold = 0.0
|
||||
model.LanPaint_InnerPatience = 1
|
||||
model.LanPaint_MinStepFrac = 1.0
|
||||
if LanPaint_PromptMode == "Image First":
|
||||
model.LanPaint_cfg_BIG = guider.cfg
|
||||
else:
|
||||
@@ -840,7 +944,7 @@ class LanPaint_VideoMaskEditor:
|
||||
recorded in the hidden ``audio_mask`` widget as [{"start": s, "end": e}].
|
||||
The same editor displays the audio waveform and paints intervals; the
|
||||
workflow attaches the mask to the audio latent (SetLatentNoiseMask) and the
|
||||
sampler resamples it to the audio latent tokens.
|
||||
sampler resamples it to the audio latent's time axis.
|
||||
|
||||
Video mask convention: 1 = regenerate, 0 = keep. Both masks span the
|
||||
video's full frame count (read from the file container, no pixel decoding).
|
||||
@@ -973,7 +1077,7 @@ class LanPaint_AVEncode:
|
||||
z_audio = LanPaint_MiniMaxAudioEncode().encode(audio, audio_vae)[0]["samples"]
|
||||
|
||||
# audio mask: accept [F] or [F, 1]; the per-stream prep resamples it
|
||||
# to the audio latent tokens
|
||||
# to the audio latent's time axis
|
||||
if audio_mask.ndim == 2 and audio_mask.shape[1] == 1:
|
||||
audio_mask = audio_mask[:, 0]
|
||||
|
||||
@@ -1216,7 +1320,7 @@ class LanPaint_ImageEncode:
|
||||
m.unsqueeze(0).unsqueeze(0), size=target, mode="nearest-exact"
|
||||
)[0, 0]
|
||||
latent["noise_mask"] = m.unsqueeze(0).unsqueeze(0)
|
||||
else: # 5D video-style VAE: snap to (T, H, W), one mask frame per token
|
||||
else: # 5D video-style VAE: snap to (T, H, W), one mask slice per latent frame
|
||||
t, h, w = z.shape[-3:]
|
||||
if tuple(m.shape) != (h, w):
|
||||
m = torch.nn.functional.interpolate(
|
||||
@@ -1266,6 +1370,18 @@ class LanPaint_ImageDecode:
|
||||
)
|
||||
if image is None:
|
||||
return (img,)
|
||||
# Some VAEs decode to more channels than the source image: the Qwen Image
|
||||
# 2.1 VAE always emits a 4th channel for RGB input. ComfyUI's IMAGE type is
|
||||
# RGB, so trim to the original's channel count - which is what ComfyUI's own
|
||||
# RGBA -> RGB conversion does, compositing over white - and the merge below
|
||||
# can broadcast. Without this it raises
|
||||
# "The size of tensor a (3) must match the size of tensor b (4)".
|
||||
img_channels, orig_channels = img.shape[-1], image.shape[-1]
|
||||
if img_channels > orig_channels:
|
||||
img = img[..., :orig_channels]
|
||||
elif img_channels < orig_channels:
|
||||
pad = img.new_ones(img.shape[:-1] + (orig_channels - img_channels,))
|
||||
img = torch.cat((img, pad), dim=-1)
|
||||
target_h, target_w = image.shape[1], image.shape[2]
|
||||
if tuple(img.shape[1:3]) != (target_h, target_w):
|
||||
img = torch.nn.functional.interpolate(
|
||||
@@ -1276,6 +1392,10 @@ class LanPaint_ImageDecode:
|
||||
).movedim(1, -1)
|
||||
if mask is None:
|
||||
return (img,)
|
||||
# Keep the merge channel-agnostic: an RGBA image (a source that carries its
|
||||
# own alpha, e.g. a Qwen Image 2.1 transparent-background render) keeps its
|
||||
# 4th channel all the way through, so transparency can be inpainted alongside
|
||||
# the content. An RGB image still comes back as RGB.
|
||||
return (merge_video_with_mask(image, img, mask, blend_overlap),)
|
||||
|
||||
|
||||
|
||||
@@ -322,3 +322,26 @@ def test_prepare_step_size_handles_per_row_parameters() -> None:
|
||||
adt = (A_x * dtx).flatten()
|
||||
assert adt[0] == pytest.approx(0.2) # 1/(1-0.5) * 0.1
|
||||
assert adt[-1] == pytest.approx(0.2) # 1/(1-0.9) * 0.02 -- bounded invariant
|
||||
|
||||
|
||||
def test_hide_h3_denoise_mask_strips_and_restores(monkeypatch) -> None:
|
||||
# ComfyUI >= 0.34: the per-token denoise mask switches the MiniMax H3 DiT
|
||||
# to mask-driven row timesteps. LanPaint hides it from extra_conds during
|
||||
# the paint loop so the DiT keeps the uniform 0.33 row timesteps.
|
||||
nodes = _import_nodes(monkeypatch)
|
||||
captured = {}
|
||||
|
||||
class FakeH3Model:
|
||||
def extra_conds(self, **kwargs): # type: ignore[no-untyped-def]
|
||||
captured.update(kwargs)
|
||||
return {}
|
||||
|
||||
model = FakeH3Model()
|
||||
with nodes._hide_h3_denoise_mask(model):
|
||||
model.extra_conds(denoise_mask="MASK", latent_shapes=[(1, 2), (1, 2)])
|
||||
assert "denoise_mask" not in captured # hidden from the model
|
||||
assert captured["latent_shapes"] is not None # other conds still pass
|
||||
|
||||
captured.clear()
|
||||
model.extra_conds(denoise_mask="MASK", latent_shapes=[(1, 2), (1, 2)])
|
||||
assert captured["denoise_mask"] == "MASK" # restored after the loop
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
"""Tests for the MinStepFrac tail ramp (inner-step count reduction).
|
||||
|
||||
The ramp makes sure only one knob changes at a time: while the remaining
|
||||
noise fraction (1 - abt) is above the threshold the step size scales inside
|
||||
LanPaint and the count is constant; below it the step size is pinned and
|
||||
the count ramps down linearly.
|
||||
"""
|
||||
|
||||
|
||||
def _import_nodes():
|
||||
import LanPaint.src.LanPaint.nodes as nodes # type: ignore[attr-defined]
|
||||
|
||||
return nodes
|
||||
|
||||
|
||||
def test_disabled_returns_full_count() -> None:
|
||||
nodes = _import_nodes()
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.1, 0.0) == 5 # min_frac 0 = off
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.01, 0.0) == 5
|
||||
|
||||
|
||||
def test_above_fraction_returns_full_count() -> None:
|
||||
nodes = _import_nodes()
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.2, 0.05) == 5
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.05, 0.05) == 5 # boundary kept
|
||||
|
||||
|
||||
def test_below_fraction_ramps_linear() -> None:
|
||||
nodes = _import_nodes()
|
||||
# frac / min_frac: 0.04/0.05 = 0.8 -> 5*0.8 = 4
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.04, 0.05) == 4
|
||||
# 0.025/0.05 = 0.5 -> 2.5 -> round -> 2
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.025, 0.05) == 2
|
||||
# 0.005/0.05 = 0.1 -> 0.5 -> round(0.5) = 0 (banker's), max(0, ...) anyway
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.005, 0.05) == 0
|
||||
assert nodes.min_step_frac_effective_steps(5, 0.0, 0.05) == 0
|
||||
|
||||
|
||||
def test_zero_steps_stays_zero() -> None:
|
||||
nodes = _import_nodes()
|
||||
assert nodes.min_step_frac_effective_steps(0, 0.01, 0.05) == 0
|
||||
@@ -0,0 +1,87 @@
|
||||
"""Tests for the retired LanPaint hyperparameters and the value sanitizer.
|
||||
|
||||
Beta/Friction/EarlyStop/InnerThreshold/InnerPatience/MinStepFrac were
|
||||
removed from the sampler node widgets; old prompts still pass them, so the
|
||||
nodes accept-and-ignore them via hidden inputs. Invalid widget values fall
|
||||
back to defaults instead of crashing the node.
|
||||
"""
|
||||
|
||||
RETIRED = [
|
||||
"LanPaint_Beta",
|
||||
"LanPaint_Friction",
|
||||
"LanPaint_EarlyStop",
|
||||
"LanPaint_InnerThreshold",
|
||||
"LanPaint_InnerPatience",
|
||||
"LanPaint_MinStepFrac",
|
||||
]
|
||||
|
||||
|
||||
def _import_nodes():
|
||||
import LanPaint.src.LanPaint.nodes as nodes # type: ignore[attr-defined]
|
||||
|
||||
return nodes
|
||||
|
||||
|
||||
def _required(node_cls):
|
||||
return node_cls.INPUT_TYPES().get("required", {})
|
||||
|
||||
|
||||
def _hidden(node_cls):
|
||||
return node_cls.INPUT_TYPES().get("hidden", {})
|
||||
|
||||
|
||||
def test_retired_params_removed_from_widgets() -> None:
|
||||
nodes = _import_nodes()
|
||||
for cls in (
|
||||
nodes.LanPaint_KSampler,
|
||||
nodes.LanPaint_KSamplerAdvanced,
|
||||
nodes.LanPaint_SamplerCustom,
|
||||
nodes.LanPaint_SamplerCustomAdvanced,
|
||||
):
|
||||
req = _required(cls)
|
||||
for name in RETIRED:
|
||||
assert name not in req, f"{cls.__name__} still exposes {name}"
|
||||
|
||||
|
||||
def test_retired_params_kept_as_hidden_inputs() -> None:
|
||||
nodes = _import_nodes()
|
||||
# The two advanced nodes exposed all six params; the basic KSampler
|
||||
# only ever had MinStepFrac. Old prompts must still validate.
|
||||
assert set(_hidden(nodes.LanPaint_KSamplerAdvanced)) >= set(RETIRED)
|
||||
assert set(_hidden(nodes.LanPaint_SamplerCustomAdvanced)) >= set(RETIRED)
|
||||
assert "LanPaint_MinStepFrac" in _hidden(nodes.LanPaint_KSampler)
|
||||
|
||||
|
||||
def test_kept_widgets_still_present() -> None:
|
||||
nodes = _import_nodes()
|
||||
req = _required(nodes.LanPaint_KSamplerAdvanced)
|
||||
for name in (
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
):
|
||||
assert name in req
|
||||
|
||||
|
||||
def test_sanitize_param_combos() -> None:
|
||||
nodes = _import_nodes()
|
||||
sanitize = nodes._sanitize_param
|
||||
allowed = ("Image First", "Prompt First")
|
||||
assert sanitize("Image First", "Image First", allowed=allowed) == "Image First"
|
||||
assert sanitize("Prompt First", "Image First", allowed=allowed) == "Prompt First"
|
||||
assert sanitize(1.0, "Image First", allowed=allowed) == "Image First" # retired float
|
||||
assert sanitize("bogus", "Image First", allowed=allowed) == "Image First"
|
||||
assert sanitize(None, "Image First", allowed=allowed) == "Image First"
|
||||
|
||||
|
||||
def test_sanitize_param_numbers() -> None:
|
||||
nodes = _import_nodes()
|
||||
sanitize = nodes._sanitize_param
|
||||
assert sanitize(5, 5) == 5
|
||||
assert sanitize(3.7, 0.2) == 3.7
|
||||
assert sanitize("abc", 0.2) == 0.2
|
||||
assert sanitize(None, 0.2) == 0.2
|
||||
assert sanitize(True, 5) == 5 # bool is not a valid number
|
||||
@@ -72,3 +72,50 @@ def test_prepare_mask_accepts_hw_and_moves_device(monkeypatch) -> None:
|
||||
out = lanpaint_nodes.prepare_mask(input_mask, output_shape, device=torch.device("cpu"), video_inpainting=False)
|
||||
assert tuple(out.shape) == output_shape
|
||||
assert out.device.type == "cpu"
|
||||
|
||||
|
||||
def test_video_union_covers_picked_frames(monkeypatch) -> None:
|
||||
"""EXPERIMENTAL order (nearest-exact first, slice-level union): a stroke
|
||||
on a frame that the resample picks survives, and the sliding max spreads
|
||||
it to neighboring slices."""
|
||||
lanpaint_nodes = _import_nodes(monkeypatch, "0.6.0")
|
||||
|
||||
# 8 frames -> 2 slices; nearest-exact picks frames {2, 6}. Strokes on
|
||||
# picked frames 2 and 6, on rows/cols the 8->4 grid samples ({1,3,5,7}).
|
||||
mask = torch.zeros(8, 8, 8)
|
||||
mask[2, 5, 5] = 1.0
|
||||
mask[6, 7, 7] = 1.0
|
||||
|
||||
out = lanpaint_nodes.reshape_mask(mask, (1, 16, 2, 4, 4), video_inpainting=True)
|
||||
assert tuple(out.shape) == (1, 16, 2, 4, 4)
|
||||
# both slices are marked (slice-level union spreads the picked values)
|
||||
assert out[0, 0, 0].max().item() == 1.0
|
||||
assert out[0, 0, 1].max().item() == 1.0
|
||||
# the stroke positions land on the token containing the stroke pixel
|
||||
assert out[0, 0, 0, 2, 2].item() == 1.0
|
||||
assert out[0, 0, 1, 3, 3].item() == 1.0
|
||||
|
||||
|
||||
def test_video_union_drops_unpicked_frames(monkeypatch) -> None:
|
||||
"""EXPERIMENTAL order: a stroke on a frame that no slice picks (e.g.
|
||||
frame 3, since nearest-exact picks {2, 6}) is silently lost."""
|
||||
lanpaint_nodes = _import_nodes(monkeypatch, "0.6.0")
|
||||
|
||||
mask = torch.zeros(8, 8, 8)
|
||||
mask[3, 5, 5] = 1.0
|
||||
|
||||
out = lanpaint_nodes.reshape_mask(mask, (1, 16, 2, 4, 4), video_inpainting=True)
|
||||
assert out.max().item() == 0.0
|
||||
|
||||
|
||||
def test_video_union_short_sequence(monkeypatch) -> None:
|
||||
"""A 3-frame sequence downsamples to one slice; the stroke must survive
|
||||
if it lands on the frame the resample picks (floor(0.5*3) = frame 1)."""
|
||||
lanpaint_nodes = _import_nodes(monkeypatch, "0.6.0")
|
||||
|
||||
mask = torch.zeros(3, 8, 8)
|
||||
mask[1, 5, 5] = 1.0
|
||||
|
||||
out = lanpaint_nodes.reshape_mask(mask, (1, 16, 1, 4, 4), video_inpainting=True)
|
||||
assert tuple(out.shape) == (1, 16, 1, 4, 4)
|
||||
assert out.max().item() == 1.0
|
||||
|
||||
@@ -491,38 +491,41 @@ def _reshape(nodes, mask, output_shape):
|
||||
return nodes.reshape_mask(mask, output_shape, video_inpainting=True)
|
||||
|
||||
|
||||
def test_union_window4_max_takes_the_union(monkeypatch, tmp_path) -> None:
|
||||
def test_union_picked_frames_take_the_union(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# frame 3 (window 0: frames 0-3) and frame 5 (window 1: frames 4-7)
|
||||
# EXPERIMENTAL order (interp -> pool): 8 frames -> 2 slices, nearest-exact
|
||||
# picks frames {2, 6}; strokes must be on picked frames to survive
|
||||
mask = torch.zeros(8, 6, 8)
|
||||
mask[3, 2, 3] = 1.0
|
||||
mask[5, 4, 5] = 1.0
|
||||
mask[2, 2, 3] = 1.0
|
||||
mask[6, 4, 5] = 1.0
|
||||
out = _reshape(nodes, mask, (1, 24, 2, 6, 8))
|
||||
assert out.shape == (1, 24, 2, 6, 8)
|
||||
assert out[0, 0, 0, 2, 3] == 1.0 # window 0: union of frames 0-3
|
||||
assert out[0, 0, 1, 4, 5] == 1.0 # window 1: union of frames 4-7
|
||||
assert out[0, 0, 0, 2, 3] == 1.0 # picked frame 2 -> slice 0
|
||||
assert out[0, 0, 1, 4, 5] == 1.0 # picked frame 6 -> slice 1
|
||||
assert out[0, 0, 0].max() == 1.0 and out[0, 0, 1].max() == 1.0
|
||||
|
||||
|
||||
def test_union_sparse_stroke_survives(monkeypatch, tmp_path) -> None:
|
||||
def test_union_sparse_stroke_spreads_over_slices(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# a stroke on a single frame must NOT average away: its token regenerates
|
||||
# EXPERIMENTAL order: a stroke on a picked frame (16 -> 4 slices picks
|
||||
# {2, 6, 10, 14}) spreads to neighboring slices via the slice-level pool
|
||||
mask = torch.zeros(16, 4, 4)
|
||||
mask[7, 1, 1] = 1.0
|
||||
mask[6, 1, 1] = 1.0
|
||||
out = _reshape(nodes, mask, (1, 24, 4, 4, 4))
|
||||
assert out.shape == (1, 24, 4, 4, 4)
|
||||
assert out.max() == 1.0 # the painted frame's token is fully regenerated
|
||||
assert out[0, 0, 0].max() == 0.0 # windows without paint stay 0
|
||||
assert out.max() == 1.0 # the picked frame's stroke regenerates
|
||||
# with T=4 and kernel 5, every slice's window covers the whole sequence
|
||||
assert out[0, 0, 0].max() == 1.0 and out[0, 0, 3].max() == 1.0
|
||||
|
||||
|
||||
def test_union_resamples_to_latent_shape_nearest(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# 8 frames -> 2 windows -> nearest-exact to the latent (T=2, /2 spatial)
|
||||
# EXPERIMENTAL order: 8 frames -> 2 slices; stroke on picked frame 6
|
||||
mask = torch.zeros(8, 8, 6)
|
||||
mask[5, 4:6, 2:4] = 1.0 # window 1 (frames 4-7), a small stroke
|
||||
mask[6, 4:6, 2:4] = 1.0 # picked frame, a small stroke
|
||||
out = _reshape(nodes, mask, (1, 24, 2, 4, 3))
|
||||
assert out.shape == (1, 24, 2, 4, 3)
|
||||
assert out[0, 0, 0].max() == 0.0
|
||||
assert out[0, 0, 0].max() == 1.0 # slice-level pool spreads to slice 0
|
||||
assert out[0, 0, 1].max() == 1.0
|
||||
# nearest-exact spatial: the painted pixel survives at its mapped location
|
||||
assert out[0, 0, 1, 2, 1] == 1.0
|
||||
@@ -530,13 +533,15 @@ def test_union_resamples_to_latent_shape_nearest(monkeypatch, tmp_path) -> None:
|
||||
|
||||
def test_union_124_frames_to_37_tokens(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# EXPERIMENTAL order: nearest-exact picks 124 -> 37 slice anchors; frame
|
||||
# 62 is one of them (anchor for slice 18), frame 60 is not
|
||||
mask = torch.zeros(124, 864, 480)
|
||||
mask[60, 100:140, 100:140] = 1.0 # a brush stroke on a single frame
|
||||
mask[62, 100:140, 100:140] = 1.0 # a brush stroke on a picked frame
|
||||
out = _reshape(nodes, mask, (1, 24, 37, 30, 54))
|
||||
assert out.shape == (1, 24, 37, 30, 54)
|
||||
assert out.max() == 1.0 # the paint is covered by some token
|
||||
# most tokens are empty (no window-4 smear)
|
||||
assert (out == 0.0).float().mean() > 0.95
|
||||
assert out.max() == 1.0 # the paint is covered by some slice
|
||||
# only the slice-level union spreads: ~5 of 37 slices are marked
|
||||
assert (out == 0.0).float().mean() > 0.8
|
||||
|
||||
|
||||
def test_union_leaves_static_single_frame_mask(monkeypatch, tmp_path) -> None:
|
||||
@@ -672,25 +677,27 @@ def test_av_encode_requires_audio_track(monkeypatch, tmp_path) -> None:
|
||||
|
||||
def test_union_4d_frames_at_batch_from_set_mask(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# SetLatentNoiseMask reshapes [F,H,W] -> [F,1,H,W]; frames are at batch
|
||||
# SetLatentNoiseMask reshapes [F,H,W] -> [F,1,H,W]; frames are at batch.
|
||||
# EXPERIMENTAL order: strokes must be on picked frames {2, 6} for 8->2.
|
||||
mask = torch.zeros(8, 1, 6, 8)
|
||||
mask[3, 0, 2, 3] = 1.0 # window 0 (frames 0-3)
|
||||
mask[5, 0, 4, 5] = 1.0 # window 1 (frames 4-7)
|
||||
mask[2, 0, 2, 3] = 1.0 # picked frame 2
|
||||
mask[6, 0, 4, 5] = 1.0 # picked frame 6
|
||||
out = nodes.reshape_mask(mask, (1, 24, 2, 6, 8), video_inpainting=True)
|
||||
assert out.shape == (1, 24, 2, 6, 8)
|
||||
assert out[0, 0, 0, 2, 3] == 1.0 # window 0 union
|
||||
assert out[0, 0, 1, 4, 5] == 1.0 # window 1 union
|
||||
assert out[0, 0, 0, 2, 3] == 1.0 # slice 0 carries frame 2's stroke
|
||||
assert out[0, 0, 1, 4, 5] == 1.0 # slice 1 carries frame 6's stroke
|
||||
assert out[0, 0, 0].max() == 1.0 and out[0, 0, 1].max() == 1.0
|
||||
|
||||
|
||||
def test_union_4d_frames_at_batch_124_to_37(monkeypatch, tmp_path) -> None:
|
||||
nodes = _import_nodes(monkeypatch, tmp_path)
|
||||
# EXPERIMENTAL order: frame 62 is a picked anchor for 124 -> 37 slices
|
||||
mask = torch.zeros(124, 1, 864, 480)
|
||||
mask[60, 0, 100:140, 100:140] = 1.0
|
||||
mask[62, 0, 100:140, 100:140] = 1.0
|
||||
out = nodes.reshape_mask(mask, (1, 24, 37, 30, 54), video_inpainting=True)
|
||||
assert out.shape == (1, 24, 37, 30, 54)
|
||||
assert out.max() == 1.0 # the painted frame is covered, not dropped
|
||||
assert (out == 0.0).float().mean() > 0.95
|
||||
assert (out == 0.0).float().mean() > 0.8
|
||||
|
||||
|
||||
def test_audio_mask_4d_via_set_mask_shape(monkeypatch, tmp_path) -> None:
|
||||
|
||||
@@ -9,8 +9,186 @@ const TARGET_NODES = new Set([
|
||||
"LanPaint_SamplerCustomAdvanced",
|
||||
]);
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Key-based widgets_values migration.
|
||||
//
|
||||
// ComfyUI's graph format stores widget values as positional arrays, which
|
||||
// silently corrupt when parameters are removed mid-list (a retired float
|
||||
// would land in a combo slot and reset the kept parameters to defaults).
|
||||
// LanPaint keeps a table of every historical layout (widget count -> widget
|
||||
// names in order). On load, an old array is converted to a name-keyed map
|
||||
// and re-emitted in the current widget order, so kept parameters keep their
|
||||
// values by NAME and retired ones (Beta, Friction, ...) simply drop out.
|
||||
// ---------------------------------------------------------------------------
|
||||
// Layout rows name every historical widget position. The trailing
|
||||
// "lanpaint_star_button" entry is the info-button widget this extension
|
||||
// appends to every LanPaint node; it is serialized into widgets_values like
|
||||
// any other widget and must occupy a slot in the layout.
|
||||
const BUTTON = "lanpaint_star_button";
|
||||
const LAYOUTS = {
|
||||
"LanPaint_KSampler": {
|
||||
current: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
],
|
||||
// 6 entries: 2.0.1 params + button; 5 entries: 2.0.0 params + button
|
||||
// (or 2.0.1 without the button) -- kept positions are identical.
|
||||
6: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
"LanPaint_MinStepFrac",
|
||||
BUTTON,
|
||||
],
|
||||
5: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
BUTTON,
|
||||
],
|
||||
},
|
||||
"LanPaint_KSamplerAdvanced": {
|
||||
current: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
],
|
||||
// 13 entries: 2.0.1 params + button; 12 entries: 2.0.0 params +
|
||||
// button (or 2.0.1 without the button) -- kept positions identical.
|
||||
13: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_Beta",
|
||||
"LanPaint_Friction",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_EarlyStop",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
"LanPaint_InnerThreshold",
|
||||
"LanPaint_InnerPatience",
|
||||
"LanPaint_MinStepFrac",
|
||||
BUTTON,
|
||||
],
|
||||
12: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_Beta",
|
||||
"LanPaint_Friction",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_EarlyStop",
|
||||
"LanPaint_Info",
|
||||
"Inpainting_mode",
|
||||
"LanPaint_InnerThreshold",
|
||||
"LanPaint_InnerPatience",
|
||||
BUTTON,
|
||||
],
|
||||
},
|
||||
"LanPaint_SamplerCustomAdvanced": {
|
||||
current: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_Info",
|
||||
],
|
||||
// 12 entries: 2.0.1 params + button; 11 entries: 2.0.0 params +
|
||||
// button (or 2.0.1 without the button) -- kept positions identical.
|
||||
12: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_Beta",
|
||||
"LanPaint_Friction",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_EarlyStop",
|
||||
"LanPaint_Info",
|
||||
"LanPaint_InnerThreshold",
|
||||
"LanPaint_InnerPatience",
|
||||
"LanPaint_MinStepFrac",
|
||||
BUTTON,
|
||||
],
|
||||
11: [
|
||||
"LanPaint_NumSteps",
|
||||
"LanPaint_Lambda",
|
||||
"LanPaint_StepSize",
|
||||
"LanPaint_Beta",
|
||||
"LanPaint_Friction",
|
||||
"LanPaint_PromptMode",
|
||||
"LanPaint_EarlyStop",
|
||||
"LanPaint_Info",
|
||||
"LanPaint_InnerThreshold",
|
||||
"LanPaint_InnerPatience",
|
||||
BUTTON,
|
||||
],
|
||||
},
|
||||
};
|
||||
|
||||
function migrateWidgetsValues(nodeData) {
|
||||
const table = LAYOUTS[nodeData.type];
|
||||
if (!table || !Array.isArray(nodeData.widgets_values)) {
|
||||
return;
|
||||
}
|
||||
const wv = nodeData.widgets_values;
|
||||
const current = table.current;
|
||||
if (wv.length === current.length) {
|
||||
return; // already the current layout
|
||||
}
|
||||
const old = table[wv.length];
|
||||
if (!old) {
|
||||
return; // unknown layout: leave the array untouched
|
||||
}
|
||||
const byName = {};
|
||||
for (let i = 0; i < wv.length; i++) {
|
||||
byName[old[i]] = wv[i];
|
||||
}
|
||||
if (!current.every((name) => name in byName)) {
|
||||
return; // a kept parameter has no source value: leave as-is
|
||||
}
|
||||
nodeData.widgets_values = current.map((name) => byName[name]);
|
||||
}
|
||||
|
||||
function installWidgetMigration() {
|
||||
const LGraphClass = app.graph?.constructor;
|
||||
if (!LGraphClass || LGraphClass.prototype.__lanpaint_migrated) {
|
||||
return;
|
||||
}
|
||||
LGraphClass.prototype.__lanpaint_migrated = true;
|
||||
|
||||
const origConfigure = LGraphClass.prototype.configure;
|
||||
LGraphClass.prototype.configure = function (graphData, ...rest) {
|
||||
if (graphData && Array.isArray(graphData.nodes)) {
|
||||
for (const nodeData of graphData.nodes) {
|
||||
migrateWidgetsValues(nodeData);
|
||||
}
|
||||
const subgraphs = graphData.definitions?.subgraphs;
|
||||
if (Array.isArray(subgraphs)) {
|
||||
for (const sub of subgraphs) {
|
||||
if (Array.isArray(sub?.nodes)) {
|
||||
for (const nodeData of sub.nodes) {
|
||||
migrateWidgetsValues(nodeData);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return origConfigure.call(this, graphData, ...rest);
|
||||
};
|
||||
}
|
||||
|
||||
app.registerExtension({
|
||||
name: "LanPaint.InfoLink",
|
||||
setup() {
|
||||
installWidgetMigration();
|
||||
},
|
||||
async nodeCreated(node) {
|
||||
if (!node?.comfyClass || !TARGET_NODES.has(node.comfyClass)) {
|
||||
return;
|
||||
|
||||