Merge pull request #14 from GiusTex/Test

change model used
This commit is contained in:
GiusTex
2024-11-14 19:06:36 +01:00
committed by GitHub
6 changed files with 1006 additions and 574 deletions
@@ -0,0 +1,525 @@
{
"last_node_id": 591,
"last_link_id": 1263,
"nodes": [
{
"id": 530,
"type": "VAEDecode",
"pos": {
"0": 660,
"1": 90
},
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 1241
},
{
"name": "vae",
"type": "VAE",
"link": 1261
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1262
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
},
"widgets_values": []
},
{
"id": 584,
"type": "DiffusersImageOutpaint",
"pos": {
"0": 320,
"1": 90
},
"size": {
"0": 300,
"1": 214
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1255
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 1254
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 1258
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"link": 1251
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
1241
],
"slot_index": 0
}
],
"title": "DiffusersImageOutpaint",
"properties": {
"Node name for S&R": "DiffusersImageOutpaint"
},
"widgets_values": [
1.5,
1,
43078817542338,
"randomize",
8
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 529,
"type": "PadImageForDiffusersOutpaint",
"pos": {
"0": 0,
"1": 390
},
"size": {
"0": 290,
"1": 150
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 1263
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": null
},
{
"name": "MASK",
"type": "MASK",
"links": null
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"links": [
1251
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "PadImageForDiffusersOutpaint"
},
"widgets_values": [
720,
1280,
"Top"
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 534,
"type": "LoadDiffusersOutpaintModels",
"pos": {
"0": -480,
"1": 60
},
"size": {
"0": 320,
"1": 154
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [
1253,
1257
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "LoadDiffusersOutpaintModels"
},
"widgets_values": [
"RealVisXL_V5.0_Lightning",
"controlnet-union-sdxl-1.0",
"auto",
"auto",
false
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 588,
"type": "EncodeDiffusersOutpaintPrompt",
"pos": {
"0": -110,
"1": 220
},
"size": {
"0": 400,
"1": 96
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1257
},
{
"name": "clip",
"type": "CLIP",
"link": 1259
}
],
"outputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [],
"slot_index": 0
},
{
"name": "diffusers_conditioning",
"type": "CONDITIONING",
"links": [
1258
],
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "EncodeDiffusersOutpaintPrompt"
},
"widgets_values": [
""
],
"color": "#322",
"bgcolor": "#533"
},
{
"id": 351,
"type": "PreviewImage",
"pos": {
"0": 650,
"1": 180
},
"size": {
"0": 510,
"1": 490
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 1262
}
],
"outputs": [],
"properties": {
"Node name for S&R": "PreviewImage"
},
"widgets_values": []
},
{
"id": 591,
"type": "LoadImage",
"pos": {
"0": -350,
"1": 400
},
"size": [
320,
310
],
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1263
],
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"20230403_183417.jpg",
"image"
]
},
{
"id": 587,
"type": "EncodeDiffusersOutpaintPrompt",
"pos": {
"0": -120,
"1": 70
},
"size": {
"0": 400,
"1": 96
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1253
},
{
"name": "clip",
"type": "CLIP",
"link": 1260
}
],
"outputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [
1255
],
"slot_index": 0
},
{
"name": "diffusers_conditioning",
"type": "CONDITIONING",
"links": [
1254
],
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "EncodeDiffusersOutpaintPrompt"
},
"widgets_values": [
""
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 589,
"type": "CheckpointLoaderSimple",
"pos": {
"0": -500,
"1": 250
},
"size": {
"0": 360,
"1": 100
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": null
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
1259,
1260
],
"slot_index": 1
},
{
"name": "VAE",
"type": "VAE",
"links": [
1261
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "CheckpointLoaderSimple"
},
"widgets_values": [
"realvisxlV50_v50LightningBakedvae.safetensors"
],
"color": "#223",
"bgcolor": "#335"
}
],
"links": [
[
1241,
584,
0,
530,
0,
"LATENT"
],
[
1251,
529,
2,
584,
3,
"IMAGE"
],
[
1253,
534,
0,
587,
0,
"PIPE"
],
[
1254,
587,
1,
584,
1,
"CONDITIONING"
],
[
1255,
587,
0,
584,
0,
"PIPE"
],
[
1257,
534,
0,
588,
0,
"PIPE"
],
[
1258,
588,
1,
584,
2,
"CONDITIONING"
],
[
1259,
589,
1,
588,
1,
"CLIP"
],
[
1260,
589,
1,
587,
1,
"CLIP"
],
[
1261,
589,
2,
530,
1,
"VAE"
],
[
1262,
530,
0,
351,
0,
"IMAGE"
],
[
1263,
591,
0,
529,
0,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.8769226950000005,
"offset": [
570.3286926097892,
6.798044267632111
]
}
},
"version": 0.4
}
+415 -356
View File
@@ -1,30 +1,176 @@
{
"last_node_id": 689,
"last_link_id": 1301,
"last_node_id": 591,
"last_link_id": 1259,
"nodes": [
{
"id": 675,
"type": "LoadDiffusersOutpaintModels",
"id": 584,
"type": "DiffusersImageOutpaint",
"pos": {
"0": -210,
"1": 160
"0": 320,
"1": 90
},
"size": {
"0": 320,
"1": 154
"0": 300,
"1": 214
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1255
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 1254
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 1258
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"link": 1251
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
1241
],
"slot_index": 0
}
],
"title": "DiffusersImageOutpaint",
"properties": {
"Node name for S&R": "DiffusersImageOutpaint"
},
"widgets_values": [
1.5,
1,
1009037337630565,
"randomize",
8
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 529,
"type": "PadImageForDiffusersOutpaint",
"pos": {
"0": 0,
"1": 390
},
"size": {
"0": 290,
"1": 150
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 1005
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": null
},
{
"name": "MASK",
"type": "MASK",
"links": null
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"links": [
1251
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "PadImageForDiffusersOutpaint"
},
"widgets_values": [
720,
1280,
"Top"
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 531,
"type": "VAELoader",
"pos": {
"0": 370,
"1": 350
},
"size": {
"0": 260,
"1": 60
},
"flags": {},
"order": 0,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "VAE",
"type": "VAE",
"links": [
1007
]
}
],
"properties": {
"Node name for S&R": "VAELoader"
},
"widgets_values": [
"sdxl_vae.safetensors"
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 534,
"type": "LoadDiffusersOutpaintModels",
"pos": {
"0": -480,
"1": 60
},
"size": {
"0": 320,
"1": 154
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [
1264
1253,
1257
],
"shape": 3
"slot_index": 0
}
],
"properties": {
@@ -41,24 +187,190 @@
"bgcolor": "#335"
},
{
"id": 351,
"type": "PreviewImage",
"id": 588,
"type": "EncodeDiffusersOutpaintPrompt",
"pos": {
"0": 970,
"1": 280
"0": -110,
"1": 220
},
"size": {
"0": 510,
"1": 490
"0": 400,
"1": 96
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1257
},
{
"name": "clip",
"type": "CLIP",
"link": 1256
}
],
"outputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [],
"slot_index": 0
},
{
"name": "diffusers_conditioning",
"type": "CONDITIONING",
"links": [
1258
],
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "EncodeDiffusersOutpaintPrompt"
},
"widgets_values": [
""
],
"color": "#322",
"bgcolor": "#533"
},
{
"id": 580,
"type": "DualCLIPLoader",
"pos": {
"0": -420,
"1": 260
},
"size": {
"0": 260,
"1": 110
},
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "CLIP",
"type": "CLIP",
"links": [
1252,
1256
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DualCLIPLoader"
},
"widgets_values": [
"clip_l.safetensors",
"model.fp16.safetensors",
"sdxl"
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 530,
"type": "VAEDecode",
"pos": {
"0": 660,
"1": 90
},
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 1241
},
{
"name": "vae",
"type": "VAE",
"link": 1007
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1259
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
},
"widgets_values": []
},
{
"id": 528,
"type": "LoadImage",
"pos": {
"0": -360,
"1": 430
},
"size": [
320,
310
],
"flags": {},
"order": 3,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1005
]
},
{
"name": "MASK",
"type": "MASK",
"links": null
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"Emilia3.png",
"image"
]
},
{
"id": 351,
"type": "PreviewImage",
"pos": {
"0": 670,
"1": 190
},
"size": {
"0": 510,
"1": 490
},
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 679
"link": 1259
}
],
"outputs": [],
@@ -68,306 +380,29 @@
"widgets_values": []
},
{
"id": 402,
"type": "GetImageSizeAndCount",
"id": 587,
"type": "EncodeDiffusersOutpaintPrompt",
"pos": {
"0": 1200,
"1": 150
"0": -120,
"1": 70
},
"size": {
"0": 210,
"1": 90
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 1297
}
],
"outputs": [
{
"name": "image",
"type": "IMAGE",
"links": [
679
],
"slot_index": 0,
"shape": 3
},
{
"name": "720 width",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "1280 height",
"type": "INT",
"links": null,
"shape": 3
},
{
"name": "1 count",
"type": "INT",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "GetImageSizeAndCount"
},
"widgets_values": []
},
{
"id": 688,
"type": "VAEDecode",
"pos": {
"0": 960,
"1": 160
},
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 1301
},
{
"name": "vae",
"type": "VAE",
"link": 1296
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1297
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VAEDecode"
},
"widgets_values": []
},
{
"id": 689,
"type": "DiffusersImageOutpaint",
"pos": {
"0": 620,
"1": 150
},
"size": {
"0": 320,
"1": 190
},
"flags": {},
"order": 5,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1298
},
{
"name": "diffusers_outpaint_conditioning",
"type": "CONDITIONING",
"link": 1299
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"link": 1300
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
1301
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "DiffusersImageOutpaint"
},
"widgets_values": [
1.5,
1,
693728282818736,
"randomize",
8
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 683,
"type": "VAELoader",
"pos": {
"0": 630,
"1": 400
},
"size": {
"0": 315,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "VAE",
"type": "VAE",
"links": [
1296
],
"slot_index": 0,
"shape": 3
}
],
"properties": {
"Node name for S&R": "VAELoader"
},
"widgets_values": [
"sdxl_vae.safetensors"
],
"color": "#223",
"bgcolor": "#335"
},
{
"id": 1,
"type": "LoadImage",
"pos": {
"0": -200,
"1": 360
},
"size": [
320,
310
],
"flags": {},
"order": 2,
"mode": 0,
"inputs": [],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
1194
],
"slot_index": 0,
"shape": 3
},
{
"name": "MASK",
"type": "MASK",
"links": [],
"slot_index": 1,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"20240930_201555.jpg",
"image"
]
},
{
"id": 653,
"type": "PadImageForDiffusersOutpaint",
"pos": {
"0": 290,
"1": 300
},
"size": {
"0": 290,
"1": 150
"0": 400,
"1": 96
},
"flags": {},
"order": 4,
"mode": 0,
"inputs": [
{
"name": "image",
"type": "IMAGE",
"link": 1194
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": null,
"shape": 3
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
},
{
"name": "diffuser_outpaint_cnet_image",
"type": "IMAGE",
"links": [
1300
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "PadImageForDiffusersOutpaint"
},
"widgets_values": [
720,
1280,
"Top"
],
"color": "#232",
"bgcolor": "#353"
},
{
"id": 655,
"type": "EncodeDiffusersOutpaintPrompt",
"pos": {
"0": 130,
"1": 150
},
"size": {
"0": 463.6000061035156,
"1": 80
},
"flags": {},
"order": 3,
"mode": 0,
"inputs": [
{
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"link": 1264
"link": 1253
},
{
"name": "clip",
"type": "CLIP",
"link": 1252
}
],
"outputs": [
@@ -375,24 +410,24 @@
"name": "diffusers_outpaint_pipe",
"type": "PIPE",
"links": [
1298
1255
],
"shape": 3
"slot_index": 0
},
{
"name": "diffusers_outpaint_conditioning",
"name": "diffusers_conditioning",
"type": "CONDITIONING",
"links": [
1299
1254
],
"shape": 3
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "EncodeDiffusersOutpaintPrompt"
},
"widgets_values": [
"sitting"
""
],
"color": "#232",
"bgcolor": "#353"
@@ -400,86 +435,110 @@
],
"links": [
[
679,
402,
1005,
528,
0,
351,
529,
0,
"IMAGE"
],
[
1194,
1,
1007,
531,
0,
653,
0,
"IMAGE"
],
[
1264,
675,
0,
655,
0,
"PIPE"
],
[
1296,
683,
0,
688,
530,
1,
"VAE"
],
[
1297,
688,
1241,
584,
0,
402,
530,
0,
"LATENT"
],
[
1251,
529,
2,
584,
3,
"IMAGE"
],
[
1298,
655,
1252,
580,
0,
689,
587,
1,
"CLIP"
],
[
1253,
534,
0,
587,
0,
"PIPE"
],
[
1299,
655,
1254,
587,
1,
689,
584,
1,
"CONDITIONING"
],
[
1300,
653,
2,
689,
2,
"IMAGE"
1255,
587,
0,
584,
0,
"PIPE"
],
[
1301,
689,
1256,
580,
0,
688,
588,
1,
"CLIP"
],
[
1257,
534,
0,
"LATENT"
588,
0,
"PIPE"
],
[
1258,
588,
1,
584,
2,
"CONDITIONING"
],
[
1259,
530,
0,
351,
0,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {
"ds": {
"scale": 0.8769226950000005,
"scale": 0.7247295000000004,
"offset": [
254.45705206435736,
-58.04927351924108
679.3196754354946,
80.60613789648367
]
}
},
+20 -13
View File
@@ -1,15 +1,23 @@
ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-outpaint](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) by fffiloni.
![DiffusersImageOutpaint-Nodes-Screen](https://github.com/user-attachments/assets/2722e07c-1d6a-416e-a9d8-f26aaa9a45a7)
![image](https://github.com/user-attachments/assets/8f7665a1-dd8c-44d6-a067-fcc3f48b1865)
#### Updates:
- You can test the updates in the `Test` branch not yet merged opening a terminal in the extension folder, and typing `git checkout Test`, then `git pull` or `git pull ComfyUI-DiffusersImageOutpaint Test`.
- You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**.
- If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below).
- 22/10/2024:
- Unet and Controlnet Models Loader using ComfYUI nodes canceled, since I can't find a way to load them properly; more info at the end.
- Guide to change model used.
- 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. You can also use the Checkpoint Loader Simple node, to skip the clip selection part.
- 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**.
#### To do list to [change model used](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/14):
- - [x] ComfyUI Clip Loader Node
- ~[ ] ComfyUI Load Diffusion Model Node~
- ~[ ] ComfyUI Load Conotrolnet Model Node~
## Installation
- Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`.
- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model, but you need the specified vae and controlnet since the extension is hardcoded to use them. You can always change the code to use different models):
- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model ([RealvisXLv50-bakedVae on Civitai](https://civitai.com/models/139562/realvisxl-v50)), but you need the specified controlnet since the extension is hardcoded to use it, for now):
| **main model** | **controlnet model** |
| :-----: | :-----: |
| **[Diffuser Model folder](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/tree/main)** (you can change this model) | **[Diffuser Controlnet folder](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/tree/main)** (you need this model) |
@@ -18,14 +26,6 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o
| config.json, diffusion_pytorch_model.fp16.safetensors | |
| **Scheduler folder** | |
| scheduler_config.json | |
| **Text encoder folder** | |
| config.json, ~model.fp16.safetensors~ -> model.safetensors | |
| **Text encoder 2 folder** | |
| config.json, ~model.fp16.safetensors~ -> model.safetensors | |
| **Tokenizer folder** | |
| merges.txt, special_tokens_map.json, tokenizer_config.json, vocab.json | |
| **Tokenizer 2 folder** | |
| merges.txt, special_tokens_map.json, tokenizer_config.json, vocab.json | |
## Overview
- **Minimum VRAM**: 6 gb with 1280x720 image, rtx 3060, RealVisXL_V5.0_Lightning, sdxl-vae-fp16-fix, controlnet-union-sdxl-promax using `sequential_cpu_offload`, otherwise 8,3 gb;
@@ -39,5 +39,12 @@ The extension gives 4 nodes:
- You can also pass image and mask to `vae encode (for inpainting)` node, then pass the latent to a `sampler`, but controlnets and ip-adapters are harder to use compared to diffusers outpaint.
### Change model used
- **Main model**: On huggingface, choose a model from [text2image models](https://huggingface.co/models?pipeline_tag=text-to-image&sort=trending), then create a new folder named after it in `comfyui/models/diffusion_models`, then download in it the subfolders `unet` and `scheduler`.
- **Controlnet model**: same as above, except you don't need the `scheduler`.
#### Unet and Controlnet Models Loader using ComfYUI nodes canceled
I can load them but then they don't work in the inference code, since comfyui load diffusers models in a different format ([reddit post](https://www.reddit.com/r/comfyui/comments/17fvb49/comment/k6cz9yv/?utm_source=share&utm_medium=web3x&utm_name=web3xcss&utm_term=1&utm_content=share_button)).
## Credits
diffusers-image-outpaint by [fffiloni](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main)
+36 -24
View File
@@ -1,8 +1,7 @@
import torch
import gc
import os
from PIL import Image
from .utils import get_first_folder_list, tensor2pil, pil2tensor, encodeDiffOutpaintPrompt, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram
from .utils import get_first_folder_list, tensor2pil, pil2tensor, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram
# Get the absolute path of various directories
@@ -186,33 +185,44 @@ class EncodeDiffusersOutpaintPrompt:
return {
"required": {
"diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}),
"extra_prompt": ("STRING", {"default": "", "tooltip": "The extra prompt to append, describing attributes etc. you want to include in the image. Default: \"(extra_prompt), high quality, 4k\""}),
"text": ("STRING", {"multiline": True, "dynamicPrompts": True, "tooltip": "The text to be encoded."}),
"clip": ("CLIP", {"tooltip": "The CLIP model used for encoding the text."})
}
}
RETURN_TYPES = ("PIPE","CONDITIONING",)
RETURN_NAMES = ("diffusers_outpaint_pipe","diffusers_outpaint_conditioning",)
RETURN_NAMES = ("diffusers_outpaint_pipe","diffusers_conditioning",)
OUTPUT_TOOLTIPS = ("A conditioning containing the embedded text used to guide the diffusion model.",)
FUNCTION = "encode"
CATEGORY = "DiffusersOutpaint"
DESCRIPTION = "Encodes a text prompt using a CLIP model into an embedding that can be used to guide the diffusion model towards generating specific images."
def encode(self, diffusers_outpaint_pipe, extra_prompt=None):
model_path = diffusers_outpaint_pipe["model_path"]
def encode(self, diffusers_outpaint_pipe, text, clip):
dtype = diffusers_outpaint_pipe["dtype"]
device = diffusers_outpaint_pipe["device"]
final_prompt = f"{extra_prompt}, high quality, 4k"
prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds = encodeDiffOutpaintPrompt(model_path, dtype, final_prompt, device)
text = f"{text}, high quality, 4k"
tokens = clip.tokenize(text)
output = clip.encode_from_tokens(tokens, return_pooled=True, return_dict=True)
prompt_embeds = output.pop("cond")
diffusers_outpaint_conditioning = {
prompt_embeds = prompt_embeds.to(device, dtype=dtype)
pooled_prompt_embeds = output["pooled_output"].to(device, dtype=dtype)
bs_embed, seq_len, _ = prompt_embeds.shape
# duplicate text embeddings for each generation per prompt, using mps friendly method
prompt_embeds = prompt_embeds.repeat(1, 1, 1)
prompt_embeds = prompt_embeds.view(bs_embed * 1, seq_len, -1)
pooled_prompt_embeds = pooled_prompt_embeds.repeat(1, 1).view(bs_embed * 1, -1)
diffusers_conditioning = {
"prompt_embeds": prompt_embeds,
"negative_prompt_embeds": negative_prompt_embeds,
"pooled_prompt_embeds": pooled_prompt_embeds,
"negative_pooled_prompt_embeds": negative_pooled_prompt_embeds
}
return (diffusers_outpaint_pipe,diffusers_outpaint_conditioning,)
return (diffusers_outpaint_pipe,diffusers_conditioning,)
class DiffusersImageOutpaint:
@classmethod
@@ -220,7 +230,8 @@ class DiffusersImageOutpaint:
return {
"required": {
"diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}),
"diffusers_outpaint_conditioning": ("CONDITIONING", {"tooltip": "The prompt describing what you want."}),
"positive": ("CONDITIONING", {"tooltip": "The prompt describing what you want."}),
"negative": ("CONDITIONING", {"tooltip": "The prompt describing what you don't want."}),
"diffuser_outpaint_cnet_image": ("IMAGE", {"tooltip": "The image to outpaint."}),
"guidance_scale": ("FLOAT", {"default": 1.50, "min": 1.01, "max": 10, "step": 0.01, "tooltip": "The Classifier-Free Guidance scale balances creativity and adherence to the prompt. Higher values result in images more closely matching the prompt, however too high values will negatively impact quality."}),
"controlnet_strength": ("FLOAT", {"default": 1.00, "min": 0.00, "max": 10, "step": 0.01}),
@@ -233,7 +244,7 @@ class DiffusersImageOutpaint:
FUNCTION = "sample"
CATEGORY = "DiffusersOutpaint"
def sample(self, diffusers_outpaint_pipe, diffusers_outpaint_conditioning, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, seed, steps):
def sample(self, diffusers_outpaint_pipe, positive, negative, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, seed, steps):
cnet_image = diffuser_outpaint_cnet_image
cnet_image=tensor2pil(cnet_image)
@@ -245,17 +256,18 @@ class DiffusersImageOutpaint:
dtype = diffusers_outpaint_pipe["dtype"]
device = diffusers_outpaint_pipe["device"]
keep_model_device = diffusers_outpaint_pipe["keep_model_device"]
prompt_embeds = diffusers_outpaint_conditioning["prompt_embeds"]
negative_prompt_embeds = diffusers_outpaint_conditioning["negative_prompt_embeds"]
pooled_prompt_embeds = diffusers_outpaint_conditioning["pooled_prompt_embeds"]
negative_pooled_prompt_embeds = diffusers_outpaint_conditioning["negative_pooled_prompt_embeds"]
prompt_embeds = positive["prompt_embeds"]
pooled_prompt_embeds = positive["pooled_prompt_embeds"]
negative_prompt_embeds = negative["prompt_embeds"]
negative_pooled_prompt_embeds = negative["pooled_prompt_embeds"]
last_rgb_latent = diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet_image, dtype, controlnet_path,
prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds,
device, steps, controlnet_strength, guidance_scale,
keep_model_device)
del diffusers_outpaint_conditioning
clearVram(device)
del prompt_embeds, pooled_prompt_embeds, negative_prompt_embeds, negative_pooled_prompt_embeds
clearVram(device)
return ({"samples":last_rgb_latent},)
+8 -164
View File
@@ -67,163 +67,6 @@ def retrieve_timesteps(
return timesteps, num_inference_steps
def encode_prompt(
prompt: str,
tokenizer: None,
tokenizer_2: None,
text_encoder: None,
text_encoder_2: None,
device: Optional[torch.device] = None,
do_classifier_free_guidance: bool = True,
):
prompt = [prompt] if isinstance(prompt, str) else prompt
if prompt is not None:
batch_size = len(prompt)
# Define tokenizers and text encoders
tokenizers = (
[tokenizer, tokenizer_2]
if tokenizer is not None
else [tokenizer_2]
)
text_encoders = (
[text_encoder, text_encoder_2]
if text_encoder is not None
else [text_encoder_2]
)
prompt_2 = prompt
prompt_2 = [prompt_2] if isinstance(prompt_2, str) else prompt_2
# textual inversion: process multi-vector tokens if necessary
prompt_embeds_list = []
prompts = [prompt, prompt_2]
for prompt, tokenizer, text_encoder in zip(prompts, tokenizers, text_encoders):
text_inputs = tokenizer(
prompt,
padding="max_length",
max_length=tokenizer.model_max_length,
truncation=True,
return_tensors="pt",
)
text_input_ids = text_inputs.input_ids
prompt_embeds = text_encoder(
text_input_ids.to(device), output_hidden_states=True
)
# We are only ALWAYS interested in the pooled output of the final text encoder
pooled_prompt_embeds = prompt_embeds[0]
prompt_embeds = prompt_embeds.hidden_states[-2]
prompt_embeds_list.append(prompt_embeds)
prompt_embeds = torch.concat(prompt_embeds_list, dim=-1)
# get unconditional embeddings for classifier free guidance
zero_out_negative_prompt = True
negative_prompt_embeds = None
negative_pooled_prompt_embeds = None
if do_classifier_free_guidance and zero_out_negative_prompt:
negative_prompt_embeds = torch.zeros_like(prompt_embeds)
negative_pooled_prompt_embeds = torch.zeros_like(pooled_prompt_embeds)
elif do_classifier_free_guidance and negative_prompt_embeds is None:
negative_prompt = ""
negative_prompt_2 = negative_prompt
# normalize str to list
negative_prompt = (
batch_size * [negative_prompt]
if isinstance(negative_prompt, str)
else negative_prompt
)
negative_prompt_2 = (
batch_size * [negative_prompt_2]
if isinstance(negative_prompt_2, str)
else negative_prompt_2
)
uncond_tokens: List[str]
if prompt is not None and type(prompt) is not type(negative_prompt):
raise TypeError(
f"`negative_prompt` should be the same type to `prompt`, but got {type(negative_prompt)} !="
f" {type(prompt)}."
)
elif batch_size != len(negative_prompt):
raise ValueError(
f"`negative_prompt`: {negative_prompt} has batch size {len(negative_prompt)}, but `prompt`:"
f" {prompt} has batch size {batch_size}. Please make sure that passed `negative_prompt` matches"
" the batch size of `prompt`."
)
else:
uncond_tokens = [negative_prompt, negative_prompt_2]
negative_prompt_embeds_list = []
for negative_prompt, tokenizer, text_encoder in zip(
uncond_tokens, tokenizers, text_encoders
):
max_length = prompt_embeds.shape[1]
uncond_input = tokenizer(
negative_prompt,
padding="max_length",
max_length=max_length,
truncation=True,
return_tensors="pt",
)
negative_prompt_embeds = text_encoder(
uncond_input.input_ids.to(device),
output_hidden_states=True,
)
# We are only ALWAYS interested in the pooled output of the final text encoder
negative_pooled_prompt_embeds = negative_prompt_embeds[0]
negative_prompt_embeds = negative_prompt_embeds.hidden_states[-2]
negative_prompt_embeds_list.append(negative_prompt_embeds)
negative_prompt_embeds = torch.concat(negative_prompt_embeds_list, dim=-1)
prompt_embeds = prompt_embeds.to(dtype=text_encoder_2.dtype, device=device)
bs_embed, seq_len, _ = prompt_embeds.shape
# duplicate text embeddings for each generation per prompt, using mps friendly method
prompt_embeds = prompt_embeds.repeat(1, 1, 1)
prompt_embeds = prompt_embeds.view(bs_embed * 1, seq_len, -1)
if do_classifier_free_guidance:
# duplicate unconditional embeddings for each generation per prompt, using mps friendly method
seq_len = negative_prompt_embeds.shape[1]
if text_encoder_2 is not None:
negative_prompt_embeds = negative_prompt_embeds.to(
dtype=text_encoder_2.dtype, device=device
)
else:
negative_prompt_embeds = negative_prompt_embeds.to(
dtype=torch.float16, device=device
)
negative_prompt_embeds = negative_prompt_embeds.repeat(1, 1, 1)
negative_prompt_embeds = negative_prompt_embeds.view(
batch_size * 1, seq_len, -1
)
pooled_prompt_embeds = pooled_prompt_embeds.repeat(1, 1).view(bs_embed * 1, -1)
if do_classifier_free_guidance:
negative_pooled_prompt_embeds = negative_pooled_prompt_embeds.repeat(
1, 1
).view(bs_embed * 1, -1)
return (
prompt_embeds,
negative_prompt_embeds,
pooled_prompt_embeds,
negative_pooled_prompt_embeds,
)
class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
def __init__(
@@ -291,7 +134,7 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
# corresponds to doing no classifier free guidance.
@property
def do_classifier_free_guidance(self):
return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None # UNET <----
return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None
@property
def num_timesteps(self):
@@ -302,10 +145,11 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
self,
controlnet_model,
device,
dtype,
keep_model_device,
prompt_embeds: torch.Tensor,
negative_prompt_embeds: torch.Tensor,
pooled_prompt_embeds: torch.Tensor,
negative_prompt_embeds: torch.Tensor,
negative_pooled_prompt_embeds: torch.Tensor,
image: PipelineImageInput = None,
num_inference_steps: int = 8,
@@ -343,10 +187,10 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
num_channels_latents,
height,
width,
prompt_embeds.dtype,
dtype,
device,
)
# 7 Prepare added time ids & embeddings
add_text_embeds = pooled_prompt_embeds
@@ -376,7 +220,7 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
"time_ids": add_time_ids,
"control_type": union_control_type,
}
controlnet_prompt_embeds = prompt_embeds.to(device)
controlnet_added_cond_kwargs = added_cond_kwargs
@@ -430,9 +274,9 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin):
)[0]
if keep_model_device:
self.unet.to('cpu')
except torch.cuda.OutOfMemoryError as e: # free vram when OOM
except torch.cuda.OutOfMemoryError as e: # Free vram when OOM
self.unet.to('cpu')
print('\033[93m', 'Gpu is out of memory(爆显存了)!', '\033[0m')
print('\033[93m', 'Gpu is out of memory!', '\033[0m')
raise e
# perform guidance
+2 -17
View File
@@ -8,7 +8,7 @@ from PIL import Image
from folder_paths import map_legacy, folder_names_and_paths
from .controlnet_union import ControlNetModel_Union
from .pipeline_fill_sd_xl import encode_prompt, StableDiffusionXLFillPipeline
from .pipeline_fill_sd_xl import StableDiffusionXLFillPipeline
from diffusers import AutoencoderKL, TCDScheduler
from diffusers.models.model_loading_utils import load_state_dict
from transformers import CLIPTextModel, CLIPTextModelWithProjection, CLIPTokenizer
@@ -94,22 +94,6 @@ def clearVram(device):
# torch.ipc_collect() not available, and ipc_collect seems available only for cuda
def encodeDiffOutpaintPrompt(model_path, dtype, final_prompt, device):
tokenizer, tokenizer_2, text_encoder, text_encoder_2 = loadDiffModels1(model_path, dtype, device)
(prompt_embeds,
negative_prompt_embeds,
pooled_prompt_embeds,
negative_pooled_prompt_embeds,
) = encode_prompt(final_prompt, tokenizer, tokenizer_2, text_encoder, text_encoder_2, device, True)
del tokenizer, tokenizer_2, text_encoder, text_encoder_2
clearVram(device)
return prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds
def loadControlnetModel(device, dtype, controlnet_path):
config_file = f"{controlnet_path}/config_promax.json"
config = ControlNetModel_Union.load_config(config_file)
@@ -179,6 +163,7 @@ def diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet
controlnet_conditioning_scale=controlnet_strength,
guidance_scale=guidance_scale,
device=device,
dtype=dtype,
keep_model_device=keep_model_device,
))