From 74cf628e4e9d54ee543539281c0fd61538cc3aee Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 20:37:29 +0200 Subject: [PATCH 01/11] made available comfyui clip No more need to download tokenizers and text encoders, now we can use comfyui clip loader --- utils.py | 19 ++----------------- 1 file changed, 2 insertions(+), 17 deletions(-) diff --git a/utils.py b/utils.py index 4346724..42271eb 100644 --- a/utils.py +++ b/utils.py @@ -8,7 +8,7 @@ from PIL import Image from folder_paths import map_legacy, folder_names_and_paths from .controlnet_union import ControlNetModel_Union -from .pipeline_fill_sd_xl import encode_prompt, StableDiffusionXLFillPipeline +from .pipeline_fill_sd_xl import StableDiffusionXLFillPipeline from diffusers import AutoencoderKL, TCDScheduler from diffusers.models.model_loading_utils import load_state_dict from transformers import CLIPTextModel, CLIPTextModelWithProjection, CLIPTokenizer @@ -94,22 +94,6 @@ def clearVram(device): # torch.ipc_collect() not available, and ipc_collect seems available only for cuda -def encodeDiffOutpaintPrompt(model_path, dtype, final_prompt, device): - tokenizer, tokenizer_2, text_encoder, text_encoder_2 = loadDiffModels1(model_path, dtype, device) - - (prompt_embeds, - negative_prompt_embeds, - pooled_prompt_embeds, - negative_pooled_prompt_embeds, - ) = encode_prompt(final_prompt, tokenizer, tokenizer_2, text_encoder, text_encoder_2, device, True) - - del tokenizer, tokenizer_2, text_encoder, text_encoder_2 - - clearVram(device) - - return prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds - - def loadControlnetModel(device, dtype, controlnet_path): config_file = f"{controlnet_path}/config_promax.json" config = ControlNetModel_Union.load_config(config_file) @@ -179,6 +163,7 @@ def diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet controlnet_conditioning_scale=controlnet_strength, guidance_scale=guidance_scale, device=device, + dtype=dtype, keep_model_device=keep_model_device, )) From bc7c1ded58fa05873350a7ca6315e2afbc1616ba Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 20:38:46 +0200 Subject: [PATCH 02/11] made available comfyui clip No more need to download tokenizers and text encoders, now we can use comfyui clip loader --- nodes.py | 60 +++++++++++++++++++++++++++++++++----------------------- 1 file changed, 36 insertions(+), 24 deletions(-) diff --git a/nodes.py b/nodes.py index 3c33382..1031d6b 100644 --- a/nodes.py +++ b/nodes.py @@ -1,8 +1,7 @@ import torch -import gc import os from PIL import Image -from .utils import get_first_folder_list, tensor2pil, pil2tensor, encodeDiffOutpaintPrompt, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram +from .utils import get_first_folder_list, tensor2pil, pil2tensor, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram # Get the absolute path of various directories @@ -186,33 +185,44 @@ class EncodeDiffusersOutpaintPrompt: return { "required": { "diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}), - "extra_prompt": ("STRING", {"default": "", "tooltip": "The extra prompt to append, describing attributes etc. you want to include in the image. Default: \"(extra_prompt), high quality, 4k\""}), + "text": ("STRING", {"multiline": True, "dynamicPrompts": True, "tooltip": "The text to be encoded."}), + "clip": ("CLIP", {"tooltip": "The CLIP model used for encoding the text."}) } } - RETURN_TYPES = ("PIPE","CONDITIONING",) - RETURN_NAMES = ("diffusers_outpaint_pipe","diffusers_outpaint_conditioning",) + RETURN_NAMES = ("diffusers_outpaint_pipe","diffusers_conditioning",) + OUTPUT_TOOLTIPS = ("A conditioning containing the embedded text used to guide the diffusion model.",) FUNCTION = "encode" CATEGORY = "DiffusersOutpaint" + DESCRIPTION = "Encodes a text prompt using a CLIP model into an embedding that can be used to guide the diffusion model towards generating specific images." - def encode(self, diffusers_outpaint_pipe, extra_prompt=None): - model_path = diffusers_outpaint_pipe["model_path"] + def encode(self, diffusers_outpaint_pipe, text, clip): dtype = diffusers_outpaint_pipe["dtype"] device = diffusers_outpaint_pipe["device"] - - final_prompt = f"{extra_prompt}, high quality, 4k" - prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds = encodeDiffOutpaintPrompt(model_path, dtype, final_prompt, device) + text = f"{text}, high quality, 4k" + tokens = clip.tokenize(text) + output = clip.encode_from_tokens(tokens, return_pooled=True, return_dict=True) + prompt_embeds = output.pop("cond") - diffusers_outpaint_conditioning = { + prompt_embeds = prompt_embeds.to(device, dtype=dtype) + pooled_prompt_embeds = output["pooled_output"].to(device, dtype=dtype) + + bs_embed, seq_len, _ = prompt_embeds.shape + + # duplicate text embeddings for each generation per prompt, using mps friendly method + prompt_embeds = prompt_embeds.repeat(1, 1, 1) + prompt_embeds = prompt_embeds.view(bs_embed * 1, seq_len, -1) + + pooled_prompt_embeds = pooled_prompt_embeds.repeat(1, 1).view(bs_embed * 1, -1) + + diffusers_conditioning = { "prompt_embeds": prompt_embeds, - "negative_prompt_embeds": negative_prompt_embeds, "pooled_prompt_embeds": pooled_prompt_embeds, - "negative_pooled_prompt_embeds": negative_pooled_prompt_embeds } - - return (diffusers_outpaint_pipe,diffusers_outpaint_conditioning,) + return (diffusers_outpaint_pipe,diffusers_conditioning,) + class DiffusersImageOutpaint: @classmethod @@ -220,7 +230,8 @@ class DiffusersImageOutpaint: return { "required": { "diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}), - "diffusers_outpaint_conditioning": ("CONDITIONING", {"tooltip": "The prompt describing what you want."}), + "positive": ("CONDITIONING", {"tooltip": "The prompt describing what you want."}), + "negative": ("CONDITIONING", {"tooltip": "The prompt describing what you don't want."}), "diffuser_outpaint_cnet_image": ("IMAGE", {"tooltip": "The image to outpaint."}), "guidance_scale": ("FLOAT", {"default": 1.50, "min": 1.01, "max": 10, "step": 0.01, "tooltip": "The Classifier-Free Guidance scale balances creativity and adherence to the prompt. Higher values result in images more closely matching the prompt, however too high values will negatively impact quality."}), "controlnet_strength": ("FLOAT", {"default": 1.00, "min": 0.00, "max": 10, "step": 0.01}), @@ -233,7 +244,7 @@ class DiffusersImageOutpaint: FUNCTION = "sample" CATEGORY = "DiffusersOutpaint" - def sample(self, diffusers_outpaint_pipe, diffusers_outpaint_conditioning, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, seed, steps): + def sample(self, diffusers_outpaint_pipe, positive, negative, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, seed, steps): cnet_image = diffuser_outpaint_cnet_image cnet_image=tensor2pil(cnet_image) @@ -245,17 +256,18 @@ class DiffusersImageOutpaint: dtype = diffusers_outpaint_pipe["dtype"] device = diffusers_outpaint_pipe["device"] keep_model_device = diffusers_outpaint_pipe["keep_model_device"] - - prompt_embeds = diffusers_outpaint_conditioning["prompt_embeds"] - negative_prompt_embeds = diffusers_outpaint_conditioning["negative_prompt_embeds"] - pooled_prompt_embeds = diffusers_outpaint_conditioning["pooled_prompt_embeds"] - negative_pooled_prompt_embeds = diffusers_outpaint_conditioning["negative_pooled_prompt_embeds"] + + prompt_embeds = positive["prompt_embeds"] + pooled_prompt_embeds = positive["pooled_prompt_embeds"] + negative_prompt_embeds = negative["prompt_embeds"] + negative_pooled_prompt_embeds = negative["pooled_prompt_embeds"] last_rgb_latent = diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet_image, dtype, controlnet_path, prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds, device, steps, controlnet_strength, guidance_scale, keep_model_device) - del diffusers_outpaint_conditioning - clearVram(device) + del prompt_embeds, pooled_prompt_embeds, negative_prompt_embeds, negative_pooled_prompt_embeds + clearVram(device) + return ({"samples":last_rgb_latent},) From 1e1bac8d6e04db110f06a4587af97b895ea909d3 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 20:42:08 +0200 Subject: [PATCH 03/11] made available comfyui clip (pipeline) cleanup --- pipeline_fill_sd_xl.py | 172 ++--------------------------------------- 1 file changed, 8 insertions(+), 164 deletions(-) diff --git a/pipeline_fill_sd_xl.py b/pipeline_fill_sd_xl.py index 0356db8..8e24bd7 100644 --- a/pipeline_fill_sd_xl.py +++ b/pipeline_fill_sd_xl.py @@ -67,163 +67,6 @@ def retrieve_timesteps( return timesteps, num_inference_steps -def encode_prompt( - prompt: str, - tokenizer: None, - tokenizer_2: None, - text_encoder: None, - text_encoder_2: None, - device: Optional[torch.device] = None, - do_classifier_free_guidance: bool = True, - ): - prompt = [prompt] if isinstance(prompt, str) else prompt - - if prompt is not None: - batch_size = len(prompt) - - # Define tokenizers and text encoders - tokenizers = ( - [tokenizer, tokenizer_2] - if tokenizer is not None - else [tokenizer_2] - ) - text_encoders = ( - [text_encoder, text_encoder_2] - if text_encoder is not None - else [text_encoder_2] - ) - - prompt_2 = prompt - prompt_2 = [prompt_2] if isinstance(prompt_2, str) else prompt_2 - - # textual inversion: process multi-vector tokens if necessary - prompt_embeds_list = [] - prompts = [prompt, prompt_2] - for prompt, tokenizer, text_encoder in zip(prompts, tokenizers, text_encoders): - text_inputs = tokenizer( - prompt, - padding="max_length", - max_length=tokenizer.model_max_length, - truncation=True, - return_tensors="pt", - ) - - text_input_ids = text_inputs.input_ids - - prompt_embeds = text_encoder( - text_input_ids.to(device), output_hidden_states=True - ) - - # We are only ALWAYS interested in the pooled output of the final text encoder - pooled_prompt_embeds = prompt_embeds[0] - prompt_embeds = prompt_embeds.hidden_states[-2] - prompt_embeds_list.append(prompt_embeds) - - prompt_embeds = torch.concat(prompt_embeds_list, dim=-1) - - # get unconditional embeddings for classifier free guidance - zero_out_negative_prompt = True - negative_prompt_embeds = None - negative_pooled_prompt_embeds = None - - if do_classifier_free_guidance and zero_out_negative_prompt: - negative_prompt_embeds = torch.zeros_like(prompt_embeds) - negative_pooled_prompt_embeds = torch.zeros_like(pooled_prompt_embeds) - elif do_classifier_free_guidance and negative_prompt_embeds is None: - negative_prompt = "" - negative_prompt_2 = negative_prompt - - # normalize str to list - negative_prompt = ( - batch_size * [negative_prompt] - if isinstance(negative_prompt, str) - else negative_prompt - ) - negative_prompt_2 = ( - batch_size * [negative_prompt_2] - if isinstance(negative_prompt_2, str) - else negative_prompt_2 - ) - - uncond_tokens: List[str] - if prompt is not None and type(prompt) is not type(negative_prompt): - raise TypeError( - f"`negative_prompt` should be the same type to `prompt`, but got {type(negative_prompt)} !=" - f" {type(prompt)}." - ) - elif batch_size != len(negative_prompt): - raise ValueError( - f"`negative_prompt`: {negative_prompt} has batch size {len(negative_prompt)}, but `prompt`:" - f" {prompt} has batch size {batch_size}. Please make sure that passed `negative_prompt` matches" - " the batch size of `prompt`." - ) - else: - uncond_tokens = [negative_prompt, negative_prompt_2] - - negative_prompt_embeds_list = [] - for negative_prompt, tokenizer, text_encoder in zip( - uncond_tokens, tokenizers, text_encoders - ): - max_length = prompt_embeds.shape[1] - uncond_input = tokenizer( - negative_prompt, - padding="max_length", - max_length=max_length, - truncation=True, - return_tensors="pt", - ) - - negative_prompt_embeds = text_encoder( - uncond_input.input_ids.to(device), - output_hidden_states=True, - ) - # We are only ALWAYS interested in the pooled output of the final text encoder - negative_pooled_prompt_embeds = negative_prompt_embeds[0] - negative_prompt_embeds = negative_prompt_embeds.hidden_states[-2] - - negative_prompt_embeds_list.append(negative_prompt_embeds) - - negative_prompt_embeds = torch.concat(negative_prompt_embeds_list, dim=-1) - - prompt_embeds = prompt_embeds.to(dtype=text_encoder_2.dtype, device=device) - - bs_embed, seq_len, _ = prompt_embeds.shape - # duplicate text embeddings for each generation per prompt, using mps friendly method - prompt_embeds = prompt_embeds.repeat(1, 1, 1) - prompt_embeds = prompt_embeds.view(bs_embed * 1, seq_len, -1) - - if do_classifier_free_guidance: - # duplicate unconditional embeddings for each generation per prompt, using mps friendly method - seq_len = negative_prompt_embeds.shape[1] - - if text_encoder_2 is not None: - negative_prompt_embeds = negative_prompt_embeds.to( - dtype=text_encoder_2.dtype, device=device - ) - else: - negative_prompt_embeds = negative_prompt_embeds.to( - dtype=torch.float16, device=device - ) - - negative_prompt_embeds = negative_prompt_embeds.repeat(1, 1, 1) - negative_prompt_embeds = negative_prompt_embeds.view( - batch_size * 1, seq_len, -1 - ) - - pooled_prompt_embeds = pooled_prompt_embeds.repeat(1, 1).view(bs_embed * 1, -1) - if do_classifier_free_guidance: - negative_pooled_prompt_embeds = negative_pooled_prompt_embeds.repeat( - 1, 1 - ).view(bs_embed * 1, -1) - - return ( - prompt_embeds, - negative_prompt_embeds, - pooled_prompt_embeds, - negative_pooled_prompt_embeds, - ) - - class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): def __init__( @@ -291,7 +134,7 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): # corresponds to doing no classifier free guidance. @property def do_classifier_free_guidance(self): - return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None # UNET <---- + return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None @property def num_timesteps(self): @@ -302,10 +145,11 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): self, controlnet_model, device, + dtype, keep_model_device, prompt_embeds: torch.Tensor, - negative_prompt_embeds: torch.Tensor, pooled_prompt_embeds: torch.Tensor, + negative_prompt_embeds: torch.Tensor, negative_pooled_prompt_embeds: torch.Tensor, image: PipelineImageInput = None, num_inference_steps: int = 8, @@ -343,10 +187,10 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): num_channels_latents, height, width, - prompt_embeds.dtype, + dtype, device, ) - + # 7 Prepare added time ids & embeddings add_text_embeds = pooled_prompt_embeds @@ -376,7 +220,7 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): "time_ids": add_time_ids, "control_type": union_control_type, } - + controlnet_prompt_embeds = prompt_embeds.to(device) controlnet_added_cond_kwargs = added_cond_kwargs @@ -430,9 +274,9 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): )[0] if keep_model_device: self.unet.to('cpu') - except torch.cuda.OutOfMemoryError as e: # free vram when OOM + except torch.cuda.OutOfMemoryError as e: # Free vram when OOM self.unet.to('cpu') - print('\033[93m', 'Gpu is out of memory(爆显存了)!', '\033[0m') + print('\033[93m', 'Gpu is out of memory!', '\033[0m') raise e # perform guidance From 80a754cab45ca6b0aff3cc32830e4992d725629d Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 20:48:50 +0200 Subject: [PATCH 04/11] Update README.md --- README.md | 17 +++++------------ 1 file changed, 5 insertions(+), 12 deletions(-) diff --git a/README.md b/README.md index bc1b4a3..7ace908 100644 --- a/README.md +++ b/README.md @@ -2,13 +2,14 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o ![DiffusersImageOutpaint-Nodes-Screen](https://github.com/user-attachments/assets/2722e07c-1d6a-416e-a9d8-f26aaa9a45a7) -#### Update: -- You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. -- If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below). +#### Updates: +- 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. +- 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. +- 10/2024: If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below). ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. -- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model, but you need the specified vae and controlnet since the extension is hardcoded to use them. You can always change the code to use different models): +- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model, but you need the specified controlnet since the extension is hardcoded to use it, for now): | **main model** | **controlnet model** | | :-----: | :-----: | | **[Diffuser Model folder](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/tree/main)** (you can change this model) | **[Diffuser Controlnet folder](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/tree/main)** (you need this model) | @@ -17,14 +18,6 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o | config.json, diffusion_pytorch_model.fp16.safetensors | | | **Scheduler folder** | | | scheduler_config.json | | - | **Text encoder folder** | | - | config.json, ~model.fp16.safetensors~ -> model.safetensors | | - | **Text encoder 2 folder** | | - | config.json, ~model.fp16.safetensors~ -> model.safetensors | | - | **Tokenizer folder** | | - | merges.txt, special_tokens_map.json, tokenizer_config.json, vocab.json | | - | **Tokenizer 2 folder** | | - | merges.txt, special_tokens_map.json, tokenizer_config.json, vocab.json | | ## Overview - **Minimum VRAM**: 6 gb with 1280x720 image, rtx 3060, RealVisXL_V5.0_Lightning, sdxl-vae-fp16-fix, controlnet-union-sdxl-promax using `sequential_cpu_offload`, otherwise 8,3 gb; From 799d5c58f5d8eda9349d1db6e812486ce69fdb00 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 20:51:36 +0200 Subject: [PATCH 05/11] updated to use comfyui clip loader --- Diffusers-Outpaint-Workflow.json | 771 +++++++++++++++++-------------- 1 file changed, 415 insertions(+), 356 deletions(-) diff --git a/Diffusers-Outpaint-Workflow.json b/Diffusers-Outpaint-Workflow.json index 759f447..877a434 100644 --- a/Diffusers-Outpaint-Workflow.json +++ b/Diffusers-Outpaint-Workflow.json @@ -1,30 +1,176 @@ { - "last_node_id": 689, - "last_link_id": 1301, + "last_node_id": 591, + "last_link_id": 1259, "nodes": [ { - "id": 675, - "type": "LoadDiffusersOutpaintModels", + "id": 584, + "type": "DiffusersImageOutpaint", "pos": { - "0": -210, - "1": 160 + "0": 320, + "1": 90 }, "size": { - "0": 320, - "1": 154 + "0": 300, + "1": 214 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "link": 1255 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 1254 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 1258 + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "link": 1251 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 1241 + ], + "slot_index": 0 + } + ], + "title": "DiffusersImageOutpaint", + "properties": { + "Node name for S&R": "DiffusersImageOutpaint" + }, + "widgets_values": [ + 1.5, + 1, + 1009037337630565, + "randomize", + 8 + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 529, + "type": "PadImageForDiffusersOutpaint", + "pos": { + "0": 0, + "1": 390 + }, + "size": { + "0": 290, + "1": 150 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 1005 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + }, + { + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "links": [ + 1251 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "PadImageForDiffusersOutpaint" + }, + "widgets_values": [ + 720, + 1280, + "Top" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 531, + "type": "VAELoader", + "pos": { + "0": 370, + "1": 350 + }, + "size": { + "0": 260, + "1": 60 }, "flags": {}, "order": 0, "mode": 0, "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 1007 + ] + } + ], + "properties": { + "Node name for S&R": "VAELoader" + }, + "widgets_values": [ + "sdxl_vae.safetensors" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 534, + "type": "LoadDiffusersOutpaintModels", + "pos": { + "0": -480, + "1": 60 + }, + "size": { + "0": 320, + "1": 154 + }, + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], "outputs": [ { "name": "diffusers_outpaint_pipe", "type": "PIPE", "links": [ - 1264 + 1253, + 1257 ], - "shape": 3 + "slot_index": 0 } ], "properties": { @@ -41,24 +187,190 @@ "bgcolor": "#335" }, { - "id": 351, - "type": "PreviewImage", + "id": 588, + "type": "EncodeDiffusersOutpaintPrompt", "pos": { - "0": 970, - "1": 280 + "0": -110, + "1": 220 }, "size": { - "0": 510, - "1": 490 + "0": 400, + "1": 96 + }, + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "link": 1257 + }, + { + "name": "clip", + "type": "CLIP", + "link": 1256 + } + ], + "outputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "links": [], + "slot_index": 0 + }, + { + "name": "diffusers_conditioning", + "type": "CONDITIONING", + "links": [ + 1258 + ], + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "EncodeDiffusersOutpaintPrompt" + }, + "widgets_values": [ + "" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 580, + "type": "DualCLIPLoader", + "pos": { + "0": -420, + "1": 260 + }, + "size": { + "0": 260, + "1": 110 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 1252, + 1256 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "DualCLIPLoader" + }, + "widgets_values": [ + "clip_l.safetensors", + "model.fp16.safetensors", + "sdxl" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 530, + "type": "VAEDecode", + "pos": { + "0": 660, + "1": 90 + }, + "size": { + "0": 210, + "1": 46 }, "flags": {}, "order": 8, "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 1241 + }, + { + "name": "vae", + "type": "VAE", + "link": 1007 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1259 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 528, + "type": "LoadImage", + "pos": { + "0": -360, + "1": 430 + }, + "size": [ + 320, + 310 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1005 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "Emilia3.png", + "image" + ] + }, + { + "id": 351, + "type": "PreviewImage", + "pos": { + "0": 670, + "1": 190 + }, + "size": { + "0": 510, + "1": 490 + }, + "flags": {}, + "order": 9, + "mode": 0, "inputs": [ { "name": "images", "type": "IMAGE", - "link": 679 + "link": 1259 } ], "outputs": [], @@ -68,306 +380,29 @@ "widgets_values": [] }, { - "id": 402, - "type": "GetImageSizeAndCount", + "id": 587, + "type": "EncodeDiffusersOutpaintPrompt", "pos": { - "0": 1200, - "1": 150 + "0": -120, + "1": 70 }, "size": { - "0": 210, - "1": 90 - }, - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 1297 - } - ], - "outputs": [ - { - "name": "image", - "type": "IMAGE", - "links": [ - 679 - ], - "slot_index": 0, - "shape": 3 - }, - { - "name": "720 width", - "type": "INT", - "links": null, - "shape": 3 - }, - { - "name": "1280 height", - "type": "INT", - "links": null, - "shape": 3 - }, - { - "name": "1 count", - "type": "INT", - "links": null, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "GetImageSizeAndCount" - }, - "widgets_values": [] - }, - { - "id": 688, - "type": "VAEDecode", - "pos": { - "0": 960, - "1": 160 - }, - "size": { - "0": 210, - "1": 46 - }, - "flags": {}, - "order": 6, - "mode": 0, - "inputs": [ - { - "name": "samples", - "type": "LATENT", - "link": 1301 - }, - { - "name": "vae", - "type": "VAE", - "link": 1296 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 1297 - ], - "slot_index": 0, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "VAEDecode" - }, - "widgets_values": [] - }, - { - "id": 689, - "type": "DiffusersImageOutpaint", - "pos": { - "0": 620, - "1": 150 - }, - "size": { - "0": 320, - "1": 190 - }, - "flags": {}, - "order": 5, - "mode": 0, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 1298 - }, - { - "name": "diffusers_outpaint_conditioning", - "type": "CONDITIONING", - "link": 1299 - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "link": 1300 - } - ], - "outputs": [ - { - "name": "LATENT", - "type": "LATENT", - "links": [ - 1301 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "DiffusersImageOutpaint" - }, - "widgets_values": [ - 1.5, - 1, - 693728282818736, - "randomize", - 8 - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 683, - "type": "VAELoader", - "pos": { - "0": 630, - "1": 400 - }, - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "VAE", - "type": "VAE", - "links": [ - 1296 - ], - "slot_index": 0, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "VAELoader" - }, - "widgets_values": [ - "sdxl_vae.safetensors" - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 1, - "type": "LoadImage", - "pos": { - "0": -200, - "1": 360 - }, - "size": [ - 320, - 310 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 1194 - ], - "slot_index": 0, - "shape": 3 - }, - { - "name": "MASK", - "type": "MASK", - "links": [], - "slot_index": 1, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "20240930_201555.jpg", - "image" - ] - }, - { - "id": 653, - "type": "PadImageForDiffusersOutpaint", - "pos": { - "0": 290, - "1": 300 - }, - "size": { - "0": 290, - "1": 150 + "0": 400, + "1": 96 }, "flags": {}, "order": 4, "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 1194 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": null, - "shape": 3 - }, - { - "name": "MASK", - "type": "MASK", - "links": null, - "shape": 3 - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "links": [ - 1300 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "PadImageForDiffusersOutpaint" - }, - "widgets_values": [ - 720, - 1280, - "Top" - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 655, - "type": "EncodeDiffusersOutpaintPrompt", - "pos": { - "0": 130, - "1": 150 - }, - "size": { - "0": 463.6000061035156, - "1": 80 - }, - "flags": {}, - "order": 3, - "mode": 0, "inputs": [ { "name": "diffusers_outpaint_pipe", "type": "PIPE", - "link": 1264 + "link": 1253 + }, + { + "name": "clip", + "type": "CLIP", + "link": 1252 } ], "outputs": [ @@ -375,24 +410,24 @@ "name": "diffusers_outpaint_pipe", "type": "PIPE", "links": [ - 1298 + 1255 ], - "shape": 3 + "slot_index": 0 }, { - "name": "diffusers_outpaint_conditioning", + "name": "diffusers_conditioning", "type": "CONDITIONING", "links": [ - 1299 + 1254 ], - "shape": 3 + "slot_index": 1 } ], "properties": { "Node name for S&R": "EncodeDiffusersOutpaintPrompt" }, "widgets_values": [ - "sitting" + "" ], "color": "#232", "bgcolor": "#353" @@ -400,86 +435,110 @@ ], "links": [ [ - 679, - 402, + 1005, + 528, 0, - 351, + 529, 0, "IMAGE" ], [ - 1194, - 1, + 1007, + 531, 0, - 653, - 0, - "IMAGE" - ], - [ - 1264, - 675, - 0, - 655, - 0, - "PIPE" - ], - [ - 1296, - 683, - 0, - 688, + 530, 1, "VAE" ], [ - 1297, - 688, + 1241, + 584, 0, - 402, + 530, 0, + "LATENT" + ], + [ + 1251, + 529, + 2, + 584, + 3, "IMAGE" ], [ - 1298, - 655, + 1252, + 580, 0, - 689, + 587, + 1, + "CLIP" + ], + [ + 1253, + 534, + 0, + 587, 0, "PIPE" ], [ - 1299, - 655, + 1254, + 587, 1, - 689, + 584, 1, "CONDITIONING" ], [ - 1300, - 653, - 2, - 689, - 2, - "IMAGE" + 1255, + 587, + 0, + 584, + 0, + "PIPE" ], [ - 1301, - 689, + 1256, + 580, 0, - 688, + 588, + 1, + "CLIP" + ], + [ + 1257, + 534, 0, - "LATENT" + 588, + 0, + "PIPE" + ], + [ + 1258, + 588, + 1, + 584, + 2, + "CONDITIONING" + ], + [ + 1259, + 530, + 0, + 351, + 0, + "IMAGE" ] ], "groups": [], "config": {}, "extra": { "ds": { - "scale": 0.8769226950000005, + "scale": 0.7247295000000004, "offset": [ - 254.45705206435736, - -58.04927351924108 + 679.3196754354946, + 80.60613789648367 ] } }, From e5755a671a434c98c04279a6daa153adf527be84 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 21:03:36 +0200 Subject: [PATCH 06/11] Update README.md --- README.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/README.md b/README.md index 7ace908..6aedff8 100644 --- a/README.md +++ b/README.md @@ -7,6 +7,11 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o - 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. - 10/2024: If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below). +#### To do list to [change model used](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/14): +- [x] ComfyUI Clip Loader Node +- [ ] ComfyUI Load Diffusion Model Node +- [ ] ComfyUI Load Conotrolnet Model Node + ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. - Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model, but you need the specified controlnet since the extension is hardcoded to use it, for now): From e0b02be64bb09d2a378e111d6b321d06e47a2263 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 21:09:41 +0200 Subject: [PATCH 07/11] Add files via upload --- ...paint-RealXL_Checkpoint-Loader-Simple.json | 525 ++++++++++++++++++ 1 file changed, 525 insertions(+) create mode 100644 Diffusers-Outpaint-RealXL_Checkpoint-Loader-Simple.json diff --git a/Diffusers-Outpaint-RealXL_Checkpoint-Loader-Simple.json b/Diffusers-Outpaint-RealXL_Checkpoint-Loader-Simple.json new file mode 100644 index 0000000..24a467a --- /dev/null +++ b/Diffusers-Outpaint-RealXL_Checkpoint-Loader-Simple.json @@ -0,0 +1,525 @@ +{ + "last_node_id": 591, + "last_link_id": 1263, + "nodes": [ + { + "id": 530, + "type": "VAEDecode", + "pos": { + "0": 660, + "1": 90 + }, + "size": { + "0": 210, + "1": 46 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 1241 + }, + { + "name": "vae", + "type": "VAE", + "link": 1261 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1262 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + }, + "widgets_values": [] + }, + { + "id": 584, + "type": "DiffusersImageOutpaint", + "pos": { + "0": 320, + "1": 90 + }, + "size": { + "0": 300, + "1": 214 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "link": 1255 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 1254 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 1258 + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "link": 1251 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 1241 + ], + "slot_index": 0 + } + ], + "title": "DiffusersImageOutpaint", + "properties": { + "Node name for S&R": "DiffusersImageOutpaint" + }, + "widgets_values": [ + 1.5, + 1, + 43078817542338, + "randomize", + 8 + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 529, + "type": "PadImageForDiffusersOutpaint", + "pos": { + "0": 0, + "1": 390 + }, + "size": { + "0": 290, + "1": 150 + }, + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 1263 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": null + }, + { + "name": "MASK", + "type": "MASK", + "links": null + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "links": [ + 1251 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "PadImageForDiffusersOutpaint" + }, + "widgets_values": [ + 720, + 1280, + "Top" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 534, + "type": "LoadDiffusersOutpaintModels", + "pos": { + "0": -480, + "1": 60 + }, + "size": { + "0": 320, + "1": 154 + }, + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "links": [ + 1253, + 1257 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "LoadDiffusersOutpaintModels" + }, + "widgets_values": [ + "RealVisXL_V5.0_Lightning", + "controlnet-union-sdxl-1.0", + "auto", + "auto", + false + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 588, + "type": "EncodeDiffusersOutpaintPrompt", + "pos": { + "0": -110, + "1": 220 + }, + "size": { + "0": 400, + "1": 96 + }, + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "link": 1257 + }, + { + "name": "clip", + "type": "CLIP", + "link": 1259 + } + ], + "outputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "links": [], + "slot_index": 0 + }, + { + "name": "diffusers_conditioning", + "type": "CONDITIONING", + "links": [ + 1258 + ], + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "EncodeDiffusersOutpaintPrompt" + }, + "widgets_values": [ + "" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 351, + "type": "PreviewImage", + "pos": { + "0": 650, + "1": 180 + }, + "size": { + "0": 510, + "1": 490 + }, + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 1262 + } + ], + "outputs": [], + "properties": { + "Node name for S&R": "PreviewImage" + }, + "widgets_values": [] + }, + { + "id": 591, + "type": "LoadImage", + "pos": { + "0": -350, + "1": 400 + }, + "size": [ + 320, + 310 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 1263 + ], + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "20230403_183417.jpg", + "image" + ] + }, + { + "id": 587, + "type": "EncodeDiffusersOutpaintPrompt", + "pos": { + "0": -120, + "1": 70 + }, + "size": { + "0": 400, + "1": 96 + }, + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "link": 1253 + }, + { + "name": "clip", + "type": "CLIP", + "link": 1260 + } + ], + "outputs": [ + { + "name": "diffusers_outpaint_pipe", + "type": "PIPE", + "links": [ + 1255 + ], + "slot_index": 0 + }, + { + "name": "diffusers_conditioning", + "type": "CONDITIONING", + "links": [ + 1254 + ], + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "EncodeDiffusersOutpaintPrompt" + }, + "widgets_values": [ + "" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 589, + "type": "CheckpointLoaderSimple", + "pos": { + "0": -500, + "1": 250 + }, + "size": { + "0": 360, + "1": 100 + }, + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": null + }, + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 1259, + 1260 + ], + "slot_index": 1 + }, + { + "name": "VAE", + "type": "VAE", + "links": [ + 1261 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "realvisxlV50_v50LightningBakedvae.safetensors" + ], + "color": "#223", + "bgcolor": "#335" + } + ], + "links": [ + [ + 1241, + 584, + 0, + 530, + 0, + "LATENT" + ], + [ + 1251, + 529, + 2, + 584, + 3, + "IMAGE" + ], + [ + 1253, + 534, + 0, + 587, + 0, + "PIPE" + ], + [ + 1254, + 587, + 1, + 584, + 1, + "CONDITIONING" + ], + [ + 1255, + 587, + 0, + 584, + 0, + "PIPE" + ], + [ + 1257, + 534, + 0, + 588, + 0, + "PIPE" + ], + [ + 1258, + 588, + 1, + 584, + 2, + "CONDITIONING" + ], + [ + 1259, + 589, + 1, + 588, + 1, + "CLIP" + ], + [ + 1260, + 589, + 1, + 587, + 1, + "CLIP" + ], + [ + 1261, + 589, + 2, + 530, + 1, + "VAE" + ], + [ + 1262, + 530, + 0, + 351, + 0, + "IMAGE" + ], + [ + 1263, + 591, + 0, + 529, + 0, + "IMAGE" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.8769226950000005, + "offset": [ + 570.3286926097892, + 6.798044267632111 + ] + } + }, + "version": 0.4 +} \ No newline at end of file From 6e1a73f9869370156d17ea5577e95a16e2461ff6 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 21:12:59 +0200 Subject: [PATCH 08/11] Update README.md --- README.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index 6aedff8..c697b50 100644 --- a/README.md +++ b/README.md @@ -1,9 +1,10 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-outpaint](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) by fffiloni. -![DiffusersImageOutpaint-Nodes-Screen](https://github.com/user-attachments/assets/2722e07c-1d6a-416e-a9d8-f26aaa9a45a7) +![image](https://github.com/user-attachments/assets/8f7665a1-dd8c-44d6-a067-fcc3f48b1865) + #### Updates: -- 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. +- 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. You can also use the Checkpoint Loader Simple node, to skip the clip selection part. - 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. - 10/2024: If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below). From 462ea6bfc3e7b2b40d2407314cfb1a2a2cf6ddf6 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Sun, 20 Oct 2024 21:18:45 +0200 Subject: [PATCH 09/11] Update README.md --- README.md | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/README.md b/README.md index c697b50..e9c483d 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,6 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o #### Updates: - 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. You can also use the Checkpoint Loader Simple node, to skip the clip selection part. - 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. -- 10/2024: If your `text_encoder` and `text_encoder_2` names contain `.fp16.` or other things before `safetensors`, you need to remove it (see the table below). #### To do list to [change model used](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/14): - [x] ComfyUI Clip Loader Node @@ -15,7 +14,7 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. -- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model, but you need the specified controlnet since the extension is hardcoded to use it, for now): +- Download models in the **`comfyui/models/diffusion_models`** folder, following the grid below (you can use the links to download the suggested models; you can also change the main model ([RealvisXLv50-bakedVae on Civitai](https://civitai.com/models/139562/realvisxl-v50)), but you need the specified controlnet since the extension is hardcoded to use it, for now): | **main model** | **controlnet model** | | :-----: | :-----: | | **[Diffuser Model folder](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/tree/main)** (you can change this model) | **[Diffuser Controlnet folder](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/tree/main)** (you need this model) | From d28cb563dcfd97ebe3cb3c909dff316f7072d4e8 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Tue, 22 Oct 2024 20:03:01 +0200 Subject: [PATCH 10/11] to do list stopped and guide to change model used --- README.md | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index e9c483d..522ef64 100644 --- a/README.md +++ b/README.md @@ -4,13 +4,16 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o #### Updates: +- 22/10/2024: + - Unet and Controlnet Models Loader using ComfYUI nodes canceled, since I can't find a way to load them properly; more info at the end. + - Guide to change model used. - 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. You can also use the Checkpoint Loader Simple node, to skip the clip selection part. - 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. #### To do list to [change model used](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/14): -- [x] ComfyUI Clip Loader Node -- [ ] ComfyUI Load Diffusion Model Node -- [ ] ComfyUI Load Conotrolnet Model Node +- - [x] ComfyUI Clip Loader Node +- ~[ ] ComfyUI Load Diffusion Model Node~ +- ~[ ] ComfyUI Load Conotrolnet Model Node~ ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. @@ -36,5 +39,12 @@ The extension gives 4 nodes: - You can also pass image and mask to `vae encode (for inpainting)` node, then pass the latent to a `sampler`, but controlnets and ip-adapters are harder to use compared to diffusers outpaint. +### Change model used +- **Main model**: On huggingface, choose a model from [text2image models](https://huggingface.co/models?pipeline_tag=text-to-image&sort=trending), then create a new folder named after it in `comfyui/models/diffusion_models`, then download in it the subfolders `unet` and `scheduler`. +- **Controlnet model**: same as above, except you don't need the `scheduler`. + +#### Unet and Controlnet Models Loader using ComfYUI nodes canceled +Let me clarify: I _can_ load them, it's just that they don't work in the code, I don't understand how they are loaded differently, maybe this has something to do with the auto config (comfyui loaders don't ask you config files), or maybe it uses different classes, anyway whenever I pass the models loaded via comfyui, they give errors, so I can't use them. + ## Credits diffusers-image-outpaint by [fffiloni](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) From 6b71cee8af90e60a3c8460eb7dd31a3601ad8e30 Mon Sep 17 00:00:00 2001 From: GiusTex <112352961+GiusTex@users.noreply.github.com> Date: Thu, 14 Nov 2024 18:57:23 +0100 Subject: [PATCH 11/11] specify how comfyui load diffusers models --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 522ef64..b2fe597 100644 --- a/README.md +++ b/README.md @@ -44,7 +44,7 @@ The extension gives 4 nodes: - **Controlnet model**: same as above, except you don't need the `scheduler`. #### Unet and Controlnet Models Loader using ComfYUI nodes canceled -Let me clarify: I _can_ load them, it's just that they don't work in the code, I don't understand how they are loaded differently, maybe this has something to do with the auto config (comfyui loaders don't ask you config files), or maybe it uses different classes, anyway whenever I pass the models loaded via comfyui, they give errors, so I can't use them. +I can load them but then they don't work in the inference code, since comfyui load diffusers models in a different format ([reddit post](https://www.reddit.com/r/comfyui/comments/17fvb49/comment/k6cz9yv/?utm_source=share&utm_medium=web3x&utm_name=web3xcss&utm_term=1&utm_content=share_button)). ## Credits diffusers-image-outpaint by [fffiloni](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main)