From 4e230d69f13c2d61f1df7bdf38130922691a094e Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 16:03:02 +0200 Subject: [PATCH 01/19] Update nodes.py --- nodes.py | 142 +++++++++++++++++++++++++++++++++++-------------------- 1 file changed, 91 insertions(+), 51 deletions(-) diff --git a/nodes.py b/nodes.py index a46d274..c00dd7f 100644 --- a/nodes.py +++ b/nodes.py @@ -173,44 +173,92 @@ class PadImageForDiffusersOutpaint: return (new_image, tensor_mask, tensor_cnet_image,) -class LoadDiffusersOutpaintModels: +class LoadDiffuserModel: @classmethod def INPUT_TYPES(s): return { "required": { - "model": (get_first_folder_list("diffusion_models"), {"default": "RealVisXL_V5.0_Lightning", "tooltip": "The diffuser model used for denoising the input latent. (Put model files in a folder, in diffusion_models folder)."}), - "controlnet_model": (get_first_folder_list("diffusion_models"), {"default": "controlnet-union-sdxl-1.0", "tooltip": "The controlnet model used for denoising the input latent. (Put model files in a folder, in diffusion_models folder)."}), + "unet_name": (folder_paths.get_filename_list("diffusion_models"), {"tooltip": "The name of the unet (model) to load."}), "device": (["auto", "cuda", "cpu", "mps", "xpu", "meta"],{"default": "auto", "tooltip": "Device for inference, default is auto checked by comfyui"}), "dtype": (["auto","fp16","bf16","fp32", "fp8_e4m3fn", "fp8_e4m3fnuz", "fp8_e5m2", "fp8_e5m2fnuz"],{"default":"auto", "tooltip": "Model precision for inference, default is auto checked by comfyui"}), - "sequential_cpu_offload": ("BOOLEAN", {"default": False, "tooltip": "Inference by default needs around 8gb vram, if this option is on it will move controlnet and unet back and forth between cpu and vram, to have only one model loaded at a time (around 6 gb vram used), useful for gpus under 8gb but will impact inference speed."}), + "model_type": (get_config_folder_list("configs"), {"default": "sdxl", "tooltip": "The json configs used for the unet. (Put unet config in \"configs/your model type/unet\", and scheduler config in \"configs/your model type/scheduler\")."}), }, } - RETURN_TYPES = ("PIPE",) - RETURN_NAMES = ("diffusers_outpaint_pipe",) + RETURN_TYPES = ("MODEL", "SCHEDULER") + RETURN_NAMES = ("model", "scheduler configs") FUNCTION = "load" CATEGORY = "DiffusersOutpaint" - def load(self, model, controlnet_model, device, dtype, sequential_cpu_offload): + def load(self, unet_name, device, dtype, model_type): + # Go 2 folders back comfy_dir = os.path.dirname(os.path.dirname(my_dir)) - - model_path = f"{comfy_dir}/models/diffusion_models/{model}" - controlnet_path = f"{comfy_dir}/models/diffusion_models/{controlnet_model}" - + unet_path = f"D:/models/diffusion_models/{unet_name}" + device = get_device_by_name(device) dtype = get_dtype_by_name(dtype) - diffusers_outpaint_pipe = { - "model_path": model_path, - "controlnet_model": controlnet_model, - "controlnet_path": controlnet_path, - "device": device, - "dtype": dtype, - "keep_model_device": sequential_cpu_offload, - } + if model_type == "sdxl": + print("Loading sdxl unet...") + + unet = UNet2DConditionModel.from_config(f"{comfy_dir}/custom_nodes/ComfyUI-DiffusersImageOutpaint/configs", subfolder=f"{model_type}/unet").to(device, dtype) + unet.load_state_dict(load_file(unet_path)) - return (diffusers_outpaint_pipe,) + scheduler = TCDScheduler.from_config(f"{comfy_dir}/custom_nodes/ComfyUI-DiffusersImageOutpaint/configs", subfolder=f"{model_type}/scheduler") + + scale_model_input_method = test_scheduler_scale_model_input(comfy_dir, model_type) + + scheduler_configs = { + "scheduler": scheduler, + "scale_model_input_method": scale_model_input_method, + } + + return (unet, scheduler_configs,) + + +class LoadDiffuserControlnet: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "controlnet_model": (folder_paths.get_filename_list("controlnet"), {"tooltip": "The controlnet model used for denoising the input latent."}), + "device": (["auto", "cuda", "cpu", "mps", "xpu", "meta"],{"default": "auto", "tooltip": "Device for inference, default is auto checked by comfyui"}), + "dtype": (["auto","fp16","bf16","fp32", "fp8_e4m3fn", "fp8_e4m3fnuz", "fp8_e5m2", "fp8_e5m2fnuz"],{"default":"auto", "tooltip": "Model precision for inference, default is auto checked by comfyui"}), + "controlnet_type": (get_config_folder_list("configs"), {"default": "controlnet-sdxl-promax", "tooltip": "The json configs used for controlnet. (Put config(s) in \"configs/your controlnet type\")."}), + }, + } + + RETURN_TYPES = ("CONTROL_NET",) + FUNCTION = "load" + CATEGORY = "DiffusersOutpaint" + + def load(self, controlnet_model, device, dtype, controlnet_type): + + # Go 2 folders back + comfy_dir = os.path.dirname(os.path.dirname(my_dir)) + #controlnet_path = f"D:/models/controlnet/{controlnet_model}" <-----------FIX + + device = get_device_by_name(device) + dtype = get_dtype_by_name(dtype) + + if controlnet_type == "controlnet-sdxl-promax": + print("Loading controlnet-sdxl-promax...") + controlnet_model = ControlNetModel_Union.from_config(f"{comfy_dir}/custom_nodes/ComfyUI-DiffusersImageOutpaint/configs/{controlnet_type}/config_promax.json") + + state_dict = load_state_dict(load_file(controlnet_path)) + + model, _, _, _, _ = ControlNetModel_Union._load_pretrained_model( + controlnet_model, state_dict, controlnet_path, controlnet_path + ) + + controlnet_model.to(device, dtype) + + del model, state_dict, controlnet_path + + clearVram(device) + + return (controlnet_model,) class EncodeDiffusersOutpaintPrompt: @@ -218,21 +266,22 @@ class EncodeDiffusersOutpaintPrompt: def INPUT_TYPES(s): return { "required": { - "diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}), + "device": (["auto", "cuda", "cpu", "mps", "xpu", "meta"],{"default": "auto", "tooltip": "Device for inference, default is auto checked by comfyui"}), + "dtype": (["auto","fp16","bf16","fp32", "fp8_e4m3fn", "fp8_e4m3fnuz", "fp8_e5m2", "fp8_e5m2fnuz"],{"default":"auto", "tooltip": "Model precision for inference, default is auto checked by comfyui"}), "text": ("STRING", {"multiline": True, "dynamicPrompts": True, "tooltip": "The text to be encoded."}), "clip": ("CLIP", {"tooltip": "The CLIP model used for encoding the text."}) } } - RETURN_TYPES = ("PIPE","CONDITIONING",) - RETURN_NAMES = ("diffusers_outpaint_pipe","diffusers_conditioning",) + RETURN_TYPES = ("CONDITIONING",) + RETURN_NAMES = ("diffusers_conditioning",) OUTPUT_TOOLTIPS = ("A conditioning containing the embedded text used to guide the diffusion model.",) FUNCTION = "encode" CATEGORY = "DiffusersOutpaint" DESCRIPTION = "Encodes a text prompt using a CLIP model into an embedding that can be used to guide the diffusion model towards generating specific images." - def encode(self, diffusers_outpaint_pipe, text, clip): - dtype = diffusers_outpaint_pipe["dtype"] - device = diffusers_outpaint_pipe["device"] + def encode(self, device, dtype, text, clip): + device = get_device_by_name(device) + dtype = get_dtype_by_name(dtype) text = f"{text}, high quality, 4k" tokens = clip.tokenize(text) @@ -255,7 +304,7 @@ class EncodeDiffusersOutpaintPrompt: "pooled_prompt_embeds": pooled_prompt_embeds, } - return (diffusers_outpaint_pipe,diffusers_conditioning,) + return (diffusers_conditioning,) class DiffusersImageOutpaint: @@ -263,45 +312,36 @@ class DiffusersImageOutpaint: def INPUT_TYPES(s): return { "required": { - "diffusers_outpaint_pipe": ("PIPE", {"tooltip": "Load the diffusers outpaint models."}), + "model": ("MODEL", {"tooltip": "The model used for denoising the input latent."}), + "scheduler_configs": ("SCHEDULER",), + "control_net": ("CONTROL_NET",), "positive": ("CONDITIONING", {"tooltip": "The prompt describing what you want."}), "negative": ("CONDITIONING", {"tooltip": "The prompt describing what you don't want."}), "diffuser_outpaint_cnet_image": ("IMAGE", {"tooltip": "The image to outpaint."}), "guidance_scale": ("FLOAT", {"default": 1.50, "min": 1.01, "max": 10, "step": 0.01, "tooltip": "The Classifier-Free Guidance scale balances creativity and adherence to the prompt. Higher values result in images more closely matching the prompt, however too high values will negatively impact quality."}), "controlnet_strength": ("FLOAT", {"default": 1.00, "min": 0.00, "max": 10, "step": 0.01}), - "seed": ("INT", {"default": 0, "min": 0, "max": 0xffffffffffffffff, "tooltip": "Fake seed, workaround used to keep generating different outpaints. Set to -1 to generate different images, or a fixed number to stop that."}), "steps": ("INT", {"default": 8, "min": 4, "max": 20, "tooltip": "The number of steps used in the denoising process."}), - } + "device": (["auto", "cuda", "cpu", "mps", "xpu", "meta"],{"default": "auto", "tooltip": "Device for inference, default is auto checked by comfyui"}), + "dtype": (["auto","fp16","bf16","fp32", "fp8_e4m3fn", "fp8_e4m3fnuz", "fp8_e5m2", "fp8_e5m2fnuz"],{"default":"auto", "tooltip": "Model precision for inference, default is auto checked by comfyui"}), + "sequential_cpu_offload": ("BOOLEAN", {"default": False, "tooltip": "Inference by default needs around 8gb vram, if this option is on it will move controlnet and unet back and forth between cpu and vram, to have only one model loaded at a time (around 6 gb vram used), useful for gpus under 8gb but will impact inference speed."}), + }, } RETURN_TYPES = ("LATENT",) FUNCTION = "sample" CATEGORY = "DiffusersOutpaint" - - def sample(self, diffusers_outpaint_pipe, positive, negative, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, seed, steps): - + + def sample(self, device, dtype, sequential_cpu_offload, scheduler_configs, model, control_net, positive, negative, diffuser_outpaint_cnet_image, guidance_scale, controlnet_strength, steps): cnet_image = diffuser_outpaint_cnet_image cnet_image=tensor2pil(cnet_image) cnet_image=cnet_image.convert('RGB') - model_path = diffusers_outpaint_pipe["model_path"] - controlnet_model = diffusers_outpaint_pipe["controlnet_model"] - controlnet_path = diffusers_outpaint_pipe["controlnet_path"] - dtype = diffusers_outpaint_pipe["dtype"] - device = diffusers_outpaint_pipe["device"] - keep_model_device = diffusers_outpaint_pipe["keep_model_device"] - - prompt_embeds = positive["prompt_embeds"] - pooled_prompt_embeds = positive["pooled_prompt_embeds"] - negative_prompt_embeds = negative["prompt_embeds"] - negative_pooled_prompt_embeds = negative["pooled_prompt_embeds"] + keep_model_device = sequential_cpu_offload - last_rgb_latent = diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet_image, dtype, controlnet_path, - prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds, - device, steps, controlnet_strength, guidance_scale, - keep_model_device) + scheduler = scheduler_configs["scheduler"] + scale_model_input_method = scheduler_configs["scale_model_input_method"] + + last_rgb_latent = diffuserOutpaintSamples(device, dtype, keep_model_device, scheduler, scale_model_input_method, model, control_net, positive, negative, + cnet_image, controlnet_strength, guidance_scale, steps) - del prompt_embeds, pooled_prompt_embeds, negative_prompt_embeds, negative_pooled_prompt_embeds - clearVram(device) - return ({"samples":last_rgb_latent},) From 67ce39717c56fd2902e9014d7010b9a9f720ea3b Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 16:06:52 +0200 Subject: [PATCH 02/19] Update __init__.py --- __init__.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/__init__.py b/__init__.py index c789f9e..07f918c 100644 --- a/__init__.py +++ b/__init__.py @@ -1,15 +1,17 @@ -from .nodes import (PadImageForDiffusersOutpaint, LoadDiffusersOutpaintModels, EncodeDiffusersOutpaintPrompt, DiffusersImageOutpaint) +from .nodes import (PadImageForDiffusersOutpaint, LoadDiffuserModel, LoadDiffuserControlnet, EncodeDiffusersOutpaintPrompt, DiffusersImageOutpaint) NODE_CLASS_MAPPINGS = { "PadImageForDiffusersOutpaint": PadImageForDiffusersOutpaint, - "LoadDiffusersOutpaintModels": LoadDiffusersOutpaintModels, + "LoadDiffuserModel": LoadDiffuserModel, + "LoadDiffuserControlnet": LoadDiffuserControlnet, "EncodeDiffusersOutpaintPrompt": EncodeDiffusersOutpaintPrompt, "DiffusersImageOutpaint": DiffusersImageOutpaint } NODE_DISPLAY_NAME_MAPPINGS = { "PadImageForDiffusersOutpaint": "Pad Image For Diffusers Outpaint", - "LoadDiffusersOutpaintModels": "Load Diffusers Outpaint Models", + "LoadDiffuserModel": "Load Diffuser Model", + "LoadDiffuserControlnet": "Load Diffuser Controlnet", "EncodeDiffusersOutpaintPrompt": "Encode Diffusers Outpaint Prompt", "DiffusersImageOutpaint": "Diffusers Image Outpaint" } From 47270ff06e30ec0a4d72774062c98d65e8795658 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 16:11:43 +0200 Subject: [PATCH 03/19] Update utils.py --- utils.py | 123 ++++++++++++++++++++++--------------------------------- 1 file changed, 49 insertions(+), 74 deletions(-) diff --git a/utils.py b/utils.py index 8eebb2a..3eefd77 100644 --- a/utils.py +++ b/utils.py @@ -7,12 +7,7 @@ import comfy.model_management as mm from PIL import Image from folder_paths import map_legacy, folder_names_and_paths -from .controlnet_union import ControlNetModel_Union from .pipeline_fill_sd_xl import StableDiffusionXLFillPipeline -from diffusers import AutoencoderKL, TCDScheduler -from diffusers.models.model_loading_utils import load_state_dict -from transformers import CLIPTextModel, CLIPTextModelWithProjection, CLIPTokenizer -from diffusers import UNet2DConditionModel def get_first_folder_list(folder_name: str) -> tuple[list[str], dict[str, float], float]: @@ -23,9 +18,17 @@ def get_first_folder_list(folder_name: str) -> tuple[list[str], dict[str, float] root_folder = folders[0][0] elif folder_name == "diffusion_models": root_folder = folders[0][1] + elif folder_name == "controlnet": + root_folder = folders[0][0] visible_folders = [name for name in os.listdir(root_folder) if os.path.isdir(os.path.join(root_folder, name))] return visible_folders +def get_config_folder_list(folder_name: str) -> tuple[list[str], dict[str, float], float]: + my_dir = os.path.dirname(os.path.abspath(__file__)) + configs_dir = f"{my_dir}/{folder_name}" + + folders = [f for f in os.listdir(configs_dir) if os.path.isdir(os.path.join(configs_dir, f))] + return folders # Tensor to PIL (grabbed from WAS Suite) def tensor2pil(image: torch.Tensor) -> Image.Image: @@ -67,16 +70,6 @@ def get_dtype_by_name(dtype): return dtype - -def loadDiffModels1(model_path, dtype, device): - tokenizer = CLIPTokenizer.from_pretrained(model_path, subfolder="tokenizer", use_fast=False) - tokenizer_2 = CLIPTokenizer.from_pretrained(model_path, subfolder="tokenizer_2", use_fast=False) - text_encoder = CLIPTextModel.from_pretrained(model_path, subfolder="text_encoder", torch_dtype=dtype).requires_grad_(False).to(device) - text_encoder_2 = CLIPTextModelWithProjection.from_pretrained(model_path, subfolder="text_encoder_2", torch_dtype=dtype).requires_grad_(False).to(device) - - return tokenizer, tokenizer_2, text_encoder, text_encoder_2 - - def clearVram(device): gc.collect() @@ -91,73 +84,51 @@ def clearVram(device): torch.xpu.empty_cache() elif device.type == "meta": torch.meta.empty_cache() + + +class TCDScheduler_Custom: + def __init__(self, **kwargs): + for key, value in kwargs.items(): + setattr(self, key, value) - # torch.ipc_collect() not available, and ipc_collect seems available only for cuda - - -def loadControlnetModel(device, dtype, controlnet_path): - config_file = f"{controlnet_path}/config_promax.json" - config = ControlNetModel_Union.load_config(config_file) - controlnet_model = ControlNetModel_Union.from_config(config) + def scale_model_input(self, input, t): + scale_factor = getattr(self, 'scale_factor', 1) + return input * scale_factor - model_file = f"{controlnet_path}/diffusion_pytorch_model_promax.safetensors" - state_dict = load_state_dict(model_file) - - model, _, _, _, _ = ControlNetModel_Union._load_pretrained_model( - controlnet_model, state_dict, model_file, f"{controlnet_path}" - ) - controlnet_model.to(device, dtype) - - del model, state_dict, model_file + def __repr__(self): + attrs = {key: value for key, value in self.__dict__.items()} + return f"TCDScheduler({attrs})" - clearVram(device) - return controlnet_model +def test_scheduler_scale_model_input(comfy_dir, model_type): + scheduler_config_path = f"{comfy_dir}/custom_nodes/ComfyUI-DiffusersImageOutpaint/configs/{model_type}/scheduler/scheduler_config.json" - -def loadVaeModel(vae_path, device, dtype, enable_vae_slicing, enable_vae_tiling): - vae = AutoencoderKL.from_pretrained(f"{vae_path}").to(device, dtype) - if enable_vae_slicing: - vae.enable_slicing() - else: - vae.disable_slicing() + with open(scheduler_config_path, 'r') as f: + config = json.load(f) - if enable_vae_tiling: - vae.enable_tiling() - else: - vae.disable_tiling() - return vae + scheduler = TCDScheduler_Custom(**config) + scale_model_input_method = scheduler.scale_model_input + + return scale_model_input_method -def loadUnetModel(model_path, device, dtype): - unet = UNet2DConditionModel.from_pretrained(model_path, subfolder="unet", use_safetensors=True) - unet.to(device, dtype) - return unet - - -def diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet_image, dtype, controlnet_path, - prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds, - device, steps, controlnet_strength, guidance_scale, - keep_model_device): +def diffuserOutpaintSamples(device, dtype, keep_model_device, scheduler, scale_model_input_method, model, control_net, positive, negative, + cnet_image, controlnet_strength, guidance_scale, steps): - controlnet_model = loadControlnetModel(device, dtype, controlnet_path) - unet = loadUnetModel(model_path, device, dtype) - - with open(f"{model_path}/scheduler/scheduler_config.json", "r") as f: - scheduler_config = json.load(f) - scheduler = TCDScheduler.from_config(scheduler_config) + prompt_embeds = positive["prompt_embeds"] + pooled_prompt_embeds = positive["pooled_prompt_embeds"] + negative_prompt_embeds = negative["prompt_embeds"] + negative_pooled_prompt_embeds = negative["pooled_prompt_embeds"] + controlnet_model = control_net - pipe = StableDiffusionXLFillPipeline( - unet, - scheduler=scheduler, - ) - if not keep_model_device: - pipe.to(device) + device = get_device_by_name(device) + dtype = get_dtype_by_name(dtype) + + timesteps = None + unet = model + + pipe = StableDiffusionXLFillPipeline() - cnet_image = diffuser_outpaint_cnet_image - cnet_image=tensor2pil(cnet_image) - cnet_image=cnet_image.convert('RGB') - rgb_latents = list(pipe( prompt_embeds=prompt_embeds, negative_prompt_embeds=negative_prompt_embeds, @@ -170,13 +141,17 @@ def diffuserOutpaintSamples(model_path, controlnet_model, diffuser_outpaint_cnet guidance_scale=guidance_scale, device=device, dtype=dtype, + unet=unet, + timesteps=timesteps, + scale_model_input_method=scale_model_input_method, keep_model_device=keep_model_device, - )) + scheduler=scheduler, + )) last_rgb_latent = rgb_latents[-1] # Access the last image - del pipe, controlnet_model, scheduler, prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds - + del pipe, unet, controlnet_model, scheduler, prompt_embeds, negative_prompt_embeds, pooled_prompt_embeds, negative_pooled_prompt_embeds + clearVram(device) return last_rgb_latent From 41413fde5c390c905e59352eed33d6bde25405c9 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 16:45:15 +0200 Subject: [PATCH 04/19] Update nodes.py --- nodes.py | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/nodes.py b/nodes.py index c00dd7f..579e3c7 100644 --- a/nodes.py +++ b/nodes.py @@ -1,12 +1,31 @@ import torch import os from PIL import Image, ImageDraw -from .utils import get_first_folder_list, tensor2pil, pil2tensor, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram +from .utils import get_config_folder_list, tensor2pil, pil2tensor, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram, test_scheduler_scale_model_input +import folder_paths +from .unet_2d_condition import UNet2DConditionModel +from diffusers import TCDScheduler +from .controlnet_union import ControlNetModel_Union +from diffusers.models.model_loading_utils import load_state_dict +from safetensors.torch import load_file + +import logging # Get the absolute path of various directories my_dir = os.path.dirname(os.path.abspath(__file__)) +def update_folder_names_and_paths(key, targets=[]): + # check for existing key + base = folder_paths.folder_names_and_paths.get(key, ([], {})) + base = base[0] if isinstance(base[0], (list, set, tuple)) else [] + # find base key & add w/ fallback, sanity check + warning + target = next((x for x in targets if x in folder_paths.folder_names_and_paths), targets[0]) + orig, _ = folder_paths.folder_names_and_paths.get(target, ([], {})) + folder_paths.folder_names_and_paths[key] = (orig or base, {".gguf"}) + if base and base != orig: + logging.warning(f"Unknown file list already present on key {key}: {base}") + def can_expand(source_width, source_height, target_width, target_height, alignment): """Checks if the image can be expanded based on the alignment.""" if alignment in ("Left", "Right") and source_width >= target_width: From 3753fbffe9e0b1ea3af259e8db0494b8caee5ca4 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 16:55:25 +0200 Subject: [PATCH 05/19] Update pipeline_fill_sd_xl.py --- pipeline_fill_sd_xl.py | 37 +++++++++++++++---------------------- 1 file changed, 15 insertions(+), 22 deletions(-) diff --git a/pipeline_fill_sd_xl.py b/pipeline_fill_sd_xl.py index 8e24bd7..297f74a 100644 --- a/pipeline_fill_sd_xl.py +++ b/pipeline_fill_sd_xl.py @@ -17,12 +17,10 @@ from typing import List, Optional, Union import cv2 import PIL.Image import torch -import gc from diffusers.image_processor import PipelineImageInput, VaeImageProcessor -from diffusers.models import AutoencoderKL, UNet2DConditionModel -from diffusers.pipelines.pipeline_utils import DiffusionPipeline, StableDiffusionMixin from diffusers.schedulers import KarrasDiffusionSchedulers from diffusers.utils.torch_utils import randn_tensor +from tqdm import tqdm from .controlnet_union import ControlNetModel_Union from comfy.utils import ProgressBar @@ -71,17 +69,9 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): def __init__( self, - unet: UNet2DConditionModel, - scheduler: KarrasDiffusionSchedulers, - force_zeros_for_empty_prompt: bool = True, ): super().__init__() - self.register_modules( - unet=unet, - scheduler=scheduler, - ) - self.vae_scale_factor = 8 self.image_processor = VaeImageProcessor( vae_scale_factor=self.vae_scale_factor, do_convert_rgb=True @@ -91,10 +81,6 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): do_convert_rgb=True, do_normalize=False, ) - self.register_to_config( - force_zeros_for_empty_prompt=force_zeros_for_empty_prompt - ) - self.controlnet_model = None def prepare_image(self, image, device, dtype, do_classifier_free_guidance=False): image = self.control_image_processor.preprocess(image).to(dtype=torch.float32) @@ -134,7 +120,10 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): # corresponds to doing no classifier free guidance. @property def do_classifier_free_guidance(self): - return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None + if hasattr(self.unet, 'config'): + return self._guidance_scale > 1 and self.unet.config.time_cond_proj_dim is None + else: + return self._guidance_scale > 1 @property def num_timesteps(self): @@ -147,6 +136,10 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): device, dtype, keep_model_device, + scheduler: KarrasDiffusionSchedulers, + unet: object, + timesteps, + scale_model_input_method, prompt_embeds: torch.Tensor, pooled_prompt_embeds: torch.Tensor, negative_prompt_embeds: torch.Tensor, @@ -158,6 +151,11 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): ): self.controlnet = controlnet_model self._guidance_scale = guidance_scale + self.unet = unet + self.scheduler = scheduler + self.timesteps = timesteps + self.scale_model_input_method=scale_model_input_method + # 2. Define call parameters batch_size = 1 @@ -228,7 +226,7 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): num_warmup_steps = len(timesteps) - num_inference_steps * self.scheduler.order ComfyUI_ProgressBar = ProgressBar(int(num_inference_steps)) - with self.progress_bar(total=num_inference_steps) as progress_bar: + with tqdm(total=num_inference_steps) as pbar: for i, t in enumerate(timesteps): # expand the latents if we are doing classifier free guidance latent_model_input = ( @@ -318,10 +316,5 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): ComfyUI_ProgressBar.update(1) #yield latents_to_rgb(latents) - del self.unet - del self.controlnet - gc.collect() - torch.cuda.empty_cache() - latents = latents / 0.13025 yield latents From ca18274c11185d57a1f318b6b1550e9069f94a62 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:04:53 +0200 Subject: [PATCH 06/19] Update pipeline_fill_sd_xl.py --- pipeline_fill_sd_xl.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/pipeline_fill_sd_xl.py b/pipeline_fill_sd_xl.py index 297f74a..f447561 100644 --- a/pipeline_fill_sd_xl.py +++ b/pipeline_fill_sd_xl.py @@ -312,7 +312,8 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): if i == len(timesteps) - 1 or ( (i + 1) > num_warmup_steps and (i + 1) % self.scheduler.order == 0 ): - progress_bar.update() + #progress_bar.update() + pbar.update() ComfyUI_ProgressBar.update(1) #yield latents_to_rgb(latents) From 4475a217da471debdb7149fa4ef46bc045abdc17 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:06:17 +0200 Subject: [PATCH 07/19] Update pipeline_fill_sd_xl.py --- pipeline_fill_sd_xl.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/pipeline_fill_sd_xl.py b/pipeline_fill_sd_xl.py index f447561..04891c7 100644 --- a/pipeline_fill_sd_xl.py +++ b/pipeline_fill_sd_xl.py @@ -312,10 +312,8 @@ class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): if i == len(timesteps) - 1 or ( (i + 1) > num_warmup_steps and (i + 1) % self.scheduler.order == 0 ): - #progress_bar.update() pbar.update() ComfyUI_ProgressBar.update(1) - #yield latents_to_rgb(latents) latents = latents / 0.13025 yield latents From 59383257525022574e1c924a44089cbcf62d15c3 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:08:31 +0200 Subject: [PATCH 08/19] Update pipeline_fill_sd_xl.py --- pipeline_fill_sd_xl.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pipeline_fill_sd_xl.py b/pipeline_fill_sd_xl.py index 04891c7..16fdb16 100644 --- a/pipeline_fill_sd_xl.py +++ b/pipeline_fill_sd_xl.py @@ -65,7 +65,7 @@ def retrieve_timesteps( return timesteps, num_inference_steps -class StableDiffusionXLFillPipeline(DiffusionPipeline, StableDiffusionMixin): +class StableDiffusionXLFillPipeline: def __init__( self, From 9f75d79fe8bd8cf760231d8a9c5753f82e93b22d Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:10:38 +0200 Subject: [PATCH 09/19] Update nodes.py --- nodes.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/nodes.py b/nodes.py index 579e3c7..c92cb9c 100644 --- a/nodes.py +++ b/nodes.py @@ -3,7 +3,7 @@ import os from PIL import Image, ImageDraw from .utils import get_config_folder_list, tensor2pil, pil2tensor, diffuserOutpaintSamples, get_device_by_name, get_dtype_by_name, clearVram, test_scheduler_scale_model_input import folder_paths -from .unet_2d_condition import UNet2DConditionModel +from diffusers.models import UNet2DConditionModel from diffusers import TCDScheduler from .controlnet_union import ControlNetModel_Union from diffusers.models.model_loading_utils import load_state_dict From 5c04ccf05f41d4ce095e18f5ce12a73bec850c18 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:12:55 +0200 Subject: [PATCH 10/19] Add files via upload --- configs/controlnet-sdxl-promax/config.json | 56 +++++++++++++++ .../controlnet-sdxl-promax/config_promax.json | 57 +++++++++++++++ configs/sdxl/scheduler/scheduler_config.json | 19 +++++ configs/sdxl/unet/config.json | 71 +++++++++++++++++++ 4 files changed, 203 insertions(+) create mode 100644 configs/controlnet-sdxl-promax/config.json create mode 100644 configs/controlnet-sdxl-promax/config_promax.json create mode 100644 configs/sdxl/scheduler/scheduler_config.json create mode 100644 configs/sdxl/unet/config.json diff --git a/configs/controlnet-sdxl-promax/config.json b/configs/controlnet-sdxl-promax/config.json new file mode 100644 index 0000000..e29898b --- /dev/null +++ b/configs/controlnet-sdxl-promax/config.json @@ -0,0 +1,56 @@ +{ + "_class_name": "ControlNetModel", + "_diffusers_version": "0.20.0.dev0", + "act_fn": "silu", + "addition_embed_type": "text_time", + "addition_embed_type_num_heads": 64, + "addition_time_embed_dim": 256, + "attention_head_dim": [ + 5, + 10, + 20 + ], + "block_out_channels": [ + 320, + 640, + 1280 + ], + "class_embed_type": null, + "conditioning_channels": 3, + "conditioning_embedding_out_channels": [ + 16, + 32, + 96, + 256 + ], + "controlnet_conditioning_channel_order": "rgb", + "cross_attention_dim": 2048, + "down_block_types": [ + "DownBlock2D", + "CrossAttnDownBlock2D", + "CrossAttnDownBlock2D" + ], + "downsample_padding": 1, + "encoder_hid_dim": null, + "encoder_hid_dim_type": null, + "flip_sin_to_cos": true, + "freq_shift": 0, + "global_pool_conditions": false, + "in_channels": 4, + "layers_per_block": 2, + "mid_block_scale_factor": 1, + "norm_eps": 1e-05, + "norm_num_groups": 32, + "num_attention_heads": null, + "num_class_embeds": null, + "only_cross_attention": false, + "projection_class_embeddings_input_dim": 2816, + "resnet_time_scale_shift": "default", + "transformer_layers_per_block": [ + 1, + 2, + 10 + ], + "upcast_attention": null, + "use_linear_projection": true +} diff --git a/configs/controlnet-sdxl-promax/config_promax.json b/configs/controlnet-sdxl-promax/config_promax.json new file mode 100644 index 0000000..5419938 --- /dev/null +++ b/configs/controlnet-sdxl-promax/config_promax.json @@ -0,0 +1,57 @@ +{ + "_class_name": "ControlNetModel", + "_diffusers_version": "0.20.0.dev0", + "act_fn": "silu", + "addition_embed_type": "text_time", + "addition_embed_type_num_heads": 64, + "addition_time_embed_dim": 256, + "attention_head_dim": [ + 5, + 10, + 20 + ], + "block_out_channels": [ + 320, + 640, + 1280 + ], + "class_embed_type": null, + "conditioning_channels": 3, + "conditioning_embedding_out_channels": [ + 16, + 32, + 96, + 256 + ], + "controlnet_conditioning_channel_order": "rgb", + "cross_attention_dim": 2048, + "down_block_types": [ + "DownBlock2D", + "CrossAttnDownBlock2D", + "CrossAttnDownBlock2D" + ], + "downsample_padding": 1, + "encoder_hid_dim": null, + "encoder_hid_dim_type": null, + "flip_sin_to_cos": true, + "freq_shift": 0, + "global_pool_conditions": false, + "in_channels": 4, + "layers_per_block": 2, + "mid_block_scale_factor": 1, + "norm_eps": 1e-05, + "norm_num_groups": 32, + "num_attention_heads": null, + "num_class_embeds": null, + "only_cross_attention": false, + "projection_class_embeddings_input_dim": 2816, + "resnet_time_scale_shift": "default", + "transformer_layers_per_block": [ + 1, + 2, + 10 + ], + "upcast_attention": null, + "use_linear_projection": true, + "num_control_type": 8 +} diff --git a/configs/sdxl/scheduler/scheduler_config.json b/configs/sdxl/scheduler/scheduler_config.json new file mode 100644 index 0000000..b60b9ee --- /dev/null +++ b/configs/sdxl/scheduler/scheduler_config.json @@ -0,0 +1,19 @@ +{ + "_class_name": "DDIMScheduler", + "_diffusers_version": "0.30.0.dev0", + "beta_end": 0.012, + "beta_schedule": "scaled_linear", + "beta_start": 0.00085, + "clip_sample": false, + "clip_sample_range": 1.0, + "dynamic_thresholding_ratio": 0.995, + "num_train_timesteps": 1000, + "prediction_type": "epsilon", + "rescale_betas_zero_snr": false, + "sample_max_value": 1.0, + "set_alpha_to_one": false, + "steps_offset": 1, + "thresholding": false, + "timestep_spacing": "leading", + "trained_betas": null +} diff --git a/configs/sdxl/unet/config.json b/configs/sdxl/unet/config.json new file mode 100644 index 0000000..8eb8040 --- /dev/null +++ b/configs/sdxl/unet/config.json @@ -0,0 +1,71 @@ +{ + "_class_name": "UNet2DConditionModel", + "act_fn": "silu", + "addition_embed_type": "text_time", + "addition_embed_type_num_heads": 64, + "addition_time_embed_dim": 256, + "attention_head_dim": [ + 5, + 10, + 20 + ], + "attention_type": "default", + "block_out_channels": [ + 320, + 640, + 1280 + ], + "center_input_sample": false, + "class_embed_type": null, + "class_embeddings_concat": false, + "conv_in_kernel": 3, + "conv_out_kernel": 3, + "cross_attention_dim": 2048, + "cross_attention_norm": null, + "down_block_types": [ + "DownBlock2D", + "CrossAttnDownBlock2D", + "CrossAttnDownBlock2D" + ], + "downsample_padding": 1, + "dropout": 0.0, + "dual_cross_attention": false, + "encoder_hid_dim": null, + "encoder_hid_dim_type": null, + "flip_sin_to_cos": true, + "freq_shift": 0, + "in_channels": 4, + "layers_per_block": 2, + "mid_block_only_cross_attention": null, + "mid_block_scale_factor": 1, + "mid_block_type": "UNetMidBlock2DCrossAttn", + "norm_eps": 1e-05, + "norm_num_groups": 32, + "num_attention_heads": null, + "num_class_embeds": null, + "only_cross_attention": false, + "out_channels": 4, + "projection_class_embeddings_input_dim": 2816, + "resnet_out_scale_factor": 1.0, + "resnet_skip_time_act": false, + "resnet_time_scale_shift": "default", + "reverse_transformer_layers_per_block": null, + "sample_size": 128, + "time_cond_proj_dim": null, + "time_embedding_act_fn": null, + "time_embedding_dim": null, + "time_embedding_type": "positional", + "timestep_post_act": null, + "transformer_layers_per_block": [ + 1, + 2, + 10 + ], + "up_block_types": [ + "CrossAttnUpBlock2D", + "CrossAttnUpBlock2D", + "UpBlock2D" + ], + "upcast_attention": false, + "use_linear_projection": true +} From d2d18d08d35c5efdfbff7469849c7b74d1a71ca2 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 17:23:00 +0200 Subject: [PATCH 11/19] Update nodes.py --- nodes.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/nodes.py b/nodes.py index c92cb9c..a4ee500 100644 --- a/nodes.py +++ b/nodes.py @@ -213,8 +213,8 @@ class LoadDiffuserModel: # Go 2 folders back comfy_dir = os.path.dirname(os.path.dirname(my_dir)) - unet_path = f"D:/models/diffusion_models/{unet_name}" - + unet_path = folder_paths.get_full_path_or_raise("diffusion_models", unet_name) + device = get_device_by_name(device) dtype = get_dtype_by_name(dtype) @@ -256,8 +256,8 @@ class LoadDiffuserControlnet: # Go 2 folders back comfy_dir = os.path.dirname(os.path.dirname(my_dir)) - #controlnet_path = f"D:/models/controlnet/{controlnet_model}" <-----------FIX - + controlnet_path = folder_paths.get_full_path_or_raise("controlnet", controlnet_model) + device = get_device_by_name(device) dtype = get_dtype_by_name(dtype) From b277efcf01e25078ab2eb826ff552731ff28561f Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:14:50 +0200 Subject: [PATCH 12/19] Fix "missing loaded_keys" error --- requirements.txt | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/requirements.txt b/requirements.txt index 6c2f526..6d2dc07 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,7 +1,7 @@ torch numpy==1.26.4 -transformers +transformers==4.45.0 accelerate -diffusers +diffusers==0.32.2 fastapi<0.113.0 -opencv-python \ No newline at end of file +opencv-python From 183ef19cede41f0153ac284dbd9743101a7c8509 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:47:22 +0200 Subject: [PATCH 13/19] Update README.md --- README.md | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/README.md b/README.md index 8e0c40d..7d2a2ea 100644 --- a/README.md +++ b/README.md @@ -1,8 +1,9 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-outpaint](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) by fffiloni. -![Extension-Overview](https://github.com/user-attachments/assets/b801698e-e666-4179-98bd-42dfb1f033ba) +![image](https://github.com/user-attachments/assets/1a02c2d1-f24e-4ad2-acdc-a2cbb15a1f14) #### Updates: +- 15/05/2025: Fixed `missing 'loaded_keys'` error. More details below. - 17/11/2024: - Added more options to Pad Image node (resize image, custom resize image percentage, mask overlap percentage, overlap left/right/top/bottom). - Side notes: @@ -22,30 +23,26 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. -- Download models in comfyui/models/diffusion_models: - - model_name: - - unet: - - `diffusion_pytorch_model.fp16.safetensors` ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/blob/main/unet/diffusion_pytorch_model.fp16.safetensors)) - - `config.json` ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/blob/main/unet/config.json)) - - scheduler: - - `scheduler_config.json` ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/blob/main/scheduler/scheduler_config.json)) - - `model_index.json` ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/blob/main/model_index.json)) - - controlnet_name: - - `config_promax.json` ([example](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/blob/main/config_promax.json)), `diffusion_pytorch_model_promax.safetensors` ([example](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/blob/main/diffusion_pytorch_model_promax.safetensors)) +- Download a sdxl model ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/resolve/main/unet/diffusion_pytorch_model.fp16.safetensors)) in comfyui/models/diffusion_models; +- Download a sdxl controlnet model ([example](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/blob/main/diffusion_pytorch_model_promax.safetensors)) in comfyui/models/controlnet. + - (Dual) Clip Loader node: if you use the Clip Loader instead of Checkpoint Loader Simple, and want to use an `sdxl type` model like RealVisXL_V5.0_Lightning, you can download `clip_I` and `clip_g` from [here](https://huggingface.co/Comfy-Org/stable-diffusion-3.5-fp8/tree/main/text_encoders). You can use [this workflow](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/blob/New-Pad-Node-Options/Diffusers-Outpaint-DoubleWorkflow.json) (change model.fp16 with `clip_g`). ## Overview - **Minimum VRAM**: 6 gb with 1280x720 image, rtx 3060, RealVisXL_V5.0_Lightning, sdxl-vae-fp16-fix, controlnet-union-sdxl-promax using `sequential_cpu_offload`, otherwise 8,3 gb; - ~As seen in [this issue](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/issues/7#issuecomment-2410852908), images with **square corners** are required~. -The extension gives 4 nodes: -- **Load Diffusion Outpaint Models**: a simple node to load diffusion `models`. You can download them from Huggingface (the extension doesn't download them automatically); +The extension gives 5 nodes: +- **Load Diffuser Model**: a simple node to load diffusion `models`. You can download them from Huggingface (the extension doesn't download them automatically). Put them inside the `diffusion_models` folder; +- **Load Diffuser Controlnet**: a simple node to load diffusion `models`. You can download them from Huggingface (the extension doesn't download them automatically). Put them inside the `controlnet` folder; - **Paid Image for Diffusers Outpaint**: this node resizes the image based on the specified `width` and `height`, then resizes it again based on the `resize_image` percentage, and if possible it will put the mask based on the `alignment` specified, otherwise it will revert back to the default "middle" `alignment`; - **Encode Diffusers Outpaint Prompt**: self explanatory. Works as `clip text encode (prompt)`, and specifies what to add to the image; - **Diffusers Image Outpaint**: This is the main node, that outpaints the image. Currently the generation process is based on fffiloni's one, so you can't reproduce a specific a specific outpaint, and the `seed` option you see is only used to update the UI and generate a new image. You can specify the amount of `steps` to generate the image. You _can_ also pass image and mask to `vae encode (for inpainting)` node, then pass the latent to a `sampler`, but controlnets and ip-adapters won't always give good results like with diffusers outpaint, and they require a different workflow, not covered by this extension. +Since for now only sdxl models work, the config are chosen automatically. If in the future other types that would require different config will work, I could add more selection options. + ### Change model used - **Main model**: On huggingface, choose a model from [text2image models](https://huggingface.co/models?pipeline_tag=text-to-image&sort=trending) (**sdxl and maybe sd1.5 model types should work, while flux doesn't**), then create a new folder named after it in `comfyui/models/diffusion_models`, then download in it the subfolders `unet` (if not available use `transformer`) and `scheduler`. - Hint: sometimes in the `unet` or `transformer` folder there are more model files and not all are required. If you have `model.fp16` and `model`, I suggest you to use the fp16 variant; if you have `model-001-of-002`, `model-002-of-002`, `model`, choose model (instead of the fragmented version). @@ -54,5 +51,8 @@ You _can_ also pass image and mask to `vae encode (for inpainting)` node, then p #### Unet and Controlnet Models Loader using ComfYUI nodes canceled I can load them but then they don't work in the inference code, since comfyui load diffusers models in a different format ([reddit post](https://www.reddit.com/r/comfyui/comments/17fvb49/comment/k6cz9yv/?utm_source=share&utm_medium=web3x&utm_name=web3xcss&utm_term=1&utm_content=share_button)). +## Missing 'loaded_keys' error +Recent versions of `transformers` and `diffusers` broke somethings, you need to revert back, command with some working versions (do it inside your comfyui env): `pip install transformers==4.45.0 --upgrade diffusers==0.32.2 --upgrade`. + ## Credits diffusers-image-outpaint by [fffiloni](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) From 783d02c3a06da84f0b5f2f4a5193bcf495a00362 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:50:26 +0200 Subject: [PATCH 14/19] Update README.md --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index 7d2a2ea..68faafb 100644 --- a/README.md +++ b/README.md @@ -52,7 +52,7 @@ Since for now only sdxl models work, the config are chosen automatically. If in I can load them but then they don't work in the inference code, since comfyui load diffusers models in a different format ([reddit post](https://www.reddit.com/r/comfyui/comments/17fvb49/comment/k6cz9yv/?utm_source=share&utm_medium=web3x&utm_name=web3xcss&utm_term=1&utm_content=share_button)). ## Missing 'loaded_keys' error -Recent versions of `transformers` and `diffusers` broke somethings, you need to revert back, command with some working versions (do it inside your comfyui env): `pip install transformers==4.45.0 --upgrade diffusers==0.32.2 --upgrade`. +Recent versions of `transformers` and `diffusers` broke somethings, you need to revert back, command with some working versions (found [here](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/blob/main/requirements.txt)) (do it inside your comfyui env): `pip install transformers==4.45.0 --upgrade diffusers==0.32.2 --upgrade`. ## Credits diffusers-image-outpaint by [fffiloni](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/tree/main) From c408dd803717357d9726e8386939465be2912ed8 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:51:37 +0200 Subject: [PATCH 15/19] Delete configs/controlnet-sdxl-promax/config.json --- configs/controlnet-sdxl-promax/config.json | 56 ---------------------- 1 file changed, 56 deletions(-) delete mode 100644 configs/controlnet-sdxl-promax/config.json diff --git a/configs/controlnet-sdxl-promax/config.json b/configs/controlnet-sdxl-promax/config.json deleted file mode 100644 index e29898b..0000000 --- a/configs/controlnet-sdxl-promax/config.json +++ /dev/null @@ -1,56 +0,0 @@ -{ - "_class_name": "ControlNetModel", - "_diffusers_version": "0.20.0.dev0", - "act_fn": "silu", - "addition_embed_type": "text_time", - "addition_embed_type_num_heads": 64, - "addition_time_embed_dim": 256, - "attention_head_dim": [ - 5, - 10, - 20 - ], - "block_out_channels": [ - 320, - 640, - 1280 - ], - "class_embed_type": null, - "conditioning_channels": 3, - "conditioning_embedding_out_channels": [ - 16, - 32, - 96, - 256 - ], - "controlnet_conditioning_channel_order": "rgb", - "cross_attention_dim": 2048, - "down_block_types": [ - "DownBlock2D", - "CrossAttnDownBlock2D", - "CrossAttnDownBlock2D" - ], - "downsample_padding": 1, - "encoder_hid_dim": null, - "encoder_hid_dim_type": null, - "flip_sin_to_cos": true, - "freq_shift": 0, - "global_pool_conditions": false, - "in_channels": 4, - "layers_per_block": 2, - "mid_block_scale_factor": 1, - "norm_eps": 1e-05, - "norm_num_groups": 32, - "num_attention_heads": null, - "num_class_embeds": null, - "only_cross_attention": false, - "projection_class_embeddings_input_dim": 2816, - "resnet_time_scale_shift": "default", - "transformer_layers_per_block": [ - 1, - 2, - 10 - ], - "upcast_attention": null, - "use_linear_projection": true -} From 4b3668a77bdfdd35f6df5c6bb45c62d074329097 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:51:53 +0200 Subject: [PATCH 16/19] Delete Diffusers-Outpaint-DoubleWorkflow.json --- Diffusers-Outpaint-DoubleWorkflow.json | 1100 ------------------------ 1 file changed, 1100 deletions(-) delete mode 100644 Diffusers-Outpaint-DoubleWorkflow.json diff --git a/Diffusers-Outpaint-DoubleWorkflow.json b/Diffusers-Outpaint-DoubleWorkflow.json deleted file mode 100644 index 2eb5242..0000000 --- a/Diffusers-Outpaint-DoubleWorkflow.json +++ /dev/null @@ -1,1100 +0,0 @@ -{ - "last_node_id": 30, - "last_link_id": 35, - "nodes": [ - { - "id": 4, - "type": "VAEDecode", - "pos": [ - 1220, - -140 - ], - "size": [ - 210, - 46 - ], - "flags": {}, - "order": 17, - "mode": 0, - "inputs": [ - { - "name": "samples", - "type": "LATENT", - "link": 3 - }, - { - "name": "vae", - "type": "VAE", - "link": 4 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 1 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "VAEDecode" - } - }, - { - "id": 5, - "type": "VAELoader", - "pos": [ - 930, - 120 - ], - "size": [ - 260, - 60 - ], - "flags": {}, - "order": 0, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "VAE", - "type": "VAE", - "links": [ - 4 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "VAELoader" - }, - "widgets_values": [ - "sdxl_vae.safetensors" - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 7, - "type": "DualCLIPLoader", - "pos": [ - 160, - 40 - ], - "size": [ - 260, - 110 - ], - "flags": {}, - "order": 1, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "CLIP", - "type": "CLIP", - "links": [ - 7, - 9 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "DualCLIPLoader" - }, - "widgets_values": [ - "clip_l.safetensors", - "model.fp16.safetensors", - "sdxl" - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 9, - "type": "EncodeDiffusersOutpaintPrompt", - "pos": [ - 450, - -10 - ], - "size": [ - 400, - 96 - ], - "flags": {}, - "order": 13, - "mode": 0, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 8 - }, - { - "name": "clip", - "type": "CLIP", - "link": 9 - } - ], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [], - "slot_index": 0 - }, - { - "name": "diffusers_conditioning", - "type": "CONDITIONING", - "links": [ - 12 - ], - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "EncodeDiffusersOutpaintPrompt" - }, - "widgets_values": [ - "" - ], - "color": "#322", - "bgcolor": "#533" - }, - { - "id": 1, - "type": "PreviewImage", - "pos": [ - 1230, - -50 - ], - "size": [ - 510, - 490 - ], - "flags": {}, - "order": 19, - "mode": 0, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 1 - } - ], - "outputs": [], - "properties": { - "Node name for S&R": "PreviewImage" - } - }, - { - "id": 18, - "type": "PadImageForDiffusersOutpaint", - "pos": [ - 560, - 130 - ], - "size": [ - 290, - 310 - ], - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 20 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [] - }, - { - "name": "MASK", - "type": "MASK", - "links": null - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "links": [ - 22 - ] - } - ], - "properties": { - "Node name for S&R": "PadImageForDiffusersOutpaint" - }, - "widgets_values": [ - 1280, - 720, - "Middle", - "Full", - 50, - 10, - true, - true, - true, - true - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 8, - "type": "EncodeDiffusersOutpaintPrompt", - "pos": [ - 450, - -160 - ], - "size": [ - 400, - 96 - ], - "flags": {}, - "order": 12, - "mode": 0, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 6 - }, - { - "name": "clip", - "type": "CLIP", - "link": 7 - } - ], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [ - 10 - ], - "slot_index": 0 - }, - { - "name": "diffusers_conditioning", - "type": "CONDITIONING", - "links": [ - 11 - ], - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "EncodeDiffusersOutpaintPrompt" - }, - "widgets_values": [ - "a verdant valley with waterfalls, rainbow" - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 21, - "type": "PreviewImage", - "pos": [ - 1230, - -870 - ], - "size": [ - 510, - 490 - ], - "flags": {}, - "order": 18, - "mode": 2, - "inputs": [ - { - "name": "images", - "type": "IMAGE", - "link": 24 - } - ], - "outputs": [], - "properties": { - "Node name for S&R": "PreviewImage" - } - }, - { - "id": 22, - "type": "VAEDecode", - "pos": [ - 1220, - -960 - ], - "size": [ - 210, - 46 - ], - "flags": {}, - "order": 16, - "mode": 2, - "inputs": [ - { - "name": "samples", - "type": "LATENT", - "link": 25 - }, - { - "name": "vae", - "type": "VAE", - "link": 35 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 24 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "VAEDecode" - } - }, - { - "id": 25, - "type": "EncodeDiffusersOutpaintPrompt", - "pos": [ - 450, - -830 - ], - "size": [ - 400, - 96 - ], - "flags": {}, - "order": 10, - "mode": 2, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": null - }, - { - "name": "clip", - "type": "CLIP", - "link": 33 - } - ], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [], - "slot_index": 0 - }, - { - "name": "diffusers_conditioning", - "type": "CONDITIONING", - "links": [ - 29 - ], - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "EncodeDiffusersOutpaintPrompt" - }, - "widgets_values": [ - "" - ], - "color": "#322", - "bgcolor": "#533" - }, - { - "id": 26, - "type": "DiffusersImageOutpaint", - "pos": [ - 900, - -960 - ], - "size": [ - 300, - 214 - ], - "flags": {}, - "order": 14, - "mode": 2, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 27 - }, - { - "name": "positive", - "type": "CONDITIONING", - "link": 28 - }, - { - "name": "negative", - "type": "CONDITIONING", - "link": 29 - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "link": 30 - } - ], - "outputs": [ - { - "name": "LATENT", - "type": "LATENT", - "links": [ - 25 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "DiffusersImageOutpaint" - }, - "widgets_values": [ - 1.5, - 1, - 770989832261247, - "randomize", - 8 - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 27, - "type": "PadImageForDiffusersOutpaint", - "pos": [ - 560, - -690 - ], - "size": [ - 290, - 310 - ], - "flags": {}, - "order": 11, - "mode": 2, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 31 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [] - }, - { - "name": "MASK", - "type": "MASK", - "links": null - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "links": [ - 30 - ] - } - ], - "properties": { - "Node name for S&R": "PadImageForDiffusersOutpaint" - }, - "widgets_values": [ - 1280, - 720, - "Middle", - "Full", - 50, - 10, - true, - true, - true, - true - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 3, - "type": "LoadImage", - "pos": [ - 180, - 200 - ], - "size": [ - 320, - 310 - ], - "flags": {}, - "order": 2, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 20 - ], - "slot_index": 0 - }, - { - "name": "MASK", - "type": "MASK", - "links": null - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "20230403_183417.jpg", - "image" - ] - }, - { - "id": 24, - "type": "EncodeDiffusersOutpaintPrompt", - "pos": [ - 450, - -980 - ], - "size": [ - 400, - 96 - ], - "flags": {}, - "order": 9, - "mode": 2, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 34 - }, - { - "name": "clip", - "type": "CLIP", - "link": 32 - } - ], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [ - 27 - ], - "slot_index": 0 - }, - { - "name": "diffusers_conditioning", - "type": "CONDITIONING", - "links": [ - 28 - ], - "slot_index": 1 - } - ], - "properties": { - "Node name for S&R": "EncodeDiffusersOutpaintPrompt" - }, - "widgets_values": [ - "a verdant valley with waterfalls, rainbow" - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 10, - "type": "DiffusersImageOutpaint", - "pos": [ - 900, - -140 - ], - "size": [ - 300, - 214 - ], - "flags": {}, - "order": 15, - "mode": 0, - "inputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "link": 10 - }, - { - "name": "positive", - "type": "CONDITIONING", - "link": 11 - }, - { - "name": "negative", - "type": "CONDITIONING", - "link": 12 - }, - { - "name": "diffuser_outpaint_cnet_image", - "type": "IMAGE", - "link": 22 - } - ], - "outputs": [ - { - "name": "LATENT", - "type": "LATENT", - "links": [ - 3 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "DiffusersImageOutpaint" - }, - "widgets_values": [ - 1.5, - 1, - 770989832261247, - "randomize", - 8 - ], - "color": "#232", - "bgcolor": "#353" - }, - { - "id": 29, - "type": "LoadDiffusersOutpaintModels", - "pos": [ - 30, - -1070 - ], - "size": [ - 320, - 154 - ], - "flags": {}, - "order": 3, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [ - 34 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "LoadDiffusersOutpaintModels" - }, - "widgets_values": [ - "RealVisXL_V5.0_Lightning", - "controlnet-union-sdxl-1.0", - "auto", - "auto", - false - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 19, - "type": "CheckpointLoaderSimple", - "pos": [ - 30, - -870 - ], - "size": [ - 340, - 100 - ], - "flags": {}, - "order": 4, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": null - }, - { - "name": "CLIP", - "type": "CLIP", - "links": [ - 32, - 33 - ], - "slot_index": 1 - }, - { - "name": "VAE", - "type": "VAE", - "links": [ - 35 - ], - "slot_index": 2 - } - ], - "properties": { - "Node name for S&R": "CheckpointLoaderSimple" - }, - "widgets_values": [ - "realvisxlV50_v50LightningBakedvae.safetensors" - ], - "color": "#223", - "bgcolor": "#335" - }, - { - "id": 28, - "type": "LoadImage", - "pos": [ - 180, - -620 - ], - "size": [ - 300, - 300 - ], - "flags": {}, - "order": 5, - "mode": 2, - "inputs": [], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 31 - ], - "slot_index": 0 - }, - { - "name": "MASK", - "type": "MASK", - "links": null - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "20230403_183417.jpg", - "image" - ] - }, - { - "id": 30, - "type": "Note", - "pos": [ - 20, - -720 - ], - "size": [ - 380, - 60 - ], - "flags": {}, - "order": 6, - "mode": 2, - "inputs": [], - "outputs": [], - "properties": {}, - "widgets_values": [ - "The Checkpoint Loader Simple load a model with baked in Clip and Vae, so I don't need Clip Loader and Vae Loader" - ], - "color": "#432", - "bgcolor": "#653" - }, - { - "id": 11, - "type": "LoadDiffusersOutpaintModels", - "pos": [ - 90, - -160 - ], - "size": [ - 320, - 154 - ], - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [], - "outputs": [ - { - "name": "diffusers_outpaint_pipe", - "type": "PIPE", - "links": [ - 6, - 8 - ], - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "LoadDiffusersOutpaintModels" - }, - "widgets_values": [ - "RealVisXL_V5.0_Lightning", - "controlnet-union-sdxl-1.0", - "auto", - "auto", - false - ], - "color": "#223", - "bgcolor": "#335" - } - ], - "links": [ - [ - 1, - 4, - 0, - 1, - 0, - "IMAGE" - ], - [ - 3, - 10, - 0, - 4, - 0, - "LATENT" - ], - [ - 4, - 5, - 0, - 4, - 1, - "VAE" - ], - [ - 6, - 11, - 0, - 8, - 0, - "PIPE" - ], - [ - 7, - 7, - 0, - 8, - 1, - "CLIP" - ], - [ - 8, - 11, - 0, - 9, - 0, - "PIPE" - ], - [ - 9, - 7, - 0, - 9, - 1, - "CLIP" - ], - [ - 10, - 8, - 0, - 10, - 0, - "PIPE" - ], - [ - 11, - 8, - 1, - 10, - 1, - "CONDITIONING" - ], - [ - 12, - 9, - 1, - 10, - 2, - "CONDITIONING" - ], - [ - 20, - 3, - 0, - 18, - 0, - "IMAGE" - ], - [ - 22, - 18, - 2, - 10, - 3, - "IMAGE" - ], - [ - 24, - 22, - 0, - 21, - 0, - "IMAGE" - ], - [ - 25, - 26, - 0, - 22, - 0, - "LATENT" - ], - [ - 27, - 24, - 0, - 26, - 0, - "PIPE" - ], - [ - 28, - 24, - 1, - 26, - 1, - "CONDITIONING" - ], - [ - 29, - 25, - 1, - 26, - 2, - "CONDITIONING" - ], - [ - 30, - 27, - 2, - 26, - 3, - "IMAGE" - ], - [ - 31, - 28, - 0, - 27, - 0, - "IMAGE" - ], - [ - 32, - 19, - 1, - 24, - 1, - "CLIP" - ], - [ - 33, - 19, - 1, - 25, - 1, - "CLIP" - ], - [ - 34, - 29, - 0, - 24, - 0, - "PIPE" - ], - [ - 35, - 19, - 2, - 22, - 1, - "VAE" - ] - ], - "groups": [ - { - "id": 1, - "title": "Checkpoint Loader Simple", - "bounding": [ - -3.8856265544891357, - -1143.90380859375, - 1790.8021240234375, - 831.0167236328125 - ], - "color": "#3f789e", - "font_size": 24, - "flags": {} - }, - { - "id": 2, - "title": "Clip Loader + Vae Loader", - "bounding": [ - 11.18426513671875, - -245.8946990966797, - 1765.697265625, - 778.4640502929688 - ], - "color": "#3f789e", - "font_size": 24, - "flags": {} - } - ], - "config": {}, - "extra": { - "ds": { - "scale": 0.8769226950000009, - "offset": [ - 54.93129335580923, - 254.6595267695907 - ] - } - }, - "version": 0.4 -} \ No newline at end of file From e9b9f16fae5f3a44ff721689a3f8caf91f998101 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:52:11 +0200 Subject: [PATCH 17/19] Add files via upload --- diffusers-image-outpaint-workflow.json | 629 +++++++++++++++++++++++++ 1 file changed, 629 insertions(+) create mode 100644 diffusers-image-outpaint-workflow.json diff --git a/diffusers-image-outpaint-workflow.json b/diffusers-image-outpaint-workflow.json new file mode 100644 index 0000000..39e9025 --- /dev/null +++ b/diffusers-image-outpaint-workflow.json @@ -0,0 +1,629 @@ +{ + "id": "5e709e31-1e9f-475e-a837-14abe1d4f292", + "revision": 0, + "last_node_id": 586, + "last_link_id": 1133, + "nodes": [ + { + "id": 499, + "type": "PadImageForDiffusersOutpaint", + "pos": [ + -5320, + 420 + ], + "size": [ + 290, + 314 + ], + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 906 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [] + }, + { + "name": "MASK", + "type": "MASK", + "links": [] + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "links": [ + 1100 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "PadImageForDiffusersOutpaint", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 720, + 1280, + "Middle", + "Full", + 50, + 10, + true, + true, + true, + true + ], + "color": "#233", + "bgcolor": "#355" + }, + { + "id": 494, + "type": "VAEDecode", + "pos": [ + -4710, + -70 + ], + "size": [ + 140, + 46 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 1101 + }, + { + "name": "vae", + "type": "VAE", + "link": 908 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 899 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.30", + "Node name for S&R": "VAEDecode", + "widget_ue_connectable": {} + }, + "widgets_values": [], + "color": "#323", + "bgcolor": "#535" + }, + { + "id": 495, + "type": "PreviewImage", + "pos": [ + -4550, + -70 + ], + "size": [ + 250, + 310 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 899 + } + ], + "outputs": [], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.30", + "Node name for S&R": "PreviewImage", + "widget_ue_connectable": {} + }, + "widgets_values": [] + }, + { + "id": 497, + "type": "DualCLIPLoader", + "pos": [ + -5570, + 200 + ], + "size": [ + 270, + 130 + ], + "flags": {}, + "order": 0, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 1110, + 1111 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.30", + "Node name for S&R": "DualCLIPLoader", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "clip_l.safetensors", + "clip_g.safetensors", + "sdxl", + "default" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 569, + "type": "EncodeDiffusersOutpaintPrompt", + "pos": [ + -5280, + 200 + ], + "size": [ + 252.08065795898438, + 136 + ], + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 1111 + } + ], + "outputs": [ + { + "name": "diffusers_conditioning", + "type": "CONDITIONING", + "links": [ + 1099 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "EncodeDiffusersOutpaintPrompt", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "auto", + "auto", + "" + ], + "color": "#322", + "bgcolor": "#533" + }, + { + "id": 501, + "type": "VAELoader", + "pos": [ + -5010, + 260 + ], + "size": [ + 270, + 58 + ], + "flags": {}, + "order": 1, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "VAE", + "type": "VAE", + "links": [ + 908 + ] + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.30", + "Node name for S&R": "VAELoader", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "sdxl_vae.safetensors" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 573, + "type": "DiffusersImageOutpaint", + "pos": [ + -4990, + -60 + ], + "size": [ + 247.341796875, + 278 + ], + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 1131 + }, + { + "name": "scheduler_configs", + "type": "SCHEDULER", + "link": 1132 + }, + { + "name": "control_net", + "type": "CONTROL_NET", + "link": 1133 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 1098 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 1099 + }, + { + "name": "diffuser_outpaint_cnet_image", + "type": "IMAGE", + "link": 1100 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 1101 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "DiffusersImageOutpaint", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + 1.5, + 1, + 8, + "auto", + "auto", + false + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 531, + "type": "LoadDiffuserModel", + "pos": [ + -5610, + -190 + ], + "size": [ + 290, + 150 + ], + "flags": {}, + "order": 2, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 1131 + ] + }, + { + "name": "scheduler configs", + "type": "SCHEDULER", + "links": [ + 1132 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "LoadDiffuserModel", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "RealVisXL_V5.0_Lightning_unet.safetensors", + "auto", + "auto", + "sdxl" + ], + "color": "#223", + "bgcolor": "#335" + }, + { + "id": 500, + "type": "LoadImage", + "pos": [ + -5640, + 430 + ], + "size": [ + 270, + 314 + ], + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 906 + ] + }, + { + "name": "MASK", + "type": "MASK", + "links": null + } + ], + "properties": { + "cnr_id": "comfy-core", + "ver": "0.3.30", + "Node name for S&R": "LoadImage", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "20230403_183417.jpg", + "image" + ] + }, + { + "id": 570, + "type": "EncodeDiffusersOutpaintPrompt", + "pos": [ + -5280, + 10 + ], + "size": [ + 252.08065795898438, + 136 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 1110 + } + ], + "outputs": [ + { + "name": "diffusers_conditioning", + "type": "CONDITIONING", + "links": [ + 1098 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "EncodeDiffusersOutpaintPrompt", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "auto", + "auto", + "a verdant valley with waterfalls, rainbow" + ], + "color": "#232", + "bgcolor": "#353" + }, + { + "id": 532, + "type": "LoadDiffuserControlnet", + "pos": [ + -5650, + 10 + ], + "size": [ + 330, + 130 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [], + "outputs": [ + { + "name": "CONTROL_NET", + "type": "CONTROL_NET", + "links": [ + 1133 + ] + } + ], + "properties": { + "cnr_id": "ComfyUI-DiffusersImageOutpaint", + "ver": "6a51ce5d3baa2171a85f51d462ef9f30ff9b5d26", + "Node name for S&R": "LoadDiffuserControlnet", + "aux_id": "GiusTex/ComfyUI-DiffusersImageOutpaint", + "widget_ue_connectable": {} + }, + "widgets_values": [ + "controlnet-union-promax_sdxl.safetensors", + "auto", + "auto", + "controlnet-sdxl-promax" + ], + "color": "#432", + "bgcolor": "#653" + } + ], + "links": [ + [ + 899, + 494, + 0, + 495, + 0, + "IMAGE" + ], + [ + 906, + 500, + 0, + 499, + 0, + "IMAGE" + ], + [ + 908, + 501, + 0, + 494, + 1, + "VAE" + ], + [ + 1098, + 570, + 0, + 573, + 3, + "CONDITIONING" + ], + [ + 1099, + 569, + 0, + 573, + 4, + "CONDITIONING" + ], + [ + 1100, + 499, + 2, + 573, + 5, + "IMAGE" + ], + [ + 1101, + 573, + 0, + 494, + 0, + "LATENT" + ], + [ + 1110, + 497, + 0, + 570, + 0, + "CLIP" + ], + [ + 1111, + 497, + 0, + 569, + 0, + "CLIP" + ], + [ + 1131, + 531, + 0, + 573, + 0, + "MODEL" + ], + [ + 1132, + 531, + 1, + 573, + 1, + "SCHEDULER" + ], + [ + 1133, + 532, + 0, + 573, + 2, + "CONTROL_NET" + ] + ], + "groups": [], + "config": {}, + "extra": { + "ds": { + "scale": 0.7972024500000006, + "offset": [ + 5835.209356481534, + 220.1462025110432 + ] + }, + "frontendVersion": "1.19.9", + "groupNodes": {}, + "ue_links": [], + "links_added_by_ue": [], + "VHS_latentpreview": true, + "VHS_latentpreviewrate": 0, + "VHS_MetadataImage": true, + "VHS_KeepIntermediate": true + }, + "version": 0.4 +} \ No newline at end of file From 95cff71187ce97a75c7fd9e5c3ef5b4b6dd68425 Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 19:55:07 +0200 Subject: [PATCH 18/19] Update README.md --- README.md | 8 -------- 1 file changed, 8 deletions(-) diff --git a/README.md b/README.md index 68faafb..27df137 100644 --- a/README.md +++ b/README.md @@ -16,11 +16,6 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o - 20/10/2024: No more need to download tokenizers nor text encoders! Now comfyui clip loader works, and you can use your clip models. You can also use the Checkpoint Loader Simple node, to skip the clip selection part. - 10/2024: You don't need any more the diffusers vae, and can use the extension in low vram mode using `sequential_cpu_offload` (also thanks to [zmwv823](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/4)) that pushes the vram usage from *8,3 gb* down to **_6 gb_**. -#### To do list to [change model used](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/pull/14): -- - [x] ComfyUI Clip Loader Node -- ~[ ] ComfyUI Load Diffusion Model Node~ (more info [below](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint#unet-and-controlnet-models-loader-using-comfyui-nodes-canceled)) -- ~[ ] ComfyUI Load Conotrolnet Model Node~ (more info [below](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint#unet-and-controlnet-models-loader-using-comfyui-nodes-canceled)) - ## Installation - Download this extension or `git clone` it in comfyui/custom_nodes, then (if comfyui-manager didn't already install the requirements or you have missing modules), from comfyui virtual env write `cd your/path/to/this/extension` and `pip install -r requirements.txt`. - Download a sdxl model ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/resolve/main/unet/diffusion_pytorch_model.fp16.safetensors)) in comfyui/models/diffusion_models; @@ -48,9 +43,6 @@ Since for now only sdxl models work, the config are chosen automatically. If in - Hint: sometimes in the `unet` or `transformer` folder there are more model files and not all are required. If you have `model.fp16` and `model`, I suggest you to use the fp16 variant; if you have `model-001-of-002`, `model-002-of-002`, `model`, choose model (instead of the fragmented version). - **Controlnet model**: download `config.json` and the safetensors `model`. -#### Unet and Controlnet Models Loader using ComfYUI nodes canceled -I can load them but then they don't work in the inference code, since comfyui load diffusers models in a different format ([reddit post](https://www.reddit.com/r/comfyui/comments/17fvb49/comment/k6cz9yv/?utm_source=share&utm_medium=web3x&utm_name=web3xcss&utm_term=1&utm_content=share_button)). - ## Missing 'loaded_keys' error Recent versions of `transformers` and `diffusers` broke somethings, you need to revert back, command with some working versions (found [here](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/blob/main/requirements.txt)) (do it inside your comfyui env): `pip install transformers==4.45.0 --upgrade diffusers==0.32.2 --upgrade`. From b37d3b169265d39b9feb553ebc778b4ce42f252e Mon Sep 17 00:00:00 2001 From: Gius <112352961+GiusTex@users.noreply.github.com> Date: Thu, 15 May 2025 20:17:07 +0200 Subject: [PATCH 19/19] Update README.md --- README.md | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 27df137..d0bd45f 100644 --- a/README.md +++ b/README.md @@ -21,6 +21,17 @@ ComfyUI nodes for outpainting images with diffusers, based on [diffusers-image-o - Download a sdxl model ([example](https://huggingface.co/SG161222/RealVisXL_V5.0_Lightning/resolve/main/unet/diffusion_pytorch_model.fp16.safetensors)) in comfyui/models/diffusion_models; - Download a sdxl controlnet model ([example](https://huggingface.co/xinsir/controlnet-union-sdxl-1.0/blob/main/diffusion_pytorch_model_promax.safetensors)) in comfyui/models/controlnet. +**⚠ Choosing model and controlnet**: As of now, I only tried `RealVisXL_V5.0_Lightning` and `controlnet-union-promax_sdxl`. Mixing RealVisXL with controlnet-union (non promax version) gave error, so it could be that other models/controlnets give error as well, but I haven't tried much combinations so I can't tell. + +
+ Some considerations + + Flux is still beyond me (even if I was quite there, I think). I haven't tried integrating other model types, and after my flux failure I don't think I'll try adding other model types. + + Since for now only sdxl models work, the configs are hardcoded. + +
+ - (Dual) Clip Loader node: if you use the Clip Loader instead of Checkpoint Loader Simple, and want to use an `sdxl type` model like RealVisXL_V5.0_Lightning, you can download `clip_I` and `clip_g` from [here](https://huggingface.co/Comfy-Org/stable-diffusion-3.5-fp8/tree/main/text_encoders). You can use [this workflow](https://github.com/GiusTex/ComfyUI-DiffusersImageOutpaint/blob/New-Pad-Node-Options/Diffusers-Outpaint-DoubleWorkflow.json) (change model.fp16 with `clip_g`). ## Overview @@ -36,13 +47,6 @@ The extension gives 5 nodes: You _can_ also pass image and mask to `vae encode (for inpainting)` node, then pass the latent to a `sampler`, but controlnets and ip-adapters won't always give good results like with diffusers outpaint, and they require a different workflow, not covered by this extension. -Since for now only sdxl models work, the config are chosen automatically. If in the future other types that would require different config will work, I could add more selection options. - -### Change model used -- **Main model**: On huggingface, choose a model from [text2image models](https://huggingface.co/models?pipeline_tag=text-to-image&sort=trending) (**sdxl and maybe sd1.5 model types should work, while flux doesn't**), then create a new folder named after it in `comfyui/models/diffusion_models`, then download in it the subfolders `unet` (if not available use `transformer`) and `scheduler`. - - Hint: sometimes in the `unet` or `transformer` folder there are more model files and not all are required. If you have `model.fp16` and `model`, I suggest you to use the fp16 variant; if you have `model-001-of-002`, `model-002-of-002`, `model`, choose model (instead of the fragmented version). -- **Controlnet model**: download `config.json` and the safetensors `model`. - ## Missing 'loaded_keys' error Recent versions of `transformers` and `diffusers` broke somethings, you need to revert back, command with some working versions (found [here](https://huggingface.co/spaces/fffiloni/diffusers-image-outpaint/blob/main/requirements.txt)) (do it inside your comfyui env): `pip install transformers==4.45.0 --upgrade diffusers==0.32.2 --upgrade`.