diff --git a/imports/ComfyUI_IPAdapter_plus/.github/FUNDING.yml b/imports/ComfyUI_IPAdapter_plus/.github/FUNDING.yml new file mode 100644 index 0000000..191991e --- /dev/null +++ b/imports/ComfyUI_IPAdapter_plus/.github/FUNDING.yml @@ -0,0 +1,4 @@ +# These are supported funding model platforms + +github: cubiq +custom: ['https://www.paypal.com/paypalme/matt3o'] diff --git a/imports/ComfyUI_IPAdapter_plus/CrossAttentionPatchImport.py b/imports/ComfyUI_IPAdapter_plus/CrossAttentionPatchImport.py index 195292d..78b3601 100644 --- a/imports/ComfyUI_IPAdapter_plus/CrossAttentionPatchImport.py +++ b/imports/ComfyUI_IPAdapter_plus/CrossAttentionPatchImport.py @@ -6,10 +6,11 @@ from .utils import tensor_to_size class CrossAttentionPatchImport: # forward for patching - def __init__(self, ipadapter=None, number=0, weight=1.0, cond=None, uncond=None, weight_type="linear", mask=None, sigma_start=0.0, sigma_end=1.0, unfold_batch=False, embeds_scaling='V only'): + def __init__(self, ipadapter=None, number=0, weight=1.0, cond=None, cond_alt=None, uncond=None, weight_type="linear", mask=None, sigma_start=0.0, sigma_end=1.0, unfold_batch=False, embeds_scaling='V only'): self.weights = [weight] self.ipadapters = [ipadapter] self.conds = [cond] + self.conds_alt = [cond_alt] self.unconds = [uncond] self.weight_types = [weight_type] self.masks = [mask] @@ -18,15 +19,16 @@ class CrossAttentionPatchImport: self.unfold_batch = [unfold_batch] self.embeds_scaling = [embeds_scaling] self.number = number - self.layers = 10 if '101_to_k_ip' in ipadapter.ip_layers.to_kvs else 15 # TODO: check if this is a valid condition to detect all models + self.layers = 11 if '101_to_k_ip' in ipadapter.ip_layers.to_kvs else 16 # TODO: check if this is a valid condition to detect all models self.k_key = str(self.number*2+1) + "_to_k_ip" self.v_key = str(self.number*2+1) + "_to_v_ip" - def set_new_condition(self, ipadapter=None, number=0, weight=1.0, cond=None, uncond=None, weight_type="linear", mask=None, sigma_start=0.0, sigma_end=1.0, unfold_batch=False, embeds_scaling='V only'): + def set_new_condition(self, ipadapter=None, number=0, weight=1.0, cond=None, cond_alt=None, uncond=None, weight_type="linear", mask=None, sigma_start=0.0, sigma_end=1.0, unfold_batch=False, embeds_scaling='V only'): self.weights.append(weight) self.ipadapters.append(ipadapter) self.conds.append(cond) + self.conds_alt.append(cond_alt) self.unconds.append(uncond) self.weight_types.append(weight_type) self.masks.append(mask) @@ -52,35 +54,8 @@ class CrossAttentionPatchImport: out = optimized_attention(q, k, v, extra_options["n_heads"]) _, _, oh, ow = extra_options["original_shape"] - for weight, cond, uncond, ipadapter, mask, weight_type, sigma_start, sigma_end, unfold_batch, embeds_scaling in zip(self.weights, self.conds, self.unconds, self.ipadapters, self.masks, self.weight_types, self.sigma_starts, self.sigma_ends, self.unfold_batch, self.embeds_scaling): + for weight, cond, cond_alt, uncond, ipadapter, mask, weight_type, sigma_start, sigma_end, unfold_batch, embeds_scaling in zip(self.weights, self.conds, self.conds_alt, self.unconds, self.ipadapters, self.masks, self.weight_types, self.sigma_starts, self.sigma_ends, self.unfold_batch, self.embeds_scaling): if sigma <= sigma_start and sigma >= sigma_end: - if unfold_batch and cond.shape[0] > 1: - # Check AnimateDiff context window - if ad_params is not None and ad_params["sub_idxs"] is not None: - # if image length matches or exceeds full_length get sub_idx images - if cond.shape[0] >= ad_params["full_length"]: - cond = torch.Tensor(cond[ad_params["sub_idxs"]]) - uncond = torch.Tensor(uncond[ad_params["sub_idxs"]]) - # otherwise get sub_idxs images - else: - cond = tensor_to_size(cond, ad_params["full_length"]) - uncond = tensor_to_size(uncond, ad_params["full_length"]) - cond = cond[ad_params["sub_idxs"]] - uncond = uncond[ad_params["sub_idxs"]] - - cond = tensor_to_size(cond, batch_prompt) - uncond = tensor_to_size(uncond, batch_prompt) - - k_cond = ipadapter.ip_layers.to_kvs[self.k_key](cond) - k_uncond = ipadapter.ip_layers.to_kvs[self.k_key](uncond) - v_cond = ipadapter.ip_layers.to_kvs[self.v_key](cond) - v_uncond = ipadapter.ip_layers.to_kvs[self.v_key](uncond) - else: - k_cond = ipadapter.ip_layers.to_kvs[self.k_key](cond).repeat(batch_prompt, 1, 1) - k_uncond = ipadapter.ip_layers.to_kvs[self.k_key](uncond).repeat(batch_prompt, 1, 1) - v_cond = ipadapter.ip_layers.to_kvs[self.v_key](cond).repeat(batch_prompt, 1, 1) - v_uncond = ipadapter.ip_layers.to_kvs[self.v_key](uncond).repeat(batch_prompt, 1, 1) - if weight_type == 'ease in': weight = weight * (0.05 + 0.95 * (1 - t_idx / self.layers)) elif weight_type == 'ease out': @@ -97,9 +72,68 @@ class CrossAttentionPatchImport: weight = weight * 0.2 elif weight_type == 'strong middle' and (block_type == 'input' or block_type == 'output'): weight = weight * 0.2 - elif weight_type.startswith('style transfer'): - if t_idx != 6: - weight = 0.0 + elif isinstance(weight, dict): + if t_idx not in weight: + continue + + weight = weight[t_idx] + + if cond_alt is not None and t_idx in cond_alt: + cond = cond_alt[t_idx] + del cond_alt + + if unfold_batch: + # Check AnimateDiff context window + if ad_params is not None and ad_params["sub_idxs"] is not None: + if isinstance(weight, torch.Tensor): + weight = tensor_to_size(weight, ad_params["full_length"]) + weight = torch.Tensor(weight[ad_params["sub_idxs"]]) + if torch.all(weight == 0): + continue + weight = weight.repeat(len(cond_or_uncond), 1, 1) # repeat for cond and uncond + elif weight == 0: + continue + + # if image length matches or exceeds full_length get sub_idx images + if cond.shape[0] >= ad_params["full_length"]: + cond = torch.Tensor(cond[ad_params["sub_idxs"]]) + uncond = torch.Tensor(uncond[ad_params["sub_idxs"]]) + # otherwise get sub_idxs images + else: + cond = tensor_to_size(cond, ad_params["full_length"]) + uncond = tensor_to_size(uncond, ad_params["full_length"]) + cond = cond[ad_params["sub_idxs"]] + uncond = uncond[ad_params["sub_idxs"]] + else: + if isinstance(weight, torch.Tensor): + weight = tensor_to_size(weight, batch_prompt) + if torch.all(weight == 0): + continue + weight = weight.repeat(len(cond_or_uncond), 1, 1) # repeat for cond and uncond + elif weight == 0: + continue + + cond = tensor_to_size(cond, batch_prompt) + uncond = tensor_to_size(uncond, batch_prompt) + + k_cond = ipadapter.ip_layers.to_kvs[self.k_key](cond) + k_uncond = ipadapter.ip_layers.to_kvs[self.k_key](uncond) + v_cond = ipadapter.ip_layers.to_kvs[self.v_key](cond) + v_uncond = ipadapter.ip_layers.to_kvs[self.v_key](uncond) + else: + # TODO: should we always convert the weights to a tensor? + if isinstance(weight, torch.Tensor): + weight = tensor_to_size(weight, batch_prompt) + if torch.all(weight == 0): + continue + weight = weight.repeat(len(cond_or_uncond), 1, 1) # repeat for cond and uncond + elif weight == 0: + continue + + k_cond = ipadapter.ip_layers.to_kvs[self.k_key](cond).repeat(batch_prompt, 1, 1) + k_uncond = ipadapter.ip_layers.to_kvs[self.k_key](uncond).repeat(batch_prompt, 1, 1) + v_cond = ipadapter.ip_layers.to_kvs[self.v_key](cond).repeat(batch_prompt, 1, 1) + v_uncond = ipadapter.ip_layers.to_kvs[self.v_key](uncond).repeat(batch_prompt, 1, 1) ip_k = torch.cat([(k_cond, k_uncond)[i] for i in cond_or_uncond], dim=0) ip_v = torch.cat([(v_cond, v_uncond)[i] for i in cond_or_uncond], dim=0) diff --git a/imports/ComfyUI_IPAdapter_plus/IPAdapterPlus.py b/imports/ComfyUI_IPAdapter_plus/IPAdapterPlus.py index 4ecba3b..01eaa2e 100644 --- a/imports/ComfyUI_IPAdapter_plus/IPAdapterPlus.py +++ b/imports/ComfyUI_IPAdapter_plus/IPAdapterPlus.py @@ -37,7 +37,7 @@ else: current_paths, _ = folder_paths.folder_names_and_paths["ipadapter"] folder_paths.folder_names_and_paths["ipadapter"] = (current_paths, folder_paths.supported_pt_extensions) -WEIGHT_TYPES = ["linear", "ease in", "ease out", 'ease in-out', 'reverse in-out', 'weak input', 'weak output', 'weak middle', 'strong middle', 'style transfer (SDXL)'] +WEIGHT_TYPES = ["linear", "ease in", "ease out", 'ease in-out', 'reverse in-out', 'weak input', 'weak output', 'weak middle', 'strong middle', 'style transfer', 'composition', 'strong style transfer'] """ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -148,8 +148,10 @@ def ipadapter_execute(model, clipvision, insightface=None, image=None, + image_composition=None, image_negative=None, weight=1.0, + weight_composition=1.0, weight_faceidv2=None, weight_type="linear", combine_embeds="concat", @@ -159,9 +161,12 @@ def ipadapter_execute(model, pos_embed=None, neg_embed=None, unfold_batch=False, - embeds_scaling='V only'): - dtype = torch.float16 if model_management.should_use_fp16() else torch.bfloat16 if model_management.should_use_bf16() else torch.float32 + embeds_scaling='V only', + layer_weights=None): device = model_management.get_torch_device() + dtype = model_management.unet_dtype() + if dtype not in [torch.float32, torch.float16, torch.bfloat16]: + dtype = torch.float16 if comfy.model_management.should_use_fp16() else torch.float32 is_full = "proj.3.weight" in ipadapter["image_proj"] is_portrait = "proj.2.weight" in ipadapter["image_proj"] and not "proj.3.weight" in ipadapter["image_proj"] and not "0.to_q_lora.down.weight" in ipadapter["ip_adapter"] @@ -171,10 +176,6 @@ def ipadapter_execute(model, output_cross_attention_dim = ipadapter["ip_adapter"]["1.to_k_ip.weight"].shape[1] is_sdxl = output_cross_attention_dim == 2048 - if weight_type == "style transfer (SDXL)" and not is_sdxl: - weight_type = "linear" - print("\033[33mINFO: 'Style Transfer' weight type is only available for SDXL models, falling back to 'linear'.\033[0m") - if is_faceid and not insightface: raise Exception("insightface model is required for FaceID models") @@ -187,6 +188,34 @@ def ipadapter_execute(model, if image is not None and image.shape[1] != image.shape[2]: print("\033[33mINFO: the IPAdapter reference image is not a square, CLIPImageProcessor will resize and crop it at the center. If the main focus of the picture is not in the middle the result might not be what you are expecting.\033[0m") + if isinstance(weight, list): + weight = torch.tensor(weight).unsqueeze(-1).unsqueeze(-1).to(device, dtype=dtype) if unfold_batch else weight[0] + + # special weight types + if layer_weights is not None and layer_weights != '': + weight = { int(k): float(v)*weight for k, v in [x.split(":") for x in layer_weights.split(",")] } + weight_type = "linear" + elif weight_type.startswith("style transfer"): + weight = { 6:weight } if is_sdxl else { 0:weight, 1:weight, 2:weight, 3:weight, 9:weight, 10:weight, 11:weight, 12:weight, 13:weight, 14:weight, 15:weight } + elif weight_type.startswith("composition"): + weight = { 3:weight } if is_sdxl else { 4:weight*0.25, 5:weight } + elif weight_type == "strong style transfer": + if is_sdxl: + weight = { 0:weight, 1:weight, 2:weight, 4:weight, 5:weight, 6:weight, 7:weight, 8:weight, 9:weight, 10:weight } + else: + weight = { 0:weight, 1:weight, 2:weight, 3:weight, 6:weight, 7:weight, 8:weight, 9:weight, 10:weight, 11:weight, 12:weight, 13:weight, 14:weight, 15:weight } + elif weight_type == "style and composition": + if is_sdxl: + weight = { 3:weight_composition, 6:weight } + else: + weight = { 0:weight, 1:weight, 2:weight, 3:weight, 4:weight_composition*0.25, 5:weight_composition, 9:weight, 10:weight, 11:weight, 12:weight, 13:weight, 14:weight, 15:weight } + elif weight_type == "strong style and composition": + if is_sdxl: + weight = { 0:weight, 1:weight, 2:weight, 3:weight_composition, 4:weight, 5:weight, 6:weight, 7:weight, 8:weight, 9:weight, 10:weight } + else: + weight = { 0:weight, 1:weight, 2:weight, 3:weight, 4:weight_composition, 5:weight_composition, 6:weight, 7:weight, 8:weight, 9:weight, 10:weight, 11:weight, 12:weight, 13:weight, 14:weight, 15:weight } + + img_comp_cond_embeds = None face_cond_embeds = None if is_faceid: if insightface is None: @@ -218,17 +247,24 @@ def ipadapter_execute(model, if image is not None: img_cond_embeds = encode_image_masked(clipvision, image) + if image_composition is not None: + img_comp_cond_embeds = encode_image_masked(clipvision, image_composition) if is_plus: img_cond_embeds = img_cond_embeds.penultimate_hidden_states image_negative = image_negative if image_negative is not None else torch.zeros([1, 224, 224, 3]) img_uncond_embeds = encode_image_masked(clipvision, image_negative).penultimate_hidden_states + if image_composition is not None: + img_comp_cond_embeds = img_comp_cond_embeds.penultimate_hidden_states else: img_cond_embeds = img_cond_embeds.image_embeds if not is_faceid else face_cond_embeds - if image_negative is not None: + if image_negative is not None and not is_faceid: img_uncond_embeds = encode_image_masked(clipvision, image_negative).image_embeds else: img_uncond_embeds = torch.zeros_like(img_cond_embeds) + if image_composition is not None: + img_comp_cond_embeds = img_comp_cond_embeds.image_embeds + del image, image_negative, image_composition elif pos_embed is not None: img_cond_embeds = pos_embed @@ -239,6 +275,7 @@ def ipadapter_execute(model, img_uncond_embeds = encode_image_masked(clipvision, torch.zeros([1, 224, 224, 3])).penultimate_hidden_states else: img_uncond_embeds = torch.zeros_like(img_cond_embeds) + del pos_embed, neg_embed else: raise Exception("Images or Embeds are required") @@ -247,6 +284,8 @@ def ipadapter_execute(model, img_cond_embeds = img_cond_embeds.to(device, dtype=dtype) img_uncond_embeds = img_uncond_embeds.to(device, dtype=dtype) + if img_comp_cond_embeds is not None: + img_comp_cond_embeds = img_comp_cond_embeds.to(device, dtype=dtype) # combine the embeddings if needed if combine_embeds != "concat" and img_cond_embeds.shape[0] > 1 and not unfold_batch: @@ -254,20 +293,29 @@ def ipadapter_execute(model, img_cond_embeds = torch.sum(img_cond_embeds, dim=0).unsqueeze(0) if face_cond_embeds is not None: face_cond_embeds = torch.sum(face_cond_embeds, dim=0).unsqueeze(0) + if img_comp_cond_embeds is not None: + img_comp_cond_embeds = torch.sum(img_comp_cond_embeds, dim=0).unsqueeze(0) elif combine_embeds == "subtract": img_cond_embeds = img_cond_embeds[0] - torch.mean(img_cond_embeds[1:], dim=0) img_cond_embeds = img_cond_embeds.unsqueeze(0) if face_cond_embeds is not None: face_cond_embeds = face_cond_embeds[0] - torch.mean(face_cond_embeds[1:], dim=0) face_cond_embeds = face_cond_embeds.unsqueeze(0) + if img_comp_cond_embeds is not None: + img_comp_cond_embeds = img_comp_cond_embeds[0] - torch.mean(img_comp_cond_embeds[1:], dim=0) + img_comp_cond_embeds = img_comp_cond_embeds.unsqueeze(0) elif combine_embeds == "average": img_cond_embeds = torch.mean(img_cond_embeds, dim=0).unsqueeze(0) if face_cond_embeds is not None: face_cond_embeds = torch.mean(face_cond_embeds, dim=0).unsqueeze(0) + if img_comp_cond_embeds is not None: + img_comp_cond_embeds = torch.mean(img_comp_cond_embeds, dim=0).unsqueeze(0) elif combine_embeds == "norm average": img_cond_embeds = torch.mean(img_cond_embeds / torch.norm(img_cond_embeds, dim=0, keepdim=True), dim=0).unsqueeze(0) if face_cond_embeds is not None: face_cond_embeds = torch.mean(face_cond_embeds / torch.norm(face_cond_embeds, dim=0, keepdim=True), dim=0).unsqueeze(0) + if img_comp_cond_embeds is not None: + img_comp_cond_embeds = torch.mean(img_comp_cond_embeds / torch.norm(img_comp_cond_embeds, dim=0, keepdim=True), dim=0).unsqueeze(0) img_uncond_embeds = img_uncond_embeds[0].unsqueeze(0) # TODO: better strategy for uncond could be to average them if attn_mask is not None: @@ -287,25 +335,31 @@ def ipadapter_execute(model, if is_faceid and is_plus: cond = ipa.get_image_embeds_faceid_plus(face_cond_embeds, img_cond_embeds, weight_faceidv2, is_faceidv2) - # TODO: check if noise helps with the uncod face embeds - uncod = ipa.get_image_embeds_faceid_plus(torch.zeros_like(face_cond_embeds), img_uncond_embeds, weight_faceidv2, is_faceidv2) + # TODO: check if noise helps with the uncond face embeds + uncond = ipa.get_image_embeds_faceid_plus(torch.zeros_like(face_cond_embeds), img_uncond_embeds, weight_faceidv2, is_faceidv2) else: - cond, uncod = ipa.get_image_embeds(img_cond_embeds, img_uncond_embeds) + cond, uncond = ipa.get_image_embeds(img_cond_embeds, img_uncond_embeds) + if img_comp_cond_embeds is not None: + cond_comp = ipa.get_image_embeds(img_comp_cond_embeds, img_uncond_embeds)[0] cond = cond.to(device, dtype=dtype) - uncod = uncod.to(device, dtype=dtype) + uncond = uncond.to(device, dtype=dtype) + cond_alt = None + if img_comp_cond_embeds is not None: + cond_alt = { 3: cond_comp.to(device, dtype=dtype) } - del img_cond_embeds, img_uncond_embeds + del img_cond_embeds, img_uncond_embeds, img_comp_cond_embeds, face_cond_embeds - sigma_start = model.model.model_sampling.percent_to_sigma(start_at) - sigma_end = model.model.model_sampling.percent_to_sigma(end_at) + sigma_start = model.get_model_object("model_sampling").percent_to_sigma(start_at) + sigma_end = model.get_model_object("model_sampling").percent_to_sigma(end_at) patch_kwargs = { "ipadapter": ipa, "number": 0, "weight": weight, "cond": cond, - "uncond": uncod, + "cond_alt": cond_alt, + "uncond": uncond, "weight_type": weight_type, "mask": attn_mask, "sigma_start": sigma_start, @@ -340,7 +394,210 @@ def ipadapter_execute(model, return model -class IPAdapterAdvancedImport: +""" +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + Loaders +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +""" +class IPAdapterUnifiedLoader: + def __init__(self): + self.lora = None + self.clipvision = { "file": None, "model": None } + self.ipadapter = { "file": None, "model": None } + self.insightface = { "provider": None, "model": None } + + @classmethod + def INPUT_TYPES(s): + return {"required": { + "model": ("MODEL", ), + "preset": (['LIGHT - SD1.5 only (low strength)', 'STANDARD (medium strength)', 'VIT-G (medium strength)', 'PLUS (high strength)', 'PLUS FACE (portraits)', 'FULL FACE - SD1.5 only (portraits stronger)'], ), + }, + "optional": { + "ipadapter": ("IPADAPTER", ), + }} + + RETURN_TYPES = ("MODEL", "IPADAPTER", ) + RETURN_NAMES = ("model", "ipadapter", ) + FUNCTION = "load_models" + CATEGORY = "ipadapter" + + def load_models(self, model, preset, lora_strength=0.0, provider="CPU", ipadapter=None): + pipeline = { "clipvision": { 'file': None, 'model': None }, "ipadapter": { 'file': None, 'model': None }, "insightface": { 'provider': None, 'model': None } } + if ipadapter is not None: + pipeline = ipadapter + + # 1. Load the clipvision model + clipvision_file = get_clipvision_file(preset) + if clipvision_file is None: + raise Exception("ClipVision model not found.") + + if clipvision_file != self.clipvision['file']: + if clipvision_file != pipeline['clipvision']['file']: + self.clipvision['file'] = clipvision_file + self.clipvision['model'] = load_clip_vision(clipvision_file) + print(f"\033[33mINFO: Clip Vision model loaded from {clipvision_file}\033[0m") + else: + self.clipvision = pipeline['clipvision'] + + # 2. Load the ipadapter model + is_sdxl = isinstance(model.model, (comfy.model_base.SDXL, comfy.model_base.SDXLRefiner, comfy.model_base.SDXL_instructpix2pix)) + ipadapter_file, is_insightface, lora_pattern = get_ipadapter_file(preset, is_sdxl) + if ipadapter_file is None: + raise Exception("IPAdapter model not found.") + + if ipadapter_file != self.ipadapter['file']: + if pipeline['ipadapter']['file'] != ipadapter_file: + self.ipadapter['file'] = ipadapter_file + self.ipadapter['model'] = ipadapter_model_loader(ipadapter_file) + print(f"\033[33mINFO: IPAdapter model loaded from {ipadapter_file}\033[0m") + else: + self.ipadapter = pipeline['ipadapter'] + + # 3. Load the lora model if needed + if lora_pattern is not None: + lora_file = get_lora_file(lora_pattern) + lora_model = None + if lora_file is None: + raise Exception("LoRA model not found.") + + if self.lora is not None: + if lora_file == self.lora['file']: + lora_model = self.lora['model'] + else: + self.lora = None + torch.cuda.empty_cache() + + if lora_model is None: + lora_model = comfy.utils.load_torch_file(lora_file, safe_load=True) + self.lora = { 'file': lora_file, 'model': lora_model } + print(f"\033[33mINFO: LoRA model loaded from {lora_file}\033[0m") + + if lora_strength > 0: + model, _ = load_lora_for_models(model, None, lora_model, lora_strength, 0) + + # 4. Load the insightface model if needed + if is_insightface: + if provider != self.insightface['provider']: + if pipeline['insightface']['provider'] != provider: + self.insightface['provider'] = provider + self.insightface['model'] = insightface_loader(provider) + print(f"\033[33mINFO: InsightFace model loaded with {provider} provider\033[0m") + else: + self.insightface = pipeline['insightface'] + + return (model, { 'clipvision': self.clipvision, 'ipadapter': self.ipadapter, 'insightface': self.insightface }, ) + +class IPAdapterUnifiedLoaderFaceID(IPAdapterUnifiedLoader): + @classmethod + def INPUT_TYPES(s): + return {"required": { + "model": ("MODEL", ), + "preset": (['FACEID', 'FACEID PLUS - SD1.5 only', 'FACEID PLUS V2', 'FACEID PORTRAIT (style transfer)'], ), + "lora_strength": ("FLOAT", { "default": 0.6, "min": 0, "max": 1, "step": 0.01 }), + "provider": (["CPU", "CUDA", "ROCM", "DirectML", "OpenVINO", "CoreML"], ), + }, + "optional": { + "ipadapter": ("IPADAPTER", ), + }} + + RETURN_NAMES = ("MODEL", "ipadapter", ) + CATEGORY = "ipadapter/faceid" + +class IPAdapterUnifiedLoaderCommunity(IPAdapterUnifiedLoader): + @classmethod + def INPUT_TYPES(s): + return {"required": { + "model": ("MODEL", ), + "preset": (['Composition',], ), + }, + "optional": { + "ipadapter": ("IPADAPTER", ), + }} + + CATEGORY = "ipadapter/loaders" + +class IPAdapterModelLoader: + @classmethod + def INPUT_TYPES(s): + return {"required": { "ipadapter_file": (folder_paths.get_filename_list("ipadapter"), )}} + + RETURN_TYPES = ("IPADAPTER",) + FUNCTION = "load_ipadapter_model" + CATEGORY = "ipadapter/loaders" + + def load_ipadapter_model(self, ipadapter_file): + ipadapter_file = folder_paths.get_full_path("ipadapter", ipadapter_file) + return (ipadapter_model_loader(ipadapter_file),) + +class IPAdapterInsightFaceLoader: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "provider": (["CPU", "CUDA", "ROCM"], ), + }, + } + + RETURN_TYPES = ("INSIGHTFACE",) + FUNCTION = "load_insightface" + CATEGORY = "ipadapter/loaders" + + def load_insightface(self, provider): + return (insightface_loader(provider),) + +""" +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + Main Apply Nodes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +""" +class IPAdapterSimple: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 3, "step": 0.05 }), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "weight_type": (['standard', 'prompt is more important', 'style transfer'], ), + }, + "optional": { + "attn_mask": ("MASK",), + } + } + + RETURN_TYPES = ("MODEL",) + FUNCTION = "apply_ipadapter" + CATEGORY = "ipadapter" + + def apply_ipadapter(self, model, ipadapter, image, weight, start_at, end_at, weight_type, attn_mask=None): + if weight_type.startswith("style"): + weight_type = "style transfer" + elif weight_type == "prompt is more important": + weight_type = "ease out" + else: + weight_type = "linear" + + ipa_args = { + "image": image, + "weight": weight, + "start_at": start_at, + "end_at": end_at, + "attn_mask": attn_mask, + "weight_type": weight_type, + "insightface": ipadapter['insightface']['model'] if 'insightface' in ipadapter else None, + } + + if 'ipadapter' not in ipadapter: + raise Exception("IPAdapter model not present in the pipeline. Please load the models with the IPAdapterUnifiedLoader node.") + if 'clipvision' not in ipadapter: + raise Exception("CLIPVision model not present in the pipeline. Please load the models with the IPAdapterUnifiedLoader node.") + + return (ipadapter_execute(model.clone(), ipadapter['ipadapter']['model'], ipadapter['clipvision']['model'], **ipa_args), ) + +class IPAdapterAdvanced: def __init__(self): self.unfold_batch = False @@ -351,7 +608,7 @@ class IPAdapterAdvancedImport: "model": ("MODEL", ), "ipadapter": ("IPADAPTER", ), "image": ("IMAGE",), - "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 3, "step": 0.05 }), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), "weight_type": (WEIGHT_TYPES, ), "combine_embeds": (["concat", "add", "subtract", "average", "norm average"],), "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), @@ -369,11 +626,26 @@ class IPAdapterAdvancedImport: FUNCTION = "apply_ipadapter" CATEGORY = "ipadapter" - def apply_ipadapter(self, model, ipadapter, image, weight, weight_type, start_at, end_at, combine_embeds="concat", weight_faceidv2=None, image_negative=None, clip_vision=None, attn_mask=None, insightface=None, embeds_scaling='V only'): + def apply_ipadapter(self, model, ipadapter, start_at, end_at, weight = 1.0, weight_style=1.0, weight_composition=1.0, expand_style=False, weight_type="linear", combine_embeds="concat", weight_faceidv2=None, image=None, image_style=None, image_composition=None, image_negative=None, clip_vision=None, attn_mask=None, insightface=None, embeds_scaling='V only', layer_weights=None): + is_sdxl = isinstance(model.model, (comfy.model_base.SDXL, comfy.model_base.SDXLRefiner, comfy.model_base.SDXL_instructpix2pix)) + + if image_style is not None: # we are doing style + composition transfer + if not is_sdxl: + raise Exception("Style + Composition transfer is only available for SDXL models at the moment.") # TODO: check feasibility for SD1.5 models + + image = image_style + weight = weight_style + if image_composition is None: + image_composition = image_style + + weight_type = "strong style and composition" if expand_style else "style and composition" + ipa_args = { "image": image, + "image_composition": image_composition, "image_negative": image_negative, "weight": weight, + "weight_composition": weight_composition, "weight_faceidv2": weight_faceidv2, "weight_type": weight_type, "combine_embeds": combine_embeds, @@ -382,7 +654,8 @@ class IPAdapterAdvancedImport: "attn_mask": attn_mask, "unfold_batch": self.unfold_batch, "embeds_scaling": embeds_scaling, - "insightface": insightface if insightface is not None else ipadapter['insightface']['model'] if 'insightface' in ipadapter else None + "insightface": insightface if insightface is not None else ipadapter['insightface']['model'] if 'insightface' in ipadapter else None, + "layer_weights": layer_weights, } if 'ipadapter' in ipadapter: @@ -399,10 +672,113 @@ class IPAdapterAdvancedImport: return (ipadapter_execute(model.clone(), ipadapter_model, clip_vision, **ipa_args), ) +class IPAdapterBatch(IPAdapterAdvanced): + def __init__(self): + self.unfold_batch = True + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "weight_type": (WEIGHT_TYPES, ), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } +class IPAdapterStyleComposition(IPAdapterAdvanced): + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image_style": ("IMAGE",), + "image_composition": ("IMAGE",), + "weight_style": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "weight_composition": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "expand_style": ("BOOLEAN", { "default": False }), + "combine_embeds": (["concat", "add", "subtract", "average", "norm average"], {"default": "average"}), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } -class IPAdapterTiledImport: + CATEGORY = "ipadapter/style_composition" + +class IPAdapterStyleCompositionBatch(IPAdapterStyleComposition): + def __init__(self): + self.unfold_batch = True + + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image_style": ("IMAGE",), + "image_composition": ("IMAGE",), + "weight_style": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "weight_composition": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "expand_style": ("BOOLEAN", { "default": False }), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } + +class IPAdapterFaceID(IPAdapterAdvanced): + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 3, "step": 0.05 }), + "weight_faceidv2": ("FLOAT", { "default": 1.0, "min": -1, "max": 5.0, "step": 0.05 }), + "weight_type": (WEIGHT_TYPES, ), + "combine_embeds": (["concat", "add", "subtract", "average", "norm average"],), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + "insightface": ("INSIGHTFACE",), + } + } + + CATEGORY = "ipadapter/faceid" + +class IPAAdapterFaceIDBatch(IPAdapterFaceID): + def __init__(self): + self.unfold_batch = True + +class IPAdapterTiled: def __init__(self): self.unfold_batch = False @@ -431,7 +807,7 @@ class IPAdapterTiledImport: RETURN_TYPES = ("MODEL", "IMAGE", "MASK", ) RETURN_NAMES = ("MODEL", "tiles", "masks", ) FUNCTION = "apply_tiled" - CATEGORY = "ipadapter" + CATEGORY = "ipadapter/tiled" def apply_tiled(self, model, ipadapter, image, weight, weight_type, start_at, end_at, sharpening, combine_embeds="concat", image_negative=None, attn_mask=None, clip_vision=None, embeds_scaling='V only'): # 1. Select the models @@ -537,9 +913,274 @@ class IPAdapterTiledImport: return (model, torch.cat(tiles), torch.cat(masks), ) +class IPAdapterTiledBatch(IPAdapterTiled): + def __init__(self): + self.unfold_batch = True + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 3, "step": 0.05 }), + "weight_type": (WEIGHT_TYPES, ), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "sharpening": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.05 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } -class PrepImageForClipVisionImport: +class IPAdapterEmbeds: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "pos_embed": ("EMBEDS",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 3, "step": 0.05 }), + "weight_type": (WEIGHT_TYPES, ), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + }, + "optional": { + "neg_embed": ("EMBEDS",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } + + RETURN_TYPES = ("MODEL",) + FUNCTION = "apply_ipadapter" + CATEGORY = "ipadapter/embeds" + + def apply_ipadapter(self, model, ipadapter, pos_embed, weight, weight_type, start_at, end_at, neg_embed=None, attn_mask=None, clip_vision=None, embeds_scaling='V only'): + ipa_args = { + "pos_embed": pos_embed, + "neg_embed": neg_embed, + "weight": weight, + "weight_type": weight_type, + "start_at": start_at, + "end_at": end_at, + "attn_mask": attn_mask, + "embeds_scaling": embeds_scaling, + } + + if 'ipadapter' in ipadapter: + ipadapter_model = ipadapter['ipadapter']['model'] + clip_vision = clip_vision if clip_vision is not None else ipadapter['clipvision']['model'] + else: + ipadapter_model = ipadapter + clip_vision = clip_vision + + if clip_vision is None and neg_embed is None: + raise Exception("Missing CLIPVision model.") + + del ipadapter + + return (ipadapter_execute(model.clone(), ipadapter_model, clip_vision, **ipa_args), ) + +class IPAdapterMS(IPAdapterAdvanced): + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL", ), + "ipadapter": ("IPADAPTER", ), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1, "max": 5, "step": 0.05 }), + "weight_faceidv2": ("FLOAT", { "default": 1.0, "min": -1, "max": 5.0, "step": 0.05 }), + "weight_type": (WEIGHT_TYPES, ), + "combine_embeds": (["concat", "add", "subtract", "average", "norm average"],), + "start_at": ("FLOAT", { "default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "end_at": ("FLOAT", { "default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }), + "embeds_scaling": (['V only', 'K+V', 'K+V w/ C penalty', 'K+mean(V) w/ C penalty'], ), + "layer_weights": ("STRING", { "default": "", "multiline": True }), + }, + "optional": { + "image_negative": ("IMAGE",), + "attn_mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + "insightface": ("INSIGHTFACE",), + } + } + + CATEGORY = "ipadapter/dev" + +""" +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + Helpers +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +""" +class IPAdapterEncoder: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "ipadapter": ("IPADAPTER",), + "image": ("IMAGE",), + "weight": ("FLOAT", { "default": 1.0, "min": -1.0, "max": 3.0, "step": 0.01 }), + }, + "optional": { + "mask": ("MASK",), + "clip_vision": ("CLIP_VISION",), + } + } + + RETURN_TYPES = ("EMBEDS", "EMBEDS",) + RETURN_NAMES = ("pos_embed", "neg_embed",) + FUNCTION = "encode" + CATEGORY = "ipadapter/embeds" + + def encode(self, ipadapter, image, weight, mask=None, clip_vision=None): + if 'ipadapter' in ipadapter: + ipadapter_model = ipadapter['ipadapter']['model'] + clip_vision = clip_vision if clip_vision is not None else ipadapter['clipvision']['model'] + else: + ipadapter_model = ipadapter + clip_vision = clip_vision + + if clip_vision is None: + raise Exception("Missing CLIPVision model.") + + is_plus = "proj.3.weight" in ipadapter_model["image_proj"] or "latents" in ipadapter_model["image_proj"] or "perceiver_resampler.proj_in.weight" in ipadapter_model["image_proj"] + + # resize and crop the mask to 224x224 + if mask is not None and mask.shape[1:3] != torch.Size([224, 224]): + mask = mask.unsqueeze(1) + transforms = T.Compose([ + T.CenterCrop(min(mask.shape[2], mask.shape[3])), + T.Resize((224, 224), interpolation=T.InterpolationMode.BICUBIC, antialias=True), + ]) + mask = transforms(mask).squeeze(1) + #mask = T.Resize((image.shape[1], image.shape[2]), interpolation=T.InterpolationMode.BICUBIC, antialias=True)(mask.unsqueeze(1)).squeeze(1) + + img_cond_embeds = encode_image_masked(clip_vision, image, mask) + + if is_plus: + img_cond_embeds = img_cond_embeds.penultimate_hidden_states + img_uncond_embeds = encode_image_masked(clip_vision, torch.zeros([1, 224, 224, 3])).penultimate_hidden_states + else: + img_cond_embeds = img_cond_embeds.image_embeds + img_uncond_embeds = torch.zeros_like(img_cond_embeds) + + if weight != 1: + img_cond_embeds = img_cond_embeds * weight + + return (img_cond_embeds, img_uncond_embeds, ) + +class IPAdapterCombineEmbeds: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "embed1": ("EMBEDS",), + "method": (["concat", "add", "subtract", "average", "norm average", "max", "min"], ), + }, + "optional": { + "embed2": ("EMBEDS",), + "embed3": ("EMBEDS",), + "embed4": ("EMBEDS",), + "embed5": ("EMBEDS",), + }} + + RETURN_TYPES = ("EMBEDS",) + FUNCTION = "batch" + CATEGORY = "ipadapter/embeds" + + def batch(self, embed1, method, embed2=None, embed3=None, embed4=None, embed5=None): + if method=='concat' and embed2 is None and embed3 is None and embed4 is None and embed5 is None: + return (embed1, ) + + embeds = [embed1, embed2, embed3, embed4, embed5] + embeds = [embed for embed in embeds if embed is not None] + embeds = torch.cat(embeds, dim=0) + + if method == "add": + embeds = torch.sum(embeds, dim=0).unsqueeze(0) + elif method == "subtract": + embeds = embeds[0] - torch.mean(embeds[1:], dim=0) + embeds = embeds.unsqueeze(0) + elif method == "average": + embeds = torch.mean(embeds, dim=0).unsqueeze(0) + elif method == "norm average": + embeds = torch.mean(embeds / torch.norm(embeds, dim=0, keepdim=True), dim=0).unsqueeze(0) + elif method == "max": + embeds = torch.max(embeds, dim=0).values.unsqueeze(0) + elif method == "min": + embeds = torch.min(embeds, dim=0).values.unsqueeze(0) + + return (embeds, ) + +class IPAdapterNoise: + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "type": (["fade", "dissolve", "gaussian", "shuffle"], ), + "strength": ("FLOAT", { "default": 1.0, "min": 0, "max": 1, "step": 0.05 }), + "blur": ("INT", { "default": 0, "min": 0, "max": 32, "step": 1 }), + }, + "optional": { + "image_optional": ("IMAGE",), + } + } + + RETURN_TYPES = ("IMAGE",) + FUNCTION = "make_noise" + CATEGORY = "ipadapter/utils" + + def make_noise(self, type, strength, blur, image_optional=None): + if image_optional is None: + image = torch.zeros([1, 224, 224, 3]) + else: + transforms = T.Compose([ + T.CenterCrop(min(image_optional.shape[1], image_optional.shape[2])), + T.Resize((224, 224), interpolation=T.InterpolationMode.BICUBIC, antialias=True), + ]) + image = transforms(image_optional.permute([0,3,1,2])).permute([0,2,3,1]) + + seed = int(torch.sum(image).item()) % 1000000007 # hash the image to get a seed, grants predictability + torch.manual_seed(seed) + + if type == "fade": + noise = torch.rand_like(image) + noise = image * (1 - strength) + noise * strength + elif type == "dissolve": + mask = (torch.rand_like(image) < strength).float() + noise = torch.rand_like(image) + noise = image * (1-mask) + noise * mask + elif type == "gaussian": + noise = torch.randn_like(image) * strength + noise = image + noise + elif type == "shuffle": + transforms = T.Compose([ + T.ElasticTransform(alpha=75.0, sigma=(1-strength)*3.5), + T.RandomVerticalFlip(p=1.0), + T.RandomHorizontalFlip(p=1.0), + ]) + image = transforms(image.permute([0,3,1,2])).permute([0,2,3,1]) + noise = torch.randn_like(image) * (strength*0.75) + noise = image * (1-noise) + noise + + del image + noise = torch.clamp(noise, 0, 1) + + if blur > 0: + if blur % 2 == 0: + blur += 1 + noise = T.functional.gaussian_blur(noise.permute([0,3,1,2]), blur).permute([0,2,3,1]) + + return (noise, ) + +class PrepImageForClipVision: @classmethod def INPUT_TYPES(s): return {"required": { @@ -553,7 +1194,7 @@ class PrepImageForClipVisionImport: RETURN_TYPES = ("IMAGE",) FUNCTION = "prep_image" - CATEGORY = "ipadapter" + CATEGORY = "ipadapter/utils" def prep_image(self, image, interpolation="LANCZOS", crop_position="center", sharpening=0.0): size = (224, 224) @@ -602,64 +1243,172 @@ class PrepImageForClipVisionImport: return (output, ) +class IPAdapterSaveEmbeds: + def __init__(self): + self.output_dir = folder_paths.get_output_directory() -class IPAdapterNoiseImport: @classmethod def INPUT_TYPES(s): - return { - "required": { - "type": (["fade", "dissolve", "gaussian", "shuffle"], ), - "strength": ("FLOAT", { "default": 1.0, "min": 0, "max": 1, "step": 0.05 }), - "blur": ("INT", { "default": 0, "min": 0, "max": 32, "step": 1 }), + return {"required": { + "embeds": ("EMBEDS",), + "filename_prefix": ("STRING", {"default": "IP_embeds"}) }, - "optional": { - "image_optional": ("IMAGE",), - } } - RETURN_TYPES = ("IMAGE",) - FUNCTION = "make_noise" - CATEGORY = "ipadapter" + RETURN_TYPES = () + FUNCTION = "save" + OUTPUT_NODE = True + CATEGORY = "ipadapter/embeds" - def make_noise(self, type, strength, blur, image_optional=None): - if image_optional is None: - image = torch.zeros([1, 224, 224, 3]) - else: - transforms = T.Compose([ - T.CenterCrop(min(image_optional.shape[1], image_optional.shape[2])), - T.Resize((224, 224), interpolation=T.InterpolationMode.BICUBIC, antialias=True), - ]) - image = transforms(image_optional.permute([0,3,1,2])).permute([0,2,3,1]) + def save(self, embeds, filename_prefix): + full_output_folder, filename, counter, subfolder, filename_prefix = folder_paths.get_save_image_path(filename_prefix, self.output_dir) + file = f"{filename}_{counter:05}.ipadpt" + file = os.path.join(full_output_folder, file) - seed = int(torch.sum(image).item()) % 1000000007 # hash the image to get a seed, grants predictability - torch.manual_seed(seed) + torch.save(embeds, file) + return (None, ) - if type == "fade": - noise = torch.rand_like(image) - noise = image * (1 - strength) + noise * strength - elif type == "dissolve": - mask = (torch.rand_like(image) < strength).float() - noise = torch.rand_like(image) - noise = image * (1-mask) + noise * mask - elif type == "gaussian": - noise = torch.randn_like(image) * strength - noise = image + noise - elif type == "shuffle": - transforms = T.Compose([ - T.ElasticTransform(alpha=75.0, sigma=(1-strength)*3.5), - T.RandomVerticalFlip(p=1.0), - T.RandomHorizontalFlip(p=1.0), - ]) - image = transforms(image.permute([0,3,1,2])).permute([0,2,3,1]) - noise = torch.randn_like(image) * (strength*0.75) - noise = image * (1-noise) + noise +class IPAdapterLoadEmbeds: + @classmethod + def INPUT_TYPES(s): + input_dir = folder_paths.get_input_directory() + files = [os.path.relpath(os.path.join(root, file), input_dir) for root, dirs, files in os.walk(input_dir) for file in files if file.endswith('.ipadpt')] + return {"required": {"embeds": [sorted(files), ]}, } - del image - noise = torch.clamp(noise, 0, 1) + RETURN_TYPES = ("EMBEDS", ) + FUNCTION = "load" + CATEGORY = "ipadapter/embeds" - if blur > 0: - if blur % 2 == 0: - blur += 1 - noise = T.functional.gaussian_blur(noise.permute([0,3,1,2]), blur).permute([0,2,3,1]) + def load(self, embeds): + path = folder_paths.get_annotated_filepath(embeds) + return (torch.load(path).cpu(), ) - return (noise, ) \ No newline at end of file +class IPAdapterWeights: + @classmethod + def INPUT_TYPES(s): + return {"required": { + "weights": ("STRING", {"default": '1.0', "multiline": True }), + "timing": (["custom", "linear", "ease_in_out", "ease_in", "ease_out", "reverse_in_out", "random"], ), + "frames": ("INT", {"default": 0, "min": 0, "max": 9999, "step": 1 }), + "start_frame": ("INT", {"default": 0, "min": 0, "max": 9999, "step": 1 }), + "end_frame": ("INT", {"default": 9999, "min": 0, "max": 9999, "step": 1 }), + }, + } + + RETURN_TYPES = ("FLOAT",) + FUNCTION = "weights" + + CATEGORY = "ipadapter/utils" + + def weights(self, weights, timing, frames, start_frame, end_frame): + import random + + # convert the string to a list of floats separated by commas or newlines + weights = weights.replace("\n", ",") + weights = [float(weight) for weight in weights.split(",") if weight.strip() != ""] + + if timing != "custom": + start = 0.0 + end = 1.0 + + if len(weights) > 0: + start = weights[0] + end = weights[-1] + + weights = [] + + end_frame = min(end_frame, frames) + duration = end_frame - start_frame + if start_frame > 0: + weights.extend([start] * start_frame) + + for i in range(duration): + n = duration - 1 + if timing == "linear": + weights.append(start + (end - start) * i / n) + elif timing == "ease_in_out": + weights.append(start + (end - start) * (1 - math.cos(i / n * math.pi)) / 2) + elif timing == "ease_in": + weights.append(start + (end - start) * math.sin(i / n * math.pi / 2)) + elif timing == "ease_out": + weights.append(start + (end - start) * (1 - math.cos(i / n * math.pi / 2))) + elif timing == "reverse_in_out": + weights.append(start + (end - start) * (1 - math.sin((1 - i / n) * math.pi / 2))) + elif timing == "random": + weights.append(random.uniform(start, end)) + weights[-1] = end if timing != "random" else weights[-1] + + if end_frame < frames: + weights.extend([end] * (frames - end_frame)) + + if len(weights) == 0: + weights = [0.0] + + return (weights, ) + +""" +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + Register +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +""" +NODE_CLASS_MAPPINGS = { + # Main Apply Nodes + "IPAdapter": IPAdapterSimple, + "IPAdapterAdvanced": IPAdapterAdvanced, + "IPAdapterBatch": IPAdapterBatch, + "IPAdapterFaceID": IPAdapterFaceID, + "IPAAdapterFaceIDBatch": IPAAdapterFaceIDBatch, + "IPAdapterTiled": IPAdapterTiled, + "IPAdapterTiledBatch": IPAdapterTiledBatch, + "IPAdapterEmbeds": IPAdapterEmbeds, + "IPAdapterStyleComposition": IPAdapterStyleComposition, + "IPAdapterStyleCompositionBatch": IPAdapterStyleCompositionBatch, + "IPAdapterMS": IPAdapterMS, + + # Loaders + "IPAdapterUnifiedLoader": IPAdapterUnifiedLoader, + "IPAdapterUnifiedLoaderFaceID": IPAdapterUnifiedLoaderFaceID, + "IPAdapterModelLoader": IPAdapterModelLoader, + "IPAdapterInsightFaceLoader": IPAdapterInsightFaceLoader, + "IPAdapterUnifiedLoaderCommunity": IPAdapterUnifiedLoaderCommunity, + + # Helpers + "IPAdapterEncoder": IPAdapterEncoder, + "IPAdapterCombineEmbeds": IPAdapterCombineEmbeds, + "IPAdapterNoise": IPAdapterNoise, + "PrepImageForClipVision": PrepImageForClipVision, + "IPAdapterSaveEmbeds": IPAdapterSaveEmbeds, + "IPAdapterLoadEmbeds": IPAdapterLoadEmbeds, + "IPAdapterWeights": IPAdapterWeights, +} + +NODE_DISPLAY_NAME_MAPPINGS = { + # Main Apply Nodes + "IPAdapter": "IPAdapter", + "IPAdapterAdvanced": "IPAdapter Advanced", + "IPAdapterBatch": "IPAdapter Batch (Adv.)", + "IPAdapterFaceID": "IPAdapter FaceID", + "IPAAdapterFaceIDBatch": "IPAdapter FaceID Batch", + "IPAdapterTiled": "IPAdapter Tiled", + "IPAdapterTiledBatch": "IPAdapter Tiled Batch", + "IPAdapterEmbeds": "IPAdapter Embeds", + "IPAdapterStyleComposition": "IPAdapter Style & Composition SDXL", + "IPAdapterStyleCompositionBatch": "IPAdapter Style & Composition Batch SDXL", + "IPAdapterMS": "IPAdapter Mad Scientist", + + # Loaders + "IPAdapterUnifiedLoader": "IPAdapter Unified Loader", + "IPAdapterUnifiedLoaderFaceID": "IPAdapter Unified Loader FaceID", + "IPAdapterModelLoader": "IPAdapter Model Loader", + "IPAdapterInsightFaceLoader": "IPAdapter InsightFace Loader", + "IPAdapterUnifiedLoaderCommunity": "IPAdapter Unified Loader Community", + + # Helpers + "IPAdapterEncoder": "IPAdapter Encoder", + "IPAdapterCombineEmbeds": "IPAdapter Combine Embeds", + "IPAdapterNoise": "IPAdapter Noise", + "PrepImageForClipVision": "Prep Image For ClipVision", + "IPAdapterSaveEmbeds": "IPAdapter Save Embeds", + "IPAdapterLoadEmbeds": "IPAdapter Load Embeds", + "IPAdapterWeights": "IPAdapter Weights", +} \ No newline at end of file diff --git a/imports/ComfyUI_IPAdapter_plus/NODES.md b/imports/ComfyUI_IPAdapter_plus/NODES.md new file mode 100644 index 0000000..e4c33a2 --- /dev/null +++ b/imports/ComfyUI_IPAdapter_plus/NODES.md @@ -0,0 +1,54 @@ +# Nodes reference + +Below I'm trying to document all the nodes. It's still very incomplete, be sure to check back later. + +## Loaders + +### :knot: IPAdapter Unified Loader + +Loads the full stack of models needed for IPAdapter to function. The returned object will contain information regarding the **ipadapter** and **clip vision models**. + +Multiple unified loaders should always be daisy chained through the `ipadapter` in/out. **Failing to do so will cause all models to be loaded twice.** For **the first** unified loader the `ipadapter` input **should never be connected**. + +#### Inputs +- **model**, main ComfyUI model pipeline + +#### Optional Inputs +- **ipadapter**, it's important to note that this is optional and used exclusively to daisy chain unified loaders. **The `ipadapter` input is never connected in the first `IPAdapter Unified Loader` of the chain.** + +#### Outputs +- **model**, the model pipeline is used exclusively for configuration, the model comes out of this node untouched and it can be considered a reroute. Note that this is different from the Unified Loader FaceID that actually alters the model with a LoRA. +- **ipadapter**, connect this to any ipadater node. Each node will automatically detect if the `ipadapter` object contains the full stack of models or just one (like in the case [IPAdapter Model Loader](#ipadapter-model-loader)). + +### :knot: IPAdapter Model Loader + +Loads the IPAdapter model only. The returned object will be the IPAdapter model contrary to the [Unified loader](#ipadapter-unified-loader) that contains the full stack of models. + +#### Configuration parameters +- **ipadapter_file**, the main IPAdapter model. It must be located into `ComfyUI/models/ipadapter` or in any path specified in the `extra_model_paths.yaml` configuration file. + +#### Outputs +- **IPADAPTER**, contains the loaded model only. Note that `IPADAPTER` will have a different structure when loaded by the [Unified Loader](#ipadapter-unified-loader). + +## Main IPAdapter Apply Nodes + +### :knot: IPAdapter Advanced + +This node contains all the options to fine tune the IPAdapter models. It is a drop in replacement for the old `IPAdapter Apply` that is no longer available. If you have an old workflow, delete the existing `IPadapter Apply` node, add `IPAdapter Advanced` and connect all the pipes as before. + +#### Inputs +- **model**, main model pipeline. +- **ipadapter**, the IPAdapter model. It can be connected to the [IPAdapter Model Loader](#ipadapter-model-loader) or any of the Unified Loaders. If a Unified loader is used anywhere in the workflow and you don't need a different model, it's always adviced to reuse the previous `ipadapter` pipeline. +- **image**, the reference image used to generate the positive conditioning. It should be a square image, other aspect ratios are automatically cropped in the center. + +#### Optional inputs +- **image_negative**, image used to generate the negative conditioning. This is optional and normally handled by the code. It is possible to send noise or actually any image to instruct the model about what we don't want to see in the composition. +- **attn_mask**, a mask that will be applied during the image generation. **The mask should have the same size or at least the same aspect ratio of the latent**. The mask will define the area of influence of the IPAdapter models on the final image. Black zones won't be affected, white zones will get maximum influence. It can be a grayscale mask. +- **clip_vision**, this is optional if using any of the Unified loaders. If using the [IPAdapter Model Loader](#knot-ipadapter-model-loader) you also have to provide the clip vision model with a `Load CLIP Vision` node. + +#### Configuration parameters +- **weight**, weight of the IPAdapter model. For `linear` `weight_type` (the default), a good starting point is 0.8. If you use other weight types you can experiment with higher values. +- **weight_type**, this is how the IPAdapter is applied to the UNet block. For example `ease-in` means that the input blocks have higher weight than the output ones. `week input` means that the whole input block has lower weight. `style transfer (SDXL)` only works with SDXL and it's a very powerful tool to tranfer only the style of an image but not its content. This parameter hugely impacts how the composition reacts to the text prompting. +- **combine_embeds**, when sending more than one reference image the embeddings can be sent one after the other (`concat`) or combined in various ways. For low spec GPUs it is adviced to `average` the embeds if you send multiple images. `subtract` subtracts the embeddings of the second image to the first; in case of 3 or more images they are averaged and subtracted to the first. +- **start_at/end_at**, this is the timestepping. Defines at what percentage point of the generation to start applying the IPAdapter model. The initial steps are the most important so if you start later (eg: `start_at=0.3`) the generated image will have a very light conditioning. +- **embeds_scaling**, the way the IPAdapter models are applied to the K,V. This parameter has a small impact on how the model reacts to text prompting. `K+mean(V) w/ C penalty` grants good quality at high weights (>1.0) without burning the image. diff --git a/imports/ComfyUI_IPAdapter_plus/README.md b/imports/ComfyUI_IPAdapter_plus/README.md index f7fed54..85f9bab 100644 --- a/imports/ComfyUI_IPAdapter_plus/README.md +++ b/imports/ComfyUI_IPAdapter_plus/README.md @@ -1,45 +1,49 @@ # ComfyUI IPAdapter plus [ComfyUI](https://github.com/comfyanonymous/ComfyUI) reference implementation for [IPAdapter](https://github.com/tencent-ailab/IP-Adapter/) models. -IPAdapter implementation that follows the ComfyUI way of doing things. The code is memory efficient, fast, and shouldn't break with Comfy updates. +The IPAdapter are very powerful models for image-to-image conditioning. The subject or even just the style of the reference image(s) can be easily transferred to a generation. Think of it as a 1-image lora. -# Open source for you but not free for me... +# Sponsorship -I started working on IPAdapter because I needed it for my work. As the project evolved I'm inevitably receiving feature requests, bug reports and support requests. +
-I'm an open source advocate and I'm happy to share all my code for free but maintaining the IPAdapter, the [Essentials](https://github.com/cubiq/ComfyUI_essentials), [InstantID](https://github.com/cubiq/ComfyUI_InstantID) and [Face Analysis](https://github.com/cubiq/ComfyUI_FaceAnalysis) takes time. +**[:heart: Github Sponsor](https://github.com/sponsors/cubiq) | [:coin: Paypal](https://paypal.me/matt3o)** -**I'm not expecting donations but if you are making a profit from my projects it is only fair that you give something back.** I'm talking especially to companies here, I know the struggles of being a freelancer. +
-Please contact me if you are interested in a sponsorship at _matt3o@gmail_ or consider a contribution via [PayPal](https://paypal.me/matt3o) (Matteo "matt3o" Spinelli, Firenze, IT). That will help maintaining the code, adding new features and working on better documentation. +If you like my work and wish to see updates and new features please consider sponsoring my projects. -And in that regard I really need to thank [Nathan Shipley](https://www.nathanshipley.com/) for his generous donation. Go check his website, he's terribly talented. +- [ComfyUI IPAdapter Plus](https://github.com/cubiq/ComfyUI_IPAdapter_plus) +- [ComfyUI InstantID (Native)](https://github.com/cubiq/ComfyUI_InstantID) +- [ComfyUI Essentials](https://github.com/cubiq/ComfyUI_essentials) +- [ComfyUI FaceAnalysis](https://github.com/cubiq/ComfyUI_FaceAnalysis) +- [Comfy Dungeon](https://github.com/cubiq/Comfy_Dungeon) -## :warning: IPAdapter V2: complete Code rewrite warning +Not to mention the documentation and videos tutorials. Check my **ComfyUI Advanced Understanding** videos on YouTube for example, [part 1](https://www.youtube.com/watch?v=_C7kR2TFIX0) and [part 2](https://www.youtube.com/watch?v=ijqXnW_9gzc) -A code cleanup was long overdue and with the occasion I also added a few new important features. The code should be faster and should take less resources but with such an important code rewrite it's inevitable to have introduced some new bugs. +The only way to keep the code open and free is by sponsoring its development. The more sponsorships the more time I can dedicate to my open source projects. -**At the moment I'm releasing this completely undocumented!** I will post better documentation and video tutorials in the coming days. In the meantime you can check the `example` directory for most of the old and new features. +Please consider a [Github Sponsorship](https://github.com/sponsors/cubiq) or [PayPal donation](https://paypal.me/matt3o) (Matteo "matt3o" Spinelli). For sponsorships of $50+, let me know if you'd like to be mentioned in this readme file, you can find me on [Discord](https://latent.vision/discord) or _matt3o :snail: gmail.com_. ## Important updates +**2024/04/12**: Added scheduled weights. Useful for animations. + +**2024/04/09**: Added experimental Style/Composition transfer for SD1.5. The results are often not as good as SDXL. Optimal weight seems to be from 0.8 to 2.0. The **Style+Composition node doesn't work for SD1.5** at the moment, you can only alter either the Style or the Composition, I need more time for testing. Old workflows will still work **but you may need to refresh the page and re-select the weight type!** + +**2024/04/04**: Added Style & Composition node. It's now possible to apply both Style and Composition from the same node + +**2024/04/01**: Added Composition only transfer weight type for SDXL + +**2024/03/27**: Added Style transfer weight type for SDXL + **2024/03/23**: Complete code rewrite!. **This is a breaking update!** Your previous workflows won't work and you'll need to recreate them. You've been warned! After the update, refresh your browser, delete the old IPAdapter nodes and create the new ones. -**2024/02/02**: Added experimental [tiled IPAdapter](#tiled-ipadapter). It lets you easily handle reference images that are not square. Can be useful for upscaling. +*(I removed all previous updates because they were about the previous version of the extension)* -**2024/01/19**: Support for FaceID Portrait models. +## Example workflows -**2024/01/16**: Notably increased quality of FaceID Plus/v2 models. Check the [comparison](https://github.com/cubiq/ComfyUI_IPAdapter_plus/issues/195) of all face models. - -*(previous updates removed for better readability)* - -## What is it? - -The IPAdapter are very powerful models for image-to-image conditioning. Given one or more reference images you can do variations augmented by text prompt, controlnets and masks. Think of it as a 1-image lora. - -## Example workflow - -The [example directory](./examples/) has many workflows that cover all IPAdapter functionalities. +The [examples directory](./examples/) has many workflows that cover all IPAdapter functionalities. ![IPAdapter Example workflow](./examples/demo_workflow.jpg) @@ -53,74 +57,110 @@ The [example directory](./examples/) has many workflows that cover all IPAdapter The following videos are about the previous version of IPAdapter, but they still contain valuable information. -**:nerd_face: [Basic usage video](https://youtu.be/7m9ZZFU3HWo)** - -**:rocket: [Advanced features video](https://www.youtube.com/watch?v=mJQ62ly7jrg)** - -**:japanese_goblin: [Attention Masking video](https://www.youtube.com/watch?v=vqG1VXKteQg)** - -**:movie_camera: [Animation Features video](https://www.youtube.com/watch?v=ddYbhv3WgWw)** +**:nerd_face: [Basic usage video](https://youtu.be/7m9ZZFU3HWo)**, **:rocket: [Advanced features video](https://www.youtube.com/watch?v=mJQ62ly7jrg)**, **:japanese_goblin: [Attention Masking video](https://www.youtube.com/watch?v=vqG1VXKteQg)**, **:movie_camera: [Animation Features video](https://www.youtube.com/watch?v=ddYbhv3WgWw)** ## Installation -Download or git clone this repository inside `ComfyUI/custom_nodes/` directory or use the Manager. Beware that the automatic update of the manager sometimes doesn't work and you may need to upgrade manually. +Download or git clone this repository inside `ComfyUI/custom_nodes/` directory or use the Manager. IPAdapter always requires the latest version of ComfyUI. If something doesn't work be sure to upgrade. Beware that the automatic update of the manager sometimes doesn't work and you may need to upgrade manually. -IPAdapter always requires the latest version of ComfyUI. If something doesn't work be sure to upgrade! +There's now a *Unified Model Loader*, for it to work you need to name the files exactly as described below. The legacy loaders work with any file name but you have to select them manually. The models can be placed into sub-directories. -There's now an *Unified Model Loader*, for it to work you need to name the files exactly how it is described below. +Remember you can also use any custom location setting an `ipadapter` entry in the `extra_model_paths.yaml` file. -The pre-trained models are available on [huggingface](https://huggingface.co/h94/IP-Adapter), download and place them in the `ComfyUI/models/ipadapter` directory (create it if not present). You can also use any custom location setting an `ipadapter` entry in the `extra_model_paths.yaml` file. +- `/ComfyUI/models/clip_vision` + - [CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/image_encoder/model.safetensors), download and rename + - [CLIP-ViT-bigG-14-laion2B-39B-b160k.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/image_encoder/model.safetensors), download and rename +- `/ComfyUI/models/ipadapter`, create it if not present + - [ip-adapter_sd15.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15.safetensors), Basic model, average strength + - [ip-adapter_sd15_light_v11.bin](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_light_v11.bin), Light impact model + - [ip-adapter-plus_sd15.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-plus_sd15.safetensors), Plus model, very strong + - [ip-adapter-plus-face_sd15.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-plus-face_sd15.safetensors), Face model, portraits + - [ip-adapter-full-face_sd15.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-full-face_sd15.safetensors), Stronger face model, not necessarily better + - [ip-adapter_sd15_vit-G.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_vit-G.safetensors), Base model, **requires bigG clip vision encoder** + - [ip-adapter_sdxl_vit-h.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter_sdxl_vit-h.safetensors), SDXL model + - [ip-adapter-plus_sdxl_vit-h.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter-plus_sdxl_vit-h.safetensors), SDXL plus model + - [ip-adapter-plus-face_sdxl_vit-h.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter-plus-face_sdxl_vit-h.safetensors), SDXL face model + - [ip-adapter_sdxl.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter_sdxl.safetensors), vit-G SDXL model, **requires bigG clip vision encoder** + - **Deprecated** [ip-adapter_sd15_light.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_light.safetensors), v1.0 Light impact model -IPAdapter also needs the image encoders. You need the [CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/models/image_encoder/model.safetensors) and [CLIP-ViT-bigG-14-laion2B-39B-b160k.safetensors](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/image_encoder/model.safetensors) image encoders, you may already have them. If you don't, download them but **be careful because the file name is the same for both!** Rename them and place them in the `ComfyUI/models/clip_vision/` directory. +**FaceID** models require `insightface`, you need to install it in your ComfyUI environment. Check [this issue](https://github.com/cubiq/ComfyUI_IPAdapter_plus/issues/162) for help. Remember that most FaceID models also need a LoRA. -The following table shows the combination of Checkpoint and Image encoder to use for each IPAdapter Model. Any Tensor size mismatch you may get it is likely caused by a wrong combination. +For the Unified Loader to work the files need to be named exactly as shown in the table below. -| SD v. | IPadapter | Img encoder | Notes | -|---|---|---|---| -| v1.5 | [ip-adapter_sd15](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15.safetensors) | ViT-H | Basic model, average strength | -| v1.5 | [ip-adapter_sd15_light](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_light.safetensors) | ViT-H | Light model, very light impact | -| v1.5 | [ip-adapter_sd15_light_v11](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_light_v11.bin) | ViT-H | Updated light model | -| v1.5 | [ip-adapter-plus_sd15](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-plus_sd15.safetensors) | ViT-H | Plus model, very strong | -| v1.5 | [ip-adapter-plus-face_sd15](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-plus-face_sd15.safetensors) | ViT-H | Face model, use only for faces | -| v1.5 | [ip-adapter-full-face_sd15](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter-full-face_sd15.safetensors) | ViT-H | Stronger face model, not necessarily better | -| v1.5 | [ip-adapter_sd15_vit-G](https://huggingface.co/h94/IP-Adapter/resolve/main/models/ip-adapter_sd15_vit-G.safetensors) | ViT-bigG | Base model trained with a bigG encoder | -| SDXL | [ip-adapter_sdxl](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter_sdxl.safetensors) | ViT-bigG | Base SDXL model, mostly deprecated | -| SDXL | [ip-adapter_sdxl_vit-h](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter_sdxl_vit-h.safetensors) | ViT-H | New base SDXL model | -| SDXL | [ip-adapter-plus_sdxl_vit-h](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter-plus_sdxl_vit-h.safetensors) | ViT-H | SDXL plus model, stronger | -| SDXL | [ip-adapter-plus-face_sdxl_vit-h](https://huggingface.co/h94/IP-Adapter/resolve/main/sdxl_models/ip-adapter-plus-face_sdxl_vit-h.safetensors) | ViT-H | SDXL face model | +- `/ComfyUI/models/ipadapter` + - [ip-adapter-faceid_sd15.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sd15.bin), base FaceID model + - [ip-adapter-faceid-plusv2_sd15.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sd15.bin), FaceID plus v2 + - [ip-adapter-faceid-portrait-v11_sd15.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-portrait-v11_sd15.bin), text prompt style transfer for portraits + - [ip-adapter-faceid_sdxl.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sdxl.bin), SDXL base FaceID + - [ip-adapter-faceid-plusv2_sdxl.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sdxl.bin), SDXL plus v2 + - [ip-adapter-faceid-portrait_sdxl.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-portrait_sdxl.bin), SDXL text prompt style transfer + - **Deprecated** [ip-adapter-faceid-plus_sd15.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plus_sd15.bin), FaceID plus v1 + - **Deprecated** [ip-adapter-faceid-portrait_sd15.bin](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-portrait_sd15.bin), v1 of the portrait model -**FaceID** requires `insightface`, you need to install them in your ComfyUI environment. Check [this issue](https://github.com/cubiq/ComfyUI_IPAdapter_plus/issues/162) for help. +Most FaceID models require a LoRA. If you use the `IPAdapter Unified Loader FaceID` it will be loaded automatically if you follow the naming convention. Otherwise you have to load them manually, be careful each FaceID model has to be paired with its own specific LoRA. -When the dependencies are satisfied you need: +- `/ComfyUI/models/loras` + - [ip-adapter-faceid_sd15_lora.safetensors](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sd15_lora.safetensors) + - [ip-adapter-faceid-plusv2_sd15_lora.safetensors](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sd15_lora.safetensors) + - [ip-adapter-faceid_sdxl_lora.safetensors](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sdxl_lora.safetensors), SDXL FaceID LoRA + - [ip-adapter-faceid-plusv2_sdxl_lora.safetensors](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sdxl_lora.safetensors), SDXL plus v2 LoRA + - **Deprecated** [ip-adapter-faceid-plus_sd15_lora.safetensors](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plus_sd15_lora.safetensors), LoRA for the deprecated FaceID plus v1 model -| SD v. | IPadapter | Img encoder | Lora | -|---|---|---|---| -| v1.5 | [FaceID](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sd15.bin) | (not used¹) | [FaceID Lora](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sd15_lora.safetensors) | -| v1.5 | [FaceID Plus](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plus_sd15.bin) | ViT-H | [FaceID Plus Lora](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plus_sd15_lora.safetensors) | -| v1.5 | [FaceID Plus v2](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sd15.bin) | ViT-H | [FaceID Plus v2 Lora](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sd15_lora.safetensors) | -| v1.5 | [FaceID Portrait](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-portrait_sd15.bin) | (not used¹)| not needed | -| SDXL | [FaceID](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sdxl.bin) | (not used¹) | [FaceID SDXL Lora](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid_sdxl_lora.safetensors) | -| SDXL | [FaceID Plus v2](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sdxl.bin) | ViT-H | [FaceID SDXL Lora](https://huggingface.co/h94/IP-Adapter-FaceID/resolve/main/ip-adapter-faceid-plusv2_sdxl_lora.safetensors) | +All models can be found on [huggingface](https://huggingface.co/h94). +### Community's models -¹ The base FaceID model doesn't make use of a CLIP vision encoder. Remember to pair any FaceID model together with any other Face model to make it more effective. +The community has baked some interesting IPAdapter models. -The loras need to be placed into `ComfyUI/models/loras/` directory. +- `/ComfyUI/models/ipadapter` + - [ip_plus_composition_sd15.safetensors](https://huggingface.co/ostris/ip-composition-adapter/resolve/main/ip_plus_composition_sd15.safetensors), general composition ignoring style and content, more about it [here](https://huggingface.co/ostris/ip-composition-adapter) + - [ip_plus_composition_sdxl.safetensors](https://huggingface.co/ostris/ip-composition-adapter/resolve/main/ip_plus_composition_sdxl.safetensors), SDXL version + +if you know of other models please let me know and I will add them to the unified loader. ## Generic suggestions -There's a basic workflow included in this repo and a few examples in the [examples](./examples/) directory. Usually it's a good idea to lower the `weight` to at least `0.8` and increase the steps a little. +There are many workflows included in the [examples](./examples/) directory. Please check them before asking for support. -## Documentation soon to come... +Usually it's a good idea to lower the `weight` to at least `0.8` and increase the number steps. To increase adherece to the prompt you may try to change the **weight type** in the `IPAdapter Advanced` node. -Working on it! +## Nodes reference + +I'm (slowly) documenting all nodes. Please check the [Nodes reference](./NODES.md). ## Troubleshooting -Please check the [troubleshooting](https://github.com/cubiq/ComfyUI_IPAdapter_plus/issues/108) before posting a new issue. Alse remember to check the previous closed issues. +Please check the [troubleshooting](https://github.com/cubiq/ComfyUI_IPAdapter_plus/issues/108) before posting a new issue. Also remember to check the previous closed issues. + +## Current sponsors + +It's only thanks to generous sponsors that **the whole community** can enjoy open and free software. Please join me in thanking the following companies and individuals! + +### Gold sponsors + +[![Kaiber.ai](https://f.latent.vision/imgs/kaiber.png)](https://kaiber.ai/) + +### Companies supporting my projects + +- [RunComfy](https://www.runcomfy.com/) (ComfyUI Cloud) + +### Esteemed individuals + +- [Jack Gane](https://github.com/ganeJackS) +- [Nathan Shipley](https://www.nathanshipley.com/) + +### One-time Extraordinaire + +- [Eric Rollei](https://github.com/EricRollei) +- [francaleu](https://github.com/francaleu) +- [Neta.art](https://github.com/talesofai) +- [Samwise Wang](https://github.com/tzwm) +- _And all private sponsors, you know who you are!_ ## Credits - [IPAdapter](https://github.com/tencent-ailab/IP-Adapter/) +- [InstantStyle](https://github.com/InstantStyle/InstantStyle) +- [B-Lora](https://github.com/yardenfren1996/B-LoRA/) - [ComfyUI](https://github.com/comfyanonymous/ComfyUI) -- [laksjdjf](https://github.com/laksjdjf/IPAdapter-ComfyUI/) +- [laksjdjf](https://github.com/laksjdjf/) diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_advanced.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_advanced.json index ae712dd..b085824 100644 --- a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_advanced.json +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_advanced.json @@ -57,10 +57,10 @@ 1770, 710 ], - "size": [ - 529.7760009765616, - 582.3048192804504 - ], + "size": { + "0": 529.7760009765625, + "1": 582.3048095703125 + }, "flags": {}, "order": 11, "mode": 0, @@ -121,10 +121,10 @@ 1570, 700 ], - "size": [ - 140, - 46 - ], + "size": { + "0": 140, + "1": 46 + }, "flags": {}, "order": 10, "mode": 0, @@ -187,218 +187,6 @@ 1 ] }, - { - "id": 16, - "type": "CLIPVisionLoader", - "pos": [ - 308, - 161 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 2, - "mode": 0, - "outputs": [ - { - "name": "CLIP_VISION", - "type": "CLIP_VISION", - "links": [ - 24 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "CLIPVisionLoader" - }, - "widgets_values": [ - "IPAdapter_image_encoder_sd15.safetensors" - ] - }, - { - "id": 15, - "type": "IPAdapterModelLoader", - "pos": [ - 308, - 52 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 3, - "mode": 0, - "outputs": [ - { - "name": "IPADAPTER", - "type": "IPADAPTER", - "links": [ - 21 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "IPAdapterModelLoader" - }, - "widgets_values": [ - "ip-adapter-plus_sd15.safetensors" - ] - }, - { - "id": 14, - "type": "IPAdapterAdvanced", - "pos": [ - 793, - 304 - ], - "size": { - "0": 315, - "1": 254 - }, - "flags": {}, - "order": 8, - "mode": 0, - "inputs": [ - { - "name": "model", - "type": "MODEL", - "link": 20 - }, - { - "name": "ipadapter", - "type": "IPADAPTER", - "link": 21, - "slot_index": 1 - }, - { - "name": "image", - "type": "IMAGE", - "link": 26 - }, - { - "name": "image_negative", - "type": "IMAGE", - "link": null - }, - { - "name": "attn_mask", - "type": "MASK", - "link": null - }, - { - "name": "clip_vision", - "type": "CLIP_VISION", - "link": 24, - "slot_index": 5 - } - ], - "outputs": [ - { - "name": "MODEL", - "type": "MODEL", - "links": [ - 23 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "IPAdapterAdvanced" - }, - "widgets_values": [ - 0.8, - "linear", - "concat", - 0, - 1 - ] - }, - { - "id": 17, - "type": "PrepImageForClipVision", - "pos": [ - 798, - 145 - ], - "size": { - "0": 315, - "1": 106 - }, - "flags": {}, - "order": 7, - "mode": 0, - "inputs": [ - { - "name": "image", - "type": "IMAGE", - "link": 25 - } - ], - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 26 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "PrepImageForClipVision" - }, - "widgets_values": [ - "LANCZOS", - "top", - 0.15 - ] - }, - { - "id": 12, - "type": "LoadImage", - "pos": [ - 311, - 270 - ], - "size": [ - 315, - 314 - ], - "flags": {}, - "order": 4, - "mode": 0, - "outputs": [ - { - "name": "IMAGE", - "type": "IMAGE", - "links": [ - 25 - ], - "shape": 3, - "slot_index": 0 - }, - { - "name": "MASK", - "type": "MASK", - "links": null, - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "LoadImage" - }, - "widgets_values": [ - "girl_sitting.png", - "image" - ] - }, { "id": 6, "type": "CLIPTextEncode", @@ -495,6 +283,219 @@ "karras", 1 ] + }, + { + "id": 14, + "type": "IPAdapterAdvanced", + "pos": [ + 801, + 256 + ], + "size": { + "0": 315, + "1": 278 + }, + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 20 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 21, + "slot_index": 1 + }, + { + "name": "image", + "type": "IMAGE", + "link": 26 + }, + { + "name": "image_negative", + "type": "IMAGE", + "link": null + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": 24, + "slot_index": 5 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 23 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterAdvanced" + }, + "widgets_values": [ + 0.8, + "linear", + "concat", + 0, + 1, + "V only" + ] + }, + { + "id": 17, + "type": "PrepImageForClipVision", + "pos": [ + 797, + 87 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 25 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 26 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "PrepImageForClipVision" + }, + "widgets_values": [ + "LANCZOS", + "top", + 0.15 + ] + }, + { + "id": 15, + "type": "IPAdapterModelLoader", + "pos": [ + 308, + 52 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "IPADAPTER", + "type": "IPADAPTER", + "links": [ + 21 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "IPAdapterModelLoader" + }, + "widgets_values": [ + "ip-adapter-plus_sd15.safetensors" + ] + }, + { + "id": 16, + "type": "CLIPVisionLoader", + "pos": [ + 308, + 161 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 24 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors" + ] + }, + { + "id": 12, + "type": "LoadImage", + "pos": [ + 311, + 270 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 25 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "warrior_woman.png", + "image" + ] } ], "links": [ diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_combine_embeds.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_combine_embeds.json index f961f67..df9a6be 100644 --- a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_combine_embeds.json +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_combine_embeds.json @@ -38,40 +38,6 @@ 1 ] }, - { - "id": 16, - "type": "CLIPVisionLoader", - "pos": [ - 650, - 80 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 1, - "mode": 0, - "outputs": [ - { - "name": "CLIP_VISION", - "type": "CLIP_VISION", - "links": [ - 24, - 96, - 107, - 118 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "CLIPVisionLoader" - }, - "widgets_values": [ - "IPAdapter_image_encoder_sd15.safetensors" - ] - }, { "id": 3, "type": "KSampler", @@ -143,7 +109,7 @@ "1": 98 }, "flags": {}, - "order": 2, + "order": 1, "mode": 0, "outputs": [ { @@ -197,7 +163,7 @@ "1": 314 }, "flags": {}, - "order": 3, + "order": 2, "mode": 0, "outputs": [ { @@ -274,7 +240,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 9, @@ -332,7 +298,8 @@ "linear", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -499,7 +466,7 @@ "1": 58 }, "flags": {}, - "order": 4, + "order": 3, "mode": 0, "outputs": [ { @@ -577,7 +544,7 @@ "1": 314 }, "flags": {}, - "order": 5, + "order": 4, "mode": 0, "outputs": [ { @@ -613,7 +580,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 10, @@ -671,7 +638,8 @@ "linear", "add", 0, - 1 + 1, + "V only" ] }, { @@ -881,7 +849,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 12, @@ -939,7 +907,8 @@ "linear", "norm average", 0, - 1 + 1, + "V only" ] }, { @@ -951,7 +920,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 11, @@ -1009,7 +978,8 @@ "linear", "average", 0, - 1 + 1, + "V only" ] }, { @@ -1143,6 +1113,40 @@ "widgets_values": [ "IPAdapter" ] + }, + { + "id": 16, + "type": "CLIPVisionLoader", + "pos": [ + 650, + 80 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 5, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 24, + 96, + 107, + 118 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors" + ] } ], "links": [ diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_noise_injection.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_noise_injection.json index 0146c03..c32745f 100644 --- a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_noise_injection.json +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_noise_injection.json @@ -147,37 +147,6 @@ 1 ] }, - { - "id": 16, - "type": "CLIPVisionLoader", - "pos": [ - 308, - 161 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 2, - "mode": 0, - "outputs": [ - { - "name": "CLIP_VISION", - "type": "CLIP_VISION", - "links": [ - 24 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "CLIPVisionLoader" - }, - "widgets_values": [ - "IPAdapter_image_encoder_sd15.safetensors" - ] - }, { "id": 15, "type": "IPAdapterModelLoader", @@ -190,7 +159,7 @@ "1": 58 }, "flags": {}, - "order": 3, + "order": 2, "mode": 0, "outputs": [ { @@ -259,7 +228,7 @@ "1": 314 }, "flags": {}, - "order": 4, + "order": 3, "mode": 0, "outputs": [ { @@ -293,10 +262,10 @@ 728, 290 ], - "size": [ - 210, - 106 - ], + "size": { + "0": 210, + "1": 106 + }, "flags": {}, "order": 7, "mode": 0, @@ -337,7 +306,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 9, @@ -395,7 +364,8 @@ "linear", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -464,10 +434,10 @@ 1019, 405 ], - "size": [ - 210, - 106 - ], + "size": { + "0": 210, + "1": 106 + }, "flags": {}, "order": 8, "mode": 0, @@ -537,6 +507,37 @@ "properties": { "Node name for S&R": "VAEDecode" } + }, + { + "id": 16, + "type": "CLIPVisionLoader", + "pos": [ + 308, + 161 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 24 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors" + ] } ], "links": [ diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_portrait.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_portrait.json new file mode 100644 index 0000000..e7aa5c7 --- /dev/null +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_portrait.json @@ -0,0 +1,567 @@ +{ + "last_node_id": 20, + "last_link_id": 36, + "nodes": [ + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 1640, + 710 + ], + "size": { + "0": 140, + "1": 46 + }, + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": 8 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 9 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + } + }, + { + "id": 3, + "type": "KSampler", + "pos": [ + 1280, + 710 + ], + "size": { + "0": 315, + "1": 262 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 32 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 4 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 6 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 2 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 7 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "KSampler" + }, + "widgets_values": [ + 0, + "fixed", + 30, + 6.5, + "ddpm", + "karras", + 1 + ] + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1830, + 700 + ], + "size": { + "0": 529.7760009765625, + "1": 582.3048095703125 + }, + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 9 + } + ], + "properties": {}, + "widgets_values": [ + "IPAdapter" + ] + }, + { + "id": 20, + "type": "IPAdapterUnifiedLoaderFaceID", + "pos": [ + 460, + 60 + ], + "size": { + "0": 315, + "1": 126 + }, + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 36 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": null + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 35 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "links": [ + 34 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "IPAdapterUnifiedLoaderFaceID" + }, + "widgets_values": [ + "FACEID PORTRAIT (style transfer)", + 0.6, + "CPU" + ] + }, + { + "id": 4, + "type": "CheckpointLoaderSimple", + "pos": [ + 10, + 680 + ], + "size": { + "0": 315, + "1": 98 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 36 + ], + "slot_index": 0 + }, + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 3, + 5 + ], + "slot_index": 1 + }, + { + "name": "VAE", + "type": "VAE", + "links": [ + 8 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "sdxl/juggernautXL_version8Rundiffusion.safetensors" + ] + }, + { + "id": 5, + "type": "EmptyLatentImage", + "pos": [ + 870, + 1100 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 2 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyLatentImage" + }, + "widgets_values": [ + 1024, + 1024, + 1 + ] + }, + { + "id": 12, + "type": "LoadImage", + "pos": [ + 450, + 240 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 29 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "face2.jpg", + "image" + ] + }, + { + "id": 6, + "type": "CLIPTextEncode", + "pos": [ + 760, + 620 + ], + "size": { + "0": 422.84503173828125, + "1": 164.31304931640625 + }, + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 3 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 4 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "a watercolor painting of a woman on the beach\n\nhigh quality artistry" + ] + }, + { + "id": 7, + "type": "CLIPTextEncode", + "pos": [ + 760, + 850 + ], + "size": { + "0": 425.27801513671875, + "1": 180.6060791015625 + }, + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 5 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 6 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "photo, blurry, noisy, messy, lowres, jpeg, artifacts, ill, distorted, malformed, naked" + ] + }, + { + "id": 18, + "type": "IPAdapterFaceID", + "pos": [ + 850, + 190 + ], + "size": { + "0": 315, + "1": 322 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 35 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 34, + "slot_index": 1 + }, + { + "name": "image", + "type": "IMAGE", + "link": 29 + }, + { + "name": "image_negative", + "type": "IMAGE", + "link": null + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": null + }, + { + "name": "insightface", + "type": "INSIGHTFACE", + "link": null + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 32 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterFaceID" + }, + "widgets_values": [ + 0.65, + 1, + "linear", + "concat", + 0, + 1, + "V only" + ] + } + ], + "links": [ + [ + 2, + 5, + 0, + 3, + 3, + "LATENT" + ], + [ + 3, + 4, + 1, + 6, + 0, + "CLIP" + ], + [ + 4, + 6, + 0, + 3, + 1, + "CONDITIONING" + ], + [ + 5, + 4, + 1, + 7, + 0, + "CLIP" + ], + [ + 6, + 7, + 0, + 3, + 2, + "CONDITIONING" + ], + [ + 7, + 3, + 0, + 8, + 0, + "LATENT" + ], + [ + 8, + 4, + 2, + 8, + 1, + "VAE" + ], + [ + 9, + 8, + 0, + 9, + 0, + "IMAGE" + ], + [ + 29, + 12, + 0, + 18, + 2, + "IMAGE" + ], + [ + 32, + 18, + 0, + 3, + 0, + "MODEL" + ], + [ + 34, + 20, + 1, + 18, + 1, + "IPADAPTER" + ], + [ + 35, + 20, + 0, + 18, + 0, + "MODEL" + ], + [ + 36, + 4, + 0, + 20, + 0, + "MODEL" + ] + ], + "groups": [], + "config": {}, + "extra": {}, + "version": 0.4 +} \ No newline at end of file diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_style_composition.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_style_composition.json new file mode 100644 index 0000000..1e1791e --- /dev/null +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_style_composition.json @@ -0,0 +1,612 @@ +{ + "last_node_id": 16, + "last_link_id": 25, + "nodes": [ + { + "id": 7, + "type": "CLIPTextEncode", + "pos": [ + 690, + 840 + ], + "size": { + "0": 425.27801513671875, + "1": 180.6060791015625 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 5 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 6 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "blurry, noisy, messy, lowres, jpeg, artifacts, ill, distorted, malformed" + ] + }, + { + "id": 11, + "type": "IPAdapterUnifiedLoader", + "pos": [ + 335, + 430 + ], + "size": { + "0": 315, + "1": 78 + }, + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 10 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 21 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "links": [ + 22 + ], + "shape": 3, + "slot_index": 1 + } + ], + "properties": { + "Node name for S&R": "IPAdapterUnifiedLoader" + }, + "widgets_values": [ + "PLUS (high strength)" + ] + }, + { + "id": 12, + "type": "LoadImage", + "pos": [ + -102, + -46 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 25 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "black_car.jpg", + "image" + ] + }, + { + "id": 16, + "type": "LoadImage", + "pos": [ + 310, + -40 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 24 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "bw_texture_waves.jpg", + "image" + ] + }, + { + "id": 5, + "type": "EmptyLatentImage", + "pos": [ + 801, + 1097 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 2 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyLatentImage" + }, + "widgets_values": [ + 1024, + 1024, + 1 + ] + }, + { + "id": 6, + "type": "CLIPTextEncode", + "pos": [ + 690, + 610 + ], + "size": { + "0": 422.84503173828125, + "1": 164.31304931640625 + }, + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 3 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 4 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "sports car running fast on the highway\n\nhigh quality, detailed" + ] + }, + { + "id": 15, + "type": "IPAdapterStyleComposition", + "pos": [ + 772, + 219 + ], + "size": { + "0": 315, + "1": 322 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 21 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 22 + }, + { + "name": "image_style", + "type": "IMAGE", + "link": 24 + }, + { + "name": "image_composition", + "type": "IMAGE", + "link": 25 + }, + { + "name": "image_negative", + "type": "IMAGE", + "link": null + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": null + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 23 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterStyleComposition" + }, + "widgets_values": [ + 1.2, + 1, + false, + "average", + 0, + 1, + "V only" + ] + }, + { + "id": 3, + "type": "KSampler", + "pos": [ + 1247, + 586 + ], + "size": { + "0": 315, + "1": 262 + }, + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 23 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 4 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 6 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 2 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 7 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "KSampler" + }, + "widgets_values": [ + 0, + "fixed", + 30, + 6.5, + "dpmpp_2m", + "karras", + 1 + ] + }, + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 1615, + 586 + ], + "size": { + "0": 140, + "1": 46 + }, + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": 8 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 9 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + } + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1822, + 588 + ], + "size": [ + 691.0159878487498, + 716.6239849908982 + ], + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 9 + } + ], + "properties": {}, + "widgets_values": [ + "IPAdapter" + ] + }, + { + "id": 4, + "type": "CheckpointLoaderSimple", + "pos": [ + -72, + 657 + ], + "size": { + "0": 315, + "1": 98 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 10 + ], + "slot_index": 0 + }, + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 3, + 5 + ], + "slot_index": 1 + }, + { + "name": "VAE", + "type": "VAE", + "links": [ + 8 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "sdxl/AlbedoBaseXL.safetensors" + ] + } + ], + "links": [ + [ + 2, + 5, + 0, + 3, + 3, + "LATENT" + ], + [ + 3, + 4, + 1, + 6, + 0, + "CLIP" + ], + [ + 4, + 6, + 0, + 3, + 1, + "CONDITIONING" + ], + [ + 5, + 4, + 1, + 7, + 0, + "CLIP" + ], + [ + 6, + 7, + 0, + 3, + 2, + "CONDITIONING" + ], + [ + 7, + 3, + 0, + 8, + 0, + "LATENT" + ], + [ + 8, + 4, + 2, + 8, + 1, + "VAE" + ], + [ + 9, + 8, + 0, + 9, + 0, + "IMAGE" + ], + [ + 10, + 4, + 0, + 11, + 0, + "MODEL" + ], + [ + 21, + 11, + 0, + 15, + 0, + "MODEL" + ], + [ + 22, + 11, + 1, + 15, + 1, + "IPADAPTER" + ], + [ + 23, + 15, + 0, + 3, + 0, + "MODEL" + ], + [ + 24, + 16, + 0, + 15, + 2, + "IMAGE" + ], + [ + 25, + 12, + 0, + 15, + 3, + "IMAGE" + ] + ], + "groups": [], + "config": {}, + "extra": {}, + "version": 0.4 +} \ No newline at end of file diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_tiled.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_tiled.json index 48a02c8..c71a29f 100644 --- a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_tiled.json +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_tiled.json @@ -205,38 +205,6 @@ "in a peaceful spring morning a woman wearing a white shirt is sitting in a park on a bench\n\nhigh quality, detailed, diffuse light" ] }, - { - "id": 16, - "type": "CLIPVisionLoader", - "pos": [ - 250, - 180 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 2, - "mode": 0, - "outputs": [ - { - "name": "CLIP_VISION", - "type": "CLIP_VISION", - "links": [ - 32 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "CLIPVisionLoader" - }, - "widgets_values": [ - "IPAdapter_image_encoder_sd15.safetensors" - ] - }, { "id": 5, "type": "EmptyLatentImage", @@ -249,7 +217,7 @@ "1": 106 }, "flags": {}, - "order": 3, + "order": 2, "mode": 0, "outputs": [ { @@ -329,38 +297,6 @@ 1 ] }, - { - "id": 15, - "type": "IPAdapterModelLoader", - "pos": [ - 250, - 70 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 4, - "mode": 0, - "outputs": [ - { - "name": "IPADAPTER", - "type": "IPADAPTER", - "links": [ - 31 - ], - "shape": 3, - "slot_index": 0 - } - ], - "properties": { - "Node name for S&R": "IPAdapterModelLoader" - }, - "widgets_values": [ - "ip-adapter-plus_sd15.safetensors" - ] - }, { "id": 18, "type": "IPAdapterTiled", @@ -370,7 +306,7 @@ ], "size": { "0": 315, - "1": 278 + "1": 302 }, "flags": {}, "order": 7, @@ -439,7 +375,8 @@ "concat", 0, 1, - 0 + 0, + "V only" ] }, { @@ -467,6 +404,70 @@ "widgets_values": [ "IPAdapter" ] + }, + { + "id": 15, + "type": "IPAdapterModelLoader", + "pos": [ + 250, + 70 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "IPADAPTER", + "type": "IPADAPTER", + "links": [ + 31 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterModelLoader" + }, + "widgets_values": [ + "ip-adapter-plus_sd15.safetensors" + ] + }, + { + "id": 16, + "type": "CLIPVisionLoader", + "pos": [ + 250, + 180 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 32 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors" + ] } ], "links": [ diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weight_types.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weight_types.json index 7efe9aa..9bf9288 100644 --- a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weight_types.json +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weight_types.json @@ -137,76 +137,6 @@ 1 ] }, - { - "id": 16, - "type": "CLIPVisionLoader", - "pos": [ - 308, - 161 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 2, - "mode": 0, - "outputs": [ - { - "name": "CLIP_VISION", - "type": "CLIP_VISION", - "links": [ - 24, - 38, - 49, - 60, - 71 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "CLIPVisionLoader" - }, - "widgets_values": [ - "IPAdapter_image_encoder_sd15.safetensors" - ] - }, - { - "id": 15, - "type": "IPAdapterModelLoader", - "pos": [ - 308, - 52 - ], - "size": { - "0": 315, - "1": 58 - }, - "flags": {}, - "order": 3, - "mode": 0, - "outputs": [ - { - "name": "IPADAPTER", - "type": "IPADAPTER", - "links": [ - 21, - 36, - 47, - 58, - 69 - ], - "shape": 3 - } - ], - "properties": { - "Node name for S&R": "IPAdapterModelLoader" - }, - "widgets_values": [ - "ip-adapter-plus_sd15.safetensors" - ] - }, { "id": 12, "type": "LoadImage", @@ -219,7 +149,7 @@ "1": 314 }, "flags": {}, - "order": 4, + "order": 2, "mode": 0, "outputs": [ { @@ -509,7 +439,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 7, @@ -567,7 +497,8 @@ "linear", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -579,7 +510,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 8, @@ -637,7 +568,8 @@ "ease in", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -748,7 +680,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 9, @@ -806,7 +738,8 @@ "ease out", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -1094,7 +1027,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 10, @@ -1152,7 +1085,8 @@ "ease in-out", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -1164,7 +1098,7 @@ ], "size": { "0": 315, - "1": 254 + "1": 278 }, "flags": {}, "order": 11, @@ -1222,7 +1156,8 @@ "reverse in-out", "concat", 0, - 1 + 1, + "V only" ] }, { @@ -1266,6 +1201,76 @@ "widgets_values": [ "closeup of a fierce warrior woman wearing a full armor at the end of a battle. cherry blossoms\n\nhigh quality, detailed" ] + }, + { + "id": 15, + "type": "IPAdapterModelLoader", + "pos": [ + 308, + 52 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 3, + "mode": 0, + "outputs": [ + { + "name": "IPADAPTER", + "type": "IPADAPTER", + "links": [ + 21, + 36, + 47, + 58, + 69 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "IPAdapterModelLoader" + }, + "widgets_values": [ + "ip-adapter-plus_sd15.safetensors" + ] + }, + { + "id": 16, + "type": "CLIPVisionLoader", + "pos": [ + 308, + 161 + ], + "size": { + "0": 315, + "1": 58 + }, + "flags": {}, + "order": 4, + "mode": 0, + "outputs": [ + { + "name": "CLIP_VISION", + "type": "CLIP_VISION", + "links": [ + 24, + 38, + 49, + 60, + 71 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "CLIPVisionLoader" + }, + "widgets_values": [ + "CLIP-ViT-H-14-laion2B-s32B-b79K.safetensors" + ] } ], "links": [ diff --git a/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weights.json b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weights.json new file mode 100644 index 0000000..8388cb3 --- /dev/null +++ b/imports/ComfyUI_IPAdapter_plus/examples/ipadapter_weights.json @@ -0,0 +1,733 @@ +{ + "last_node_id": 21, + "last_link_id": 37, + "nodes": [ + { + "id": 7, + "type": "CLIPTextEncode", + "pos": [ + 690, + 840 + ], + "size": { + "0": 425.27801513671875, + "1": 180.6060791015625 + }, + "flags": {}, + "order": 8, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 5 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 6 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "blurry, noisy, messy, lowres, jpeg, artifacts, ill, distorted, malformed" + ] + }, + { + "id": 8, + "type": "VAEDecode", + "pos": [ + 1570, + 700 + ], + "size": { + "0": 140, + "1": 46 + }, + "flags": {}, + "order": 11, + "mode": 0, + "inputs": [ + { + "name": "samples", + "type": "LATENT", + "link": 7 + }, + { + "name": "vae", + "type": "VAE", + "link": 8 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 9 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "VAEDecode" + } + }, + { + "id": 5, + "type": "EmptyLatentImage", + "pos": [ + 801, + 1097 + ], + "size": [ + 315, + 106 + ], + "flags": {}, + "order": 5, + "mode": 0, + "inputs": [ + { + "name": "batch_size", + "type": "INT", + "link": 35, + "widget": { + "name": "batch_size" + } + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 2 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "EmptyLatentImage" + }, + "widgets_values": [ + 512, + 512, + 6 + ] + }, + { + "id": 6, + "type": "CLIPTextEncode", + "pos": [ + 690, + 610 + ], + "size": { + "0": 422.84503173828125, + "1": 164.31304931640625 + }, + "flags": {}, + "order": 7, + "mode": 0, + "inputs": [ + { + "name": "clip", + "type": "CLIP", + "link": 3 + } + ], + "outputs": [ + { + "name": "CONDITIONING", + "type": "CONDITIONING", + "links": [ + 4 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "CLIPTextEncode" + }, + "widgets_values": [ + "in a peaceful spring morning a woman wearing a white shirt is sitting in a park on a bench\n\nhigh quality, detailed, diffuse light" + ] + }, + { + "id": 3, + "type": "KSampler", + "pos": [ + 1210, + 700 + ], + "size": { + "0": 315, + "1": 262 + }, + "flags": {}, + "order": 10, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 31 + }, + { + "name": "positive", + "type": "CONDITIONING", + "link": 4 + }, + { + "name": "negative", + "type": "CONDITIONING", + "link": 6 + }, + { + "name": "latent_image", + "type": "LATENT", + "link": 2 + } + ], + "outputs": [ + { + "name": "LATENT", + "type": "LATENT", + "links": [ + 7 + ], + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "KSampler" + }, + "widgets_values": [ + 0, + "fixed", + 30, + 6.5, + "ddpm", + "karras", + 1 + ] + }, + { + "id": 12, + "type": "LoadImage", + "pos": [ + 311, + 270 + ], + "size": { + "0": 315, + "1": 314 + }, + "flags": {}, + "order": 0, + "mode": 0, + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 25 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "MASK", + "type": "MASK", + "links": null, + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "LoadImage" + }, + "widgets_values": [ + "warrior_woman.png", + "image" + ] + }, + { + "id": 17, + "type": "PrepImageForClipVision", + "pos": [ + 797, + 87 + ], + "size": { + "0": 315, + "1": 106 + }, + "flags": {}, + "order": 3, + "mode": 0, + "inputs": [ + { + "name": "image", + "type": "IMAGE", + "link": 25 + } + ], + "outputs": [ + { + "name": "IMAGE", + "type": "IMAGE", + "links": [ + 30 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "PrepImageForClipVision" + }, + "widgets_values": [ + "LANCZOS", + "top", + 0.15 + ] + }, + { + "id": 20, + "type": "IPAdapterWeights", + "pos": [ + 757, + 318 + ], + "size": [ + 263.5047280787487, + 183.75987616018006 + ], + "flags": {}, + "order": 4, + "mode": 0, + "inputs": [ + { + "name": "frames", + "type": "INT", + "link": 34, + "widget": { + "name": "frames" + }, + "slot_index": 0 + } + ], + "outputs": [ + { + "name": "FLOAT", + "type": "FLOAT", + "links": [ + 32 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterWeights" + }, + "widgets_values": [ + "1.0,0.0", + "linear", + 6, + 0, + 9999 + ] + }, + { + "id": 21, + "type": "PrimitiveNode", + "pos": [ + 340, + 1093 + ], + "size": { + "0": 210, + "1": 82 + }, + "flags": {}, + "order": 1, + "mode": 0, + "outputs": [ + { + "name": "INT", + "type": "INT", + "links": [ + 34, + 35 + ], + "widget": { + "name": "frames" + }, + "slot_index": 0 + } + ], + "title": "frames", + "properties": { + "Run widget replace on values": false + }, + "widgets_values": [ + 6, + "fixed" + ] + }, + { + "id": 19, + "type": "IPAdapterBatch", + "pos": [ + 1173, + 251 + ], + "size": [ + 315, + 254 + ], + "flags": {}, + "order": 9, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 37 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": 29 + }, + { + "name": "image", + "type": "IMAGE", + "link": 30 + }, + { + "name": "image_negative", + "type": "IMAGE", + "link": null + }, + { + "name": "attn_mask", + "type": "MASK", + "link": null + }, + { + "name": "clip_vision", + "type": "CLIP_VISION", + "link": null + }, + { + "name": "weight", + "type": "FLOAT", + "link": 32, + "widget": { + "name": "weight" + }, + "slot_index": 6 + } + ], + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 31 + ], + "shape": 3, + "slot_index": 0 + } + ], + "properties": { + "Node name for S&R": "IPAdapterBatch" + }, + "widgets_values": [ + 1, + "linear", + 0, + 1, + "V only" + ] + }, + { + "id": 18, + "type": "IPAdapterUnifiedLoader", + "pos": [ + 303, + 132 + ], + "size": { + "0": 315, + "1": 78 + }, + "flags": {}, + "order": 6, + "mode": 0, + "inputs": [ + { + "name": "model", + "type": "MODEL", + "link": 36 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "link": null + } + ], + "outputs": [ + { + "name": "model", + "type": "MODEL", + "links": [ + 37 + ], + "shape": 3, + "slot_index": 0 + }, + { + "name": "ipadapter", + "type": "IPADAPTER", + "links": [ + 29 + ], + "shape": 3 + } + ], + "properties": { + "Node name for S&R": "IPAdapterUnifiedLoader" + }, + "widgets_values": [ + "PLUS (high strength)" + ] + }, + { + "id": 4, + "type": "CheckpointLoaderSimple", + "pos": [ + -79, + 712 + ], + "size": { + "0": 315, + "1": 98 + }, + "flags": {}, + "order": 2, + "mode": 0, + "outputs": [ + { + "name": "MODEL", + "type": "MODEL", + "links": [ + 36 + ], + "slot_index": 0 + }, + { + "name": "CLIP", + "type": "CLIP", + "links": [ + 3, + 5 + ], + "slot_index": 1 + }, + { + "name": "VAE", + "type": "VAE", + "links": [ + 8 + ], + "slot_index": 2 + } + ], + "properties": { + "Node name for S&R": "CheckpointLoaderSimple" + }, + "widgets_values": [ + "sd15/realisticVisionV51_v51VAE.safetensors" + ] + }, + { + "id": 9, + "type": "SaveImage", + "pos": [ + 1770, + 710 + ], + "size": [ + 556.2374508110479, + 892.1895739499892 + ], + "flags": {}, + "order": 12, + "mode": 0, + "inputs": [ + { + "name": "images", + "type": "IMAGE", + "link": 9 + } + ], + "properties": {}, + "widgets_values": [ + "IPAdapter" + ] + } + ], + "links": [ + [ + 2, + 5, + 0, + 3, + 3, + "LATENT" + ], + [ + 3, + 4, + 1, + 6, + 0, + "CLIP" + ], + [ + 4, + 6, + 0, + 3, + 1, + "CONDITIONING" + ], + [ + 5, + 4, + 1, + 7, + 0, + "CLIP" + ], + [ + 6, + 7, + 0, + 3, + 2, + "CONDITIONING" + ], + [ + 7, + 3, + 0, + 8, + 0, + "LATENT" + ], + [ + 8, + 4, + 2, + 8, + 1, + "VAE" + ], + [ + 9, + 8, + 0, + 9, + 0, + "IMAGE" + ], + [ + 25, + 12, + 0, + 17, + 0, + "IMAGE" + ], + [ + 29, + 18, + 1, + 19, + 1, + "IPADAPTER" + ], + [ + 30, + 17, + 0, + 19, + 2, + "IMAGE" + ], + [ + 31, + 19, + 0, + 3, + 0, + "MODEL" + ], + [ + 32, + 20, + 0, + 19, + 6, + "FLOAT" + ], + [ + 34, + 21, + 0, + 20, + 0, + "INT" + ], + [ + 35, + 21, + 0, + 5, + 0, + "INT" + ], + [ + 36, + 4, + 0, + 18, + 0, + "MODEL" + ], + [ + 37, + 18, + 0, + 19, + 0, + "MODEL" + ] + ], + "groups": [], + "config": {}, + "extra": {}, + "version": 0.4 +} \ No newline at end of file diff --git a/imports/ComfyUI_IPAdapter_plus/utils.py b/imports/ComfyUI_IPAdapter_plus/utils.py index b1fab58..e08b1d1 100644 --- a/imports/ComfyUI_IPAdapter_plus/utils.py +++ b/imports/ComfyUI_IPAdapter_plus/utils.py @@ -34,7 +34,7 @@ def get_ipadapter_file(preset, is_sdxl): if is_sdxl: raise Exception("light model is not supported for SDXL") pattern = 'sd15.light.v11\.(safetensors|bin)$' - # if light model v11 is not found, try with the old version + # if v11 is not found, try with the old version if not [e for e in ipadapter_list if re.search(pattern, e, re.IGNORECASE)]: pattern = 'sd15.light\.(safetensors|bin)$' elif preset.startswith("standard"): @@ -63,8 +63,12 @@ def get_ipadapter_file(preset, is_sdxl): pattern = 'full.face.sd15\.(safetensors|bin)$' elif preset.startswith("faceid portrait"): if is_sdxl: - raise Exception("portrait model is not supported for SDXL") - pattern = 'portrait.sd15\.(safetensors|bin)$' + pattern = 'portrait.sdxl\.(safetensors|bin)$' + else: + pattern = 'portrait.v11.sd15\.(safetensors|bin)$' + # if v11 is not found, try with the old version + if not [e for e in ipadapter_list if re.search(pattern, e, re.IGNORECASE)]: + pattern = 'portrait.sd15\.(safetensors|bin)$' is_insightface = True elif preset == "faceid": if is_sdxl: @@ -88,6 +92,12 @@ def get_ipadapter_file(preset, is_sdxl): pattern = 'faceid.plusv2.sd15\.(safetensors|bin)$' lora_pattern = 'faceid.plusv2.sd15.lora\.safetensors$' is_insightface = True + # Community's models + elif preset.startswith("composition"): + if is_sdxl: + pattern = 'plus.composition.sdxl\.safetensors$' + else: + pattern = 'plus.composition.sd15\.safetensors$' else: raise Exception(f"invalid type '{preset}'")