First working version

This commit is contained in:
matt3o
2024-02-10 12:39:25 +01:00
parent 1ec20b6b2c
commit 3fa938fa2e
7 changed files with 1829 additions and 89 deletions
+162 -84
View File
@@ -12,6 +12,7 @@ from .resampler import Resampler
from insightface.app import FaceAnalysis
import torchvision.transforms.v2 as T
import torch.nn.functional as F
MODELS_DIR = os.path.join(folder_paths.models_dir, "instantid")
if "instantid" not in folder_paths.folder_names_and_paths:
@@ -50,19 +51,6 @@ def draw_kps(image_pil, kps, color_list=[(255,0,0), (0,255,0), (0,0,255), (255,2
out_img_pil = PIL.Image.fromarray(out_img.astype(np.uint8))
return out_img_pil
def set_model_patch_replace(model, patch_kwargs, key):
to = model.model_options["transformer_options"]
if "patches_replace" not in to:
to["patches_replace"] = {}
if "attn2" not in to["patches_replace"]:
to["patches_replace"]["attn2"] = {}
if key not in to["patches_replace"]["attn2"]:
patch = CrossAttentionPatch(**patch_kwargs)
to["patches_replace"]["attn2"][key] = patch
else:
to["patches_replace"]["attn2"][key].set_new_condition(**patch_kwargs)
class CrossAttentionPatch:
# forward for patching
def __init__(self, weight, instantid, number, cond, uncond, mask=None, sigma_start=0.0, sigma_end=1.0):
@@ -90,7 +78,9 @@ class CrossAttentionPatch:
def __call__(self, n, context_attn2, value_attn2, extra_options):
org_dtype = n.dtype
cond_or_uncond = extra_options["cond_or_uncond"]
sigma = extra_options["sigmas"][0].item() if 'sigmas' in extra_options else 999999999.9
sigma = extra_options["sigmas"][0] if 'sigmas' in extra_options else None
sigma = sigma.item() if sigma else 999999999.9
q = n
k = context_attn2
@@ -102,27 +92,46 @@ class CrossAttentionPatch:
_, _, lh, lw = extra_options["original_shape"]
for weight, cond, uncond, instantid, mask, sigma_start, sigma_end in zip(self.weights, self.conds, self.unconds, self.instantid, self.masks, self.sigma_start, self.sigma_end):
#if sigma > sigma_start or sigma < sigma_end:
# continue
if sigma < sigma_start and sigma > sigma_end:
k_cond = instantid.ip_layers.to_kvs[self.k_key](cond).repeat(batch_prompt, 1, 1)
k_uncond = instantid.ip_layers.to_kvs[self.k_key](uncond).repeat(batch_prompt, 1, 1)
v_cond = instantid.ip_layers.to_kvs[self.v_key](cond).repeat(batch_prompt, 1, 1)
v_uncond = instantid.ip_layers.to_kvs[self.v_key](uncond).repeat(batch_prompt, 1, 1)
k_cond = instantid.ip_layers.to_kvs[self.k_key](cond).repeat(b, 1, 1)
k_uncond = instantid.ip_layers.to_kvs[self.k_key](uncond).repeat(batch_prompt, 1, 1)
v_cond = instantid.ip_layers.to_kvs[self.v_key](cond).repeat(b, 1, 1)
v_uncond = instantid.ip_layers.to_kvs[self.v_key](uncond).repeat(batch_prompt, 1, 1)
iid_k = torch.cat([(k_cond, k_uncond)[i] for i in cond_or_uncond], dim=0)
iid_v = torch.cat([(v_cond, v_uncond)[i] for i in cond_or_uncond], dim=0)
ip_k = torch.cat([(k_cond, k_uncond)[i] for i in cond_or_uncond], dim=0)
ip_v = torch.cat([(v_cond, v_uncond)[i] for i in cond_or_uncond], dim=0)
out_iid = optimized_attention(q, iid_k, iid_v, extra_options["n_heads"])
out_iid = out_iid * weight
out_iid = optimized_attention(q, ip_k, ip_v, extra_options["n_heads"])
out_iid = out_iid * weight
if mask is not None:
# TODO: needs checking
mask_h = lh / math.sqrt(lh * lw / qs)
mask_h = int(mask_h) + int((qs % int(mask_h)) != 0)
mask_w = qs // mask_h
out = out + out_iid
mask_downsample = F.interpolate(mask.unsqueeze(1), size=(mask_h, mask_w), mode="bicubic").squeeze(1)
# if we don't have enough masks repeat the last one until we reach the right size
if mask_downsample.shape[0] < batch_prompt:
mask_downsample = torch.cat((mask_downsample, mask_downsample[-1:, :, :].repeat((batch_prompt-mask_downsample.shape[0], 1, 1))), dim=0)
# if we have too many remove the exceeding
elif mask_downsample.shape[0] > batch_prompt:
mask_downsample = mask_downsample[:batch_prompt, :, :]
# repeat the masks
mask_downsample = mask_downsample.repeat(len(cond_or_uncond), 1, 1)
mask_downsample = mask_downsample.view(mask_downsample.shape[0], -1, 1).repeat(1, 1, out.shape[2])
out_ip = out_ip * mask_downsample
out = out + out_iid
return out.to(dtype=org_dtype)
class InstantID(torch.nn.Module):
def __init__(self, instantid_model, cross_attention_dim=1024, output_cross_attention_dim=1024, clip_embeddings_dim=1024, clip_extra_context_tokens=4):
def __init__(self, instantid_model, cross_attention_dim=1280, output_cross_attention_dim=1024, clip_embeddings_dim=512, clip_extra_context_tokens=16):
super().__init__()
self.clip_embeddings_dim = clip_embeddings_dim
@@ -150,13 +159,10 @@ class InstantID(torch.nn.Module):
@torch.inference_mode()
def get_image_embeds(self, clip_embed, clip_embed_zeroed):
image_prompt_embeds = clip_embed.clone().detach()
image_prompt_embeds = self.image_proj_model(image_prompt_embeds)
#image_prompt_embeds = image_prompt_embeds.reshape([1, -1, 512])
uncond_image_prompt_embeds = clip_embed_zeroed.clone().detach()
uncond_image_prompt_embeds = self.image_proj_model(uncond_image_prompt_embeds)
#uncond_image_prompt_embeds = uncond_image_prompt_embeds.reshape([1, -1, 512])
#image_prompt_embeds = clip_embed.clone().detach()
image_prompt_embeds = self.image_proj_model(clip_embed)
#uncond_image_prompt_embeds = clip_embed_zeroed.clone().detach()
uncond_image_prompt_embeds = self.image_proj_model(clip_embed_zeroed)
return image_prompt_embeds, uncond_image_prompt_embeds
@@ -181,8 +187,9 @@ class To_KV(torch.nn.Module):
self.to_kvs = torch.nn.ModuleDict()
for key, value in state_dict.items():
self.to_kvs[key.replace(".weight", "").replace(".", "_")] = torch.nn.Linear(value.shape[1], value.shape[0], bias=False)
self.to_kvs[key.replace(".weight", "").replace(".", "_")].weight.data = value
k = key.replace(".weight", "").replace(".", "_")
self.to_kvs[k] = torch.nn.Linear(value.shape[1], value.shape[0], bias=False)
self.to_kvs[k].weight.data = value
def set_model_patch_replace(model, patch_kwargs, key):
to = model.model_options["transformer_options"]
@@ -196,7 +203,6 @@ def set_model_patch_replace(model, patch_kwargs, key):
else:
to["patches_replace"]["attn2"][key].set_new_condition(**patch_kwargs)
class InstantIDModelLoader:
@classmethod
def INPUT_TYPES(s):
@@ -222,7 +228,45 @@ class InstantIDModelLoader:
return (model,)
class InsightFaceLoader:
def tensorToNP(image):
out = torch.clamp(255. * image.detach().cpu(), 0, 255).to(torch.uint8)
out = out[..., [2, 1, 0]]
out = out.numpy()
return out
def extractFeatures(insightface, image, extract_kps=False):
face_img = tensorToNP(image)
out = []
insightface.det_model.input_size = (640,640) # reset the detection size
for i in range(face_img.shape[0]):
for size in [(size, size) for size in range(640, 128, -64)]:
insightface.det_model.input_size = size # TODO: hacky but seems to be working
face = insightface.get(face_img[i])
if face:
face = sorted(face, key=lambda x:(x['bbox'][2]-x['bbox'][0])*x['bbox'][3]-x['bbox'][1])[-1]
if extract_kps:
out.append(draw_kps(face_img[i], face['kps']))
else:
out.append(torch.from_numpy(face['embedding']).unsqueeze(0))
if 640 not in size:
print(f"\033[33mINFO: InsightFace detection resolution lowered to {size}.\033[0m")
break
if out:
if extract_kps:
out = torch.stack(T.ToTensor()(out), dim=0).permute([0,2,3,1])
else:
out = torch.stack(out, dim=0)
else:
out = None
return out
class InstantIDFaceAnalysis:
@classmethod
def INPUT_TYPES(s):
return {
@@ -231,7 +275,7 @@ class InsightFaceLoader:
},
}
RETURN_TYPES = ("INSIGHTFACE",)
RETURN_TYPES = ("FACEANALYSIS",)
FUNCTION = "load_insight_face"
CATEGORY = "InstantID"
@@ -241,12 +285,28 @@ class InsightFaceLoader:
return (model,)
def tensorToNP(image):
out = torch.clamp(255. * image.detach().cpu(), 0, 255).to(torch.uint8)
out = out[..., [2, 1, 0]]
out = out.numpy()
class FaceKeypointsPreprocessor:
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"faceanalysis": ("FACEANALYSIS", ),
"image": ("IMAGE", ),
},
}
RETURN_TYPES = ("IMAGE",)
FUNCTION = "preprocess_image"
CATEGORY = "InstantID"
return out
def preprocess_image(self, faceanalysis, image):
face_kps = extractFeatures(faceanalysis, image, extract_kps=True)
if face_kps is None:
face_kps = torch.zeros_like(image)
print(f"\033[33mWARNING: no face detected, unable to extract the keypoints!\033[0m")
#raise Exception('Face Keypoints Image: No face detected.')
return (face_kps,)
class ApplyInstantID:
@classmethod
@@ -254,47 +314,38 @@ class ApplyInstantID:
return {
"required": {
"instantid": ("INSTANTID", ),
"insightface": ("INSIGHTFACE", ),
"insightface": ("FACEANALYSIS", ),
"image_features": ("IMAGE", ),
"model": ("MODEL", ),
"image": ("IMAGE", )
"positive": ("CONDITIONING", ),
"negative": ("CONDITIONING", ),
"weight": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01,}),
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001,}),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001,}),
},
"optional": {
"attn_mask": ("MASK",),
}
}
RETURN_TYPES = ("MODEL", "IMAGE")
RETURN_NAMES = ("MODEL", "IMAGE_KPS")
RETURN_TYPES = ("MODEL", "CONDITIONING", "CONDITIONING",)
RETURN_NAMES = ("MODEL", "POSITIVE", "NEGATIVE", )
FUNCTION = "apply_instantid"
CATEGORY = "InstantID"
def apply_instantid(self, instantid, insightface, model, image):
def apply_instantid(self, instantid, insightface, image_features, model, positive, negative, weight, start_at, end_at, attn_mask=None):
self.dtype = torch.float16 if comfy.model_management.should_use_fp16() else torch.float32
self.device = comfy.model_management.get_torch_device()
self.weight = 1.0
self.weight = weight
output_cross_attention_dim = instantid["ip_adapter"]["1.to_k_ip.weight"].shape[1]
is_sdxl = output_cross_attention_dim == 2048
cross_attention_dim = 1280
clip_extra_context_tokens = 16
insightface.det_model.input_size = (640,640) # reset the detection size
face_img = tensorToNP(image)
face_embed = []
face_kps = []
for i in range(face_img.shape[0]):
for size in [(size, size) for size in range(640, 128, -64)]:
insightface.det_model.input_size = size # TODO: hacky but seems to be working
face = insightface.get(face_img[i])
if face:
face_embed.append(torch.from_numpy(face[0].embedding).unsqueeze(0))
face_kps.append(draw_kps(face_img[i], face[0].kps))
if 640 not in size:
print(f"\033[33mINFO: InsightFace detection resolution lowered to {size}.\033[0m")
break
else:
raise Exception('InsightFace: No face detected.')
face_embed = torch.stack(face_embed, dim=0)
face_kps = torch.stack(T.ToTensor()(face_kps), dim=0).permute([0,2,3,1])
face_embed = extractFeatures(insightface, image_features)
if face_embed is None:
raise Exception('Feature Extractor: No face detected.')
clip_embed = face_embed
clip_embed_zeroed = torch.zeros_like(clip_embed)
@@ -318,38 +369,65 @@ class ApplyInstantID:
work_model = model.clone()
sigma_start = work_model.model.model_sampling.percent_to_sigma(start_at)
sigma_end = work_model.model.model_sampling.percent_to_sigma(end_at)
if attn_mask is not None:
attn_mask = attn_mask.to(self.device)
patch_kwargs = {
"number": 0,
"weight": self.weight,
"instantid": self.instantid,
"cond": image_prompt_embeds,
"uncond": uncond_image_prompt_embeds,
"mask": attn_mask,
"sigma_start": sigma_start,
"sigma_end": sigma_end,
}
for id in [4,5,7,8]: # id of input_blocks that have cross attention
block_indices = range(2) if id in [4, 5] else range(10) # transformer_depth
for index in block_indices:
set_model_patch_replace(work_model, patch_kwargs, ("input", id, index))
if not is_sdxl:
for id in [1,2,4,5,7,8]: # id of input_blocks that have cross attention
set_model_patch_replace(work_model, patch_kwargs, ("input", id))
patch_kwargs["number"] += 1
for id in range(6): # id of output_blocks that have cross attention
block_indices = range(2) if id in [3, 4, 5] else range(10) # transformer_depth
for index in block_indices:
set_model_patch_replace(work_model, patch_kwargs, ("output", id, index))
for id in [3,4,5,6,7,8,9,10,11]: # id of output_blocks that have cross attention
set_model_patch_replace(work_model, patch_kwargs, ("output", id))
patch_kwargs["number"] += 1
set_model_patch_replace(work_model, patch_kwargs, ("middle", 0))
else:
for id in [4,5,7,8]: # id of input_blocks that have cross attention
block_indices = range(2) if id in [4, 5] else range(10) # transformer_depth
for index in block_indices:
set_model_patch_replace(work_model, patch_kwargs, ("input", id, index))
patch_kwargs["number"] += 1
for id in range(6): # id of output_blocks that have cross attention
block_indices = range(2) if id in [3, 4, 5] else range(10) # transformer_depth
for index in block_indices:
set_model_patch_replace(work_model, patch_kwargs, ("output", id, index))
patch_kwargs["number"] += 1
for index in range(10):
set_model_patch_replace(work_model, patch_kwargs, ("middle", 0, index))
patch_kwargs["number"] += 1
for index in range(10):
set_model_patch_replace(work_model, patch_kwargs, ("middle", 0, index))
patch_kwargs["number"] += 1
return(work_model, face_kps, )
pos = positive.copy()
print(pos[0][1].keys())
pos[0][1]['cross_attn_controlnet'] = image_prompt_embeds.cpu()
neg = negative.copy()
neg[0][1]['cross_attn_controlnet'] = uncond_image_prompt_embeds.cpu()
return(work_model, pos, neg, )
NODE_CLASS_MAPPINGS = {
"InstantIDModelLoader": InstantIDModelLoader,
"InsightFaceLoaderIID": InsightFaceLoader,
"InstantIDFaceAnalysis": InstantIDFaceAnalysis,
"ApplyInstantID": ApplyInstantID,
"FaceKeypointsPreprocessor": FaceKeypointsPreprocessor,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"InstantIDModelLoader": "Load InstantID Model",
"InsightFaceLoaderIID": "Load InsightFace IID",
"InstantIDFaceAnalysis": "InstantID Face Analysis",
"ApplyInstantID": "Apply InstantID",
"FaceKeypointsPreprocessor": "Face Keypoints Preprocessor",
}
+38 -5
View File
@@ -1,9 +1,42 @@
## NOT WORKING YET!! do not use
# ComfyUI InstantID (Native Support)
Initial work to support [InstandID](https://github.com/InstantID/InstantID) natively in ComfyUI.
Native [InstantID](https://github.com/InstantID/InstantID) support for [ComfyUI](https://github.com/comfyanonymous/ComfyUI).
This is mostly a placeholder, more work is needed... if I get the time.
This extension differs from the many already available as it doesn't use *diffusers* but instead implements InstantID natively and it fully integrates with ComfyUI.
Model go in ComfyUI/models/instantid, you need "antelopev2" models for insightface.
Please note this still could be considered beta stage, looking forward to your feedback.
This repo is temporary and might be removed.
## Basic Workflow
In the `examples` directory you'll find some basic workflows.
![workflow](examples/instantID_workflow_posed.jpg)
## Installation
**Upgrade ComfyUI to the latest version!** ComfyUI required a small update to work with InstantID that was pushed recently.
Download or `git clone` this repository into the `ComfyUI/custom_nodes/` directory. I guess the Manager will soon have this added to the list.
InstantID requires `insightface`, you need to add it to your libraries together with `onnxruntine` and `onnxruntime-gpu`.
The **main model** can be downloaded from [HuggingFace](https://huggingface.co/InstantX/InstantID/resolve/main/ip-adapter.bin?download=true) and should be placed into the `ComfyUI/models/instantid` directory. (Note that the model is called *ip_adapter* as it is based on the [IPAdapter](https://github.com/tencent-ailab/IP-Adapter) models).
You also needs a [controlnet](https://huggingface.co/InstantX/InstantID/resolve/main/ControlNetModel/diffusion_pytorch_model.safetensors?download=true), place it in the ComfyUI controlnet directory.
**Remember at the moment this is only for SDXL.**
## Watermarks!
The training data is full of watermarks, to avoid them to show up in your generations use a resolution slightly different from 1024×1024 for example **1016×1016** works pretty well.
## Lower the CFG!
It's important to lower the CFG to at least 4/5 or you can use the `RescaleCFG` node.
## Other notes
It works very well with SDXL Turbo. Best results with community's checkpoints.
<div style="text-align:center">
<img src="examples/daydreaming.jpg" width="386" height="386" alt="Day Dreaming" />
</div>
+794
View File
@@ -0,0 +1,794 @@
{
"last_node_id": 59,
"last_link_id": 197,
"nodes": [
{
"id": 8,
"type": "VAEDecode",
"pos": [
2200,
410
],
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 7
},
{
"name": "vae",
"type": "VAE",
"link": 8
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
19
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 5,
"type": "EmptyLatentImage",
"pos": [
1410,
610
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
2
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EmptyLatentImage"
},
"widgets_values": [
1016,
1016,
1
]
},
{
"id": 23,
"type": "ControlNetApplyAdvanced",
"pos": [
1410,
380
],
"size": {
"0": 315,
"1": 166
},
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "positive",
"type": "CONDITIONING",
"link": 184
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 185
},
{
"name": "control_net",
"type": "CONTROL_NET",
"link": 53
},
{
"name": "image",
"type": "IMAGE",
"link": 192
}
],
"outputs": [
{
"name": "positive",
"type": "CONDITIONING",
"links": [
95
],
"shape": 3,
"slot_index": 0
},
{
"name": "negative",
"type": "CONDITIONING",
"links": [
102
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "ControlNetApplyAdvanced"
},
"widgets_values": [
0.65,
0,
1
]
},
{
"id": 16,
"type": "ControlNetLoader",
"pos": [
1050,
540
],
"size": {
"0": 250.07241821289062,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "CONTROL_NET",
"type": "CONTROL_NET",
"links": [
53
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "ControlNetLoader"
},
"widgets_values": [
"instantid/diffusion_pytorch_model.safetensors"
]
},
{
"id": 4,
"type": "CheckpointLoaderSimple",
"pos": [
80,
670
],
"size": {
"0": 315,
"1": 98
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
182
],
"slot_index": 0
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
122,
123
],
"slot_index": 1
},
{
"name": "VAE",
"type": "VAE",
"links": [
8
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "CheckpointLoaderSimple"
},
"widgets_values": [
"sdxl/TurboVisionXL.safetensors"
]
},
{
"id": 58,
"type": "FaceKeypointsPreprocessor",
"pos": [
1060,
640
],
"size": {
"0": 229.20001220703125,
"1": 46
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "faceanalysis",
"type": "FACEANALYSIS",
"link": 193
},
{
"name": "image",
"type": "IMAGE",
"link": 197
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
192
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "FaceKeypointsPreprocessor"
}
},
{
"id": 11,
"type": "InstantIDModelLoader",
"pos": [
690,
90
],
"size": {
"0": 238.72393798828125,
"1": 58
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "INSTANTID",
"type": "INSTANTID",
"links": [
189
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDModelLoader"
},
"widgets_values": [
"ip-adapter.bin"
]
},
{
"id": 38,
"type": "InstantIDFaceAnalysis",
"pos": [
700,
200
],
"size": {
"0": 227.09793090820312,
"1": 58
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "FACEANALYSIS",
"type": "FACEANALYSIS",
"links": [
190,
193
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDFaceAnalysis"
},
"widgets_values": [
"CPU"
]
},
{
"id": 40,
"type": "CLIPTextEncode",
"pos": [
560,
830
],
"size": {
"0": 286.3603515625,
"1": 112.35245513916016
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 123
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
187
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"low quality, blurry, malformed, distorted"
]
},
{
"id": 15,
"type": "PreviewImage",
"pos": [
2200,
510
],
"size": {
"0": 584.0855712890625,
"1": 610.4592895507812
},
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 19
}
],
"properties": {
"Node name for S&R": "PreviewImage"
}
},
{
"id": 13,
"type": "LoadImage",
"pos": [
410,
270
],
"size": {
"0": 210,
"1": 290
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
188,
197
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"face4.jpg",
"image"
]
},
{
"id": 57,
"type": "ApplyInstantID",
"pos": [
1040,
260
],
"size": [
260,
226
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "instantid",
"type": "INSTANTID",
"link": 189
},
{
"name": "insightface",
"type": "FACEANALYSIS",
"link": 190
},
{
"name": "image_features",
"type": "IMAGE",
"link": 188
},
{
"name": "model",
"type": "MODEL",
"link": 182
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 186
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 187
},
{
"name": "attn_mask",
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
196
],
"shape": 3,
"slot_index": 0
},
{
"name": "POSITIVE",
"type": "CONDITIONING",
"links": [
184
],
"shape": 3,
"slot_index": 1
},
{
"name": "NEGATIVE",
"type": "CONDITIONING",
"links": [
185
],
"shape": 3,
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "ApplyInstantID"
},
"widgets_values": [
0.8,
0,
1
]
},
{
"id": 39,
"type": "CLIPTextEncode",
"pos": [
560,
650
],
"size": {
"0": 291.9967346191406,
"1": 128.62518310546875
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 122
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
186
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"watercolors portrait of a woman (happy laughing:1.15), masterpiece, artistry"
]
},
{
"id": 3,
"type": "KSampler",
"pos": [
1810,
290
],
"size": {
"0": 315,
"1": 262
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 196
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 95
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 102
},
{
"name": "latent_image",
"type": "LATENT",
"link": 2
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
7
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "KSampler"
},
"widgets_values": [
1631588094,
"fixed",
7,
2.6,
"ddpm",
"normal",
1
]
}
],
"links": [
[
2,
5,
0,
3,
3,
"LATENT"
],
[
7,
3,
0,
8,
0,
"LATENT"
],
[
8,
4,
2,
8,
1,
"VAE"
],
[
19,
8,
0,
15,
0,
"IMAGE"
],
[
53,
16,
0,
23,
2,
"CONTROL_NET"
],
[
95,
23,
0,
3,
1,
"CONDITIONING"
],
[
102,
23,
1,
3,
2,
"CONDITIONING"
],
[
122,
4,
1,
39,
0,
"CLIP"
],
[
123,
4,
1,
40,
0,
"CLIP"
],
[
182,
4,
0,
57,
3,
"MODEL"
],
[
184,
57,
1,
23,
0,
"CONDITIONING"
],
[
185,
57,
2,
23,
1,
"CONDITIONING"
],
[
186,
39,
0,
57,
4,
"CONDITIONING"
],
[
187,
40,
0,
57,
5,
"CONDITIONING"
],
[
188,
13,
0,
57,
2,
"IMAGE"
],
[
189,
11,
0,
57,
0,
"INSTANTID"
],
[
190,
38,
0,
57,
1,
"FACEANALYSIS"
],
[
192,
58,
0,
23,
3,
"IMAGE"
],
[
193,
38,
0,
58,
0,
"FACEANALYSIS"
],
[
196,
57,
0,
3,
0,
"MODEL"
],
[
197,
13,
0,
58,
1,
"IMAGE"
]
],
"groups": [],
"config": {},
"extra": {},
"version": 0.4
}
+832
View File
@@ -0,0 +1,832 @@
{
"last_node_id": 59,
"last_link_id": 196,
"nodes": [
{
"id": 8,
"type": "VAEDecode",
"pos": [
2200,
410
],
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 7
},
{
"name": "vae",
"type": "VAE",
"link": 8
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
19
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 5,
"type": "EmptyLatentImage",
"pos": [
1410,
610
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
2
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EmptyLatentImage"
},
"widgets_values": [
1016,
1016,
1
]
},
{
"id": 23,
"type": "ControlNetApplyAdvanced",
"pos": [
1410,
380
],
"size": {
"0": 315,
"1": 166
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "positive",
"type": "CONDITIONING",
"link": 184
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 185
},
{
"name": "control_net",
"type": "CONTROL_NET",
"link": 53
},
{
"name": "image",
"type": "IMAGE",
"link": 192
}
],
"outputs": [
{
"name": "positive",
"type": "CONDITIONING",
"links": [
95
],
"shape": 3,
"slot_index": 0
},
{
"name": "negative",
"type": "CONDITIONING",
"links": [
102
],
"shape": 3,
"slot_index": 1
}
],
"properties": {
"Node name for S&R": "ControlNetApplyAdvanced"
},
"widgets_values": [
0.65,
0,
1
]
},
{
"id": 16,
"type": "ControlNetLoader",
"pos": [
1050,
540
],
"size": {
"0": 250.07241821289062,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "CONTROL_NET",
"type": "CONTROL_NET",
"links": [
53
],
"shape": 3
}
],
"properties": {
"Node name for S&R": "ControlNetLoader"
},
"widgets_values": [
"instantid/diffusion_pytorch_model.safetensors"
]
},
{
"id": 4,
"type": "CheckpointLoaderSimple",
"pos": [
80,
670
],
"size": {
"0": 315,
"1": 98
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
182
],
"slot_index": 0
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
122,
123
],
"slot_index": 1
},
{
"name": "VAE",
"type": "VAE",
"links": [
8
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "CheckpointLoaderSimple"
},
"widgets_values": [
"sdxl/TurboVisionXL.safetensors"
]
},
{
"id": 58,
"type": "FaceKeypointsPreprocessor",
"pos": [
1060,
640
],
"size": {
"0": 229.20001220703125,
"1": 46
},
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "faceanalysis",
"type": "FACEANALYSIS",
"link": 193
},
{
"name": "image",
"type": "IMAGE",
"link": 191
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
192
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "FaceKeypointsPreprocessor"
}
},
{
"id": 39,
"type": "CLIPTextEncode",
"pos": [
560,
650
],
"size": {
"0": 291.9967346191406,
"1": 128.62518310546875
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 122
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
186
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"oil paint portrait of a woman (closed eyes day dreaming:1.15)"
]
},
{
"id": 11,
"type": "InstantIDModelLoader",
"pos": [
690,
90
],
"size": {
"0": 238.72393798828125,
"1": 58
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "INSTANTID",
"type": "INSTANTID",
"links": [
189
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDModelLoader"
},
"widgets_values": [
"ip-adapter.bin"
]
},
{
"id": 38,
"type": "InstantIDFaceAnalysis",
"pos": [
700,
200
],
"size": {
"0": 227.09793090820312,
"1": 58
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "FACEANALYSIS",
"type": "FACEANALYSIS",
"links": [
190,
193
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDFaceAnalysis"
},
"widgets_values": [
"CPU"
]
},
{
"id": 13,
"type": "LoadImage",
"pos": [
410,
270
],
"size": {
"0": 210,
"1": 290
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
188
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"face4.jpg",
"image"
]
},
{
"id": 40,
"type": "CLIPTextEncode",
"pos": [
560,
830
],
"size": {
"0": 286.3603515625,
"1": 112.35245513916016
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 123
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
187
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"low quality, blurry, malformed, distorted"
]
},
{
"id": 3,
"type": "KSampler",
"pos": [
1810,
290
],
"size": {
"0": 315,
"1": 262
},
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 196
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 95
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 102
},
{
"name": "latent_image",
"type": "LATENT",
"link": 2
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
7
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "KSampler"
},
"widgets_values": [
1631588085,
"fixed",
7,
2.6,
"ddpm",
"normal",
1
]
},
{
"id": 15,
"type": "PreviewImage",
"pos": [
2200,
510
],
"size": {
"0": 584.0855712890625,
"1": 610.4592895507812
},
"flags": {},
"order": 14,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 19
}
],
"properties": {
"Node name for S&R": "PreviewImage"
}
},
{
"id": 57,
"type": "ApplyInstantID",
"pos": [
1040,
260
],
"size": [
260,
226
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "instantid",
"type": "INSTANTID",
"link": 189
},
{
"name": "insightface",
"type": "FACEANALYSIS",
"link": 190
},
{
"name": "image_features",
"type": "IMAGE",
"link": 188
},
{
"name": "model",
"type": "MODEL",
"link": 182
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 186
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 187
},
{
"name": "attn_mask",
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
196
],
"shape": 3,
"slot_index": 0
},
{
"name": "POSITIVE",
"type": "CONDITIONING",
"links": [
184
],
"shape": 3,
"slot_index": 1
},
{
"name": "NEGATIVE",
"type": "CONDITIONING",
"links": [
185
],
"shape": 3,
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "ApplyInstantID"
},
"widgets_values": [
1,
0,
1
]
},
{
"id": 42,
"type": "LoadImage",
"pos": [
1060,
750
],
"size": [
220.82643554687525,
327.5079427734379
],
"flags": {},
"order": 6,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
191
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"666561.jpg",
"image"
]
}
],
"links": [
[
2,
5,
0,
3,
3,
"LATENT"
],
[
7,
3,
0,
8,
0,
"LATENT"
],
[
8,
4,
2,
8,
1,
"VAE"
],
[
19,
8,
0,
15,
0,
"IMAGE"
],
[
53,
16,
0,
23,
2,
"CONTROL_NET"
],
[
95,
23,
0,
3,
1,
"CONDITIONING"
],
[
102,
23,
1,
3,
2,
"CONDITIONING"
],
[
122,
4,
1,
39,
0,
"CLIP"
],
[
123,
4,
1,
40,
0,
"CLIP"
],
[
182,
4,
0,
57,
3,
"MODEL"
],
[
184,
57,
1,
23,
0,
"CONDITIONING"
],
[
185,
57,
2,
23,
1,
"CONDITIONING"
],
[
186,
39,
0,
57,
4,
"CONDITIONING"
],
[
187,
40,
0,
57,
5,
"CONDITIONING"
],
[
188,
13,
0,
57,
2,
"IMAGE"
],
[
189,
11,
0,
57,
0,
"INSTANTID"
],
[
190,
38,
0,
57,
1,
"FACEANALYSIS"
],
[
191,
42,
0,
58,
1,
"IMAGE"
],
[
192,
58,
0,
23,
3,
"IMAGE"
],
[
193,
38,
0,
58,
0,
"FACEANALYSIS"
],
[
196,
57,
0,
3,
0,
"MODEL"
]
],
"groups": [],
"config": {},
"extra": {},
"version": 0.4
}
Binary file not shown.

After

Width:  |  Height:  |  Size: 129 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 249 KiB

+3
View File
@@ -0,0 +1,3 @@
insightface
onnxruntime
onnxruntime-gpu