simplified nodes

This commit is contained in:
matt3o
2024-02-20 10:32:35 +01:00
parent 40b86b4497
commit 54729bf24c
11 changed files with 3152 additions and 3416 deletions
+84 -28
View File
@@ -392,16 +392,18 @@ class ApplyInstantID:
"required": {
"instantid": ("INSTANTID", ),
"insightface": ("FACEANALYSIS", ),
"image_features": ("IMAGE", ),
"control_net": ("CONTROL_NET", ),
"image": ("IMAGE", ),
"model": ("MODEL", ),
"positive": ("CONDITIONING", ),
"negative": ("CONDITIONING", ),
"weight": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.01,}),
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001,}),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001,}),
"weight": ("FLOAT", {"default": .8, "min": 0.0, "max": 5.0, "step": 0.01, }),
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001, }),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001, }),
},
"optional": {
"attn_mask": ("MASK",),
"image_kps": ("IMAGE",),
"mask": ("MASK",),
}
}
@@ -410,22 +412,30 @@ class ApplyInstantID:
FUNCTION = "apply_instantid"
CATEGORY = "InstantID"
def apply_instantid(self, instantid, insightface, image_features, model, positive, negative, weight, start_at, end_at, attn_mask=None):
def apply_instantid(self, instantid, insightface, control_net, image, model, positive, negative, start_at, end_at, weight=.8, ip_weight=None, cn_strength=None, image_kps=None, mask=None):
self.dtype = torch.float16 if comfy.model_management.should_use_fp16() else torch.float32
self.device = comfy.model_management.get_torch_device()
self.weight = weight
ip_weight = weight if ip_weight is None else ip_weight
cn_strength = weight if cn_strength is None else cn_strength
output_cross_attention_dim = instantid["ip_adapter"]["1.to_k_ip.weight"].shape[1]
is_sdxl = output_cross_attention_dim == 2048
cross_attention_dim = 1280
clip_extra_context_tokens = 16
face_embed = extractFeatures(insightface, image_features)
face_embed = extractFeatures(insightface, image)
if face_embed is None:
raise Exception('Feature Extractor: No face detected.')
raise Exception('Reference Image: No face detected.')
face_kps = extractFeatures(insightface, image_kps if image_kps is not None else image, extract_kps=True)
if face_kps is None:
face_kps = torch.zeros_like(image) if image_kps is None else image_kps
print(f"\033[33mWARNING: No face detected in the keypoints image!\033[0m")
clip_embed = face_embed
# InstantID works better with averaged embeds (TODO:needs testing)
# InstantID works better with averaged embeds (TODO: needs testing)
if clip_embed.shape[0] > 1:
clip_embed = torch.mean(clip_embed, dim=0).unsqueeze(0)
@@ -433,6 +443,7 @@ class ApplyInstantID:
clip_embeddings_dim = face_embed.shape[-1]
# 1: patch the attention
self.instantid = InstantID(
instantid,
cross_attention_dim=cross_attention_dim,
@@ -453,16 +464,16 @@ class ApplyInstantID:
sigma_start = work_model.model.model_sampling.percent_to_sigma(start_at)
sigma_end = work_model.model.model_sampling.percent_to_sigma(end_at)
if attn_mask is not None:
attn_mask = attn_mask.to(self.device)
if mask is not None:
mask = mask.to(self.device)
patch_kwargs = {
"number": 0,
"weight": self.weight,
"weight": ip_weight,
"ipadapter": self.instantid,
"cond": image_prompt_embeds,
"uncond": uncond_image_prompt_embeds,
"mask": attn_mask,
"mask": mask,
"sigma_start": sigma_start,
"sigma_end": sigma_end,
"weight_type": "original",
@@ -491,26 +502,70 @@ class ApplyInstantID:
_set_model_patch_replace(work_model, patch_kwargs, ("middle", 0, index))
patch_kwargs["number"] += 1
pos = []
for t in positive:
n = [t[0], t[1].copy()]
n[1]['cross_attn_controlnet'] = image_prompt_embeds.to(comfy.model_management.intermediate_device())
pos.append(n)
#pos[0][1]['cross_attn_controlnet'] = image_prompt_embeds.cpu()
neg = []
for t in negative:
n = [t[0], t[1].copy()]
n[1]['cross_attn_controlnet'] = uncond_image_prompt_embeds.to(comfy.model_management.intermediate_device())
neg.append(n)
#neg[0][1]['cross_attn_controlnet'] = uncond_image_prompt_embeds.cpu()
# 2: do the ControlNet
if mask is not None and len(mask.shape) < 3:
mask = mask.unsqueeze(0)
return(work_model, pos, neg, )
cnets = {}
cond_uncond = []
for conditioning in [positive, negative]:
c = []
is_cond = True
for t in conditioning:
d = t[1].copy()
prev_cnet = d.get('control', None)
if prev_cnet in cnets:
c_net = cnets[prev_cnet]
else:
c_net = control_net.copy().set_cond_hint(face_kps.movedim(-1,1), cn_strength, (start_at, end_at))
c_net.set_previous_controlnet(prev_cnet)
cnets[prev_cnet] = c_net
d['control'] = c_net
d['control_apply_to_uncond'] = False
d['cross_attn_controlnet'] = image_prompt_embeds.to(comfy.model_management.intermediate_device()) if is_cond else uncond_image_prompt_embeds.to(comfy.model_management.intermediate_device())
if mask is not None and is_cond:
d['mask'] = mask
d['set_area_to_bounds'] = False
n = [t[0], d]
c.append(n)
is_cond = True
cond_uncond.append(c)
return(work_model, cond_uncond[0], cond_uncond[1], )
class ApplyInstantIDAdvanced(ApplyInstantID):
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"instantid": ("INSTANTID", ),
"insightface": ("FACEANALYSIS", ),
"control_net": ("CONTROL_NET", ),
"image": ("IMAGE", ),
"model": ("MODEL", ),
"positive": ("CONDITIONING", ),
"negative": ("CONDITIONING", ),
"ip_weight": ("FLOAT", {"default": .8, "min": 0.0, "max": 3.0, "step": 0.01, }),
"cn_strength": ("FLOAT", {"default": .8, "min": 0.0, "max": 10.0, "step": 0.01, }),
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001, }),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001, }),
},
"optional": {
"image_kps": ("IMAGE",),
"mask": ("MASK",),
}
}
NODE_CLASS_MAPPINGS = {
"InstantIDModelLoader": InstantIDModelLoader,
"InstantIDFaceAnalysis": InstantIDFaceAnalysis,
"ApplyInstantID": ApplyInstantID,
"ApplyInstantIDAdvanced": ApplyInstantIDAdvanced,
"FaceKeypointsPreprocessor": FaceKeypointsPreprocessor,
}
@@ -518,5 +573,6 @@ NODE_DISPLAY_NAME_MAPPINGS = {
"InstantIDModelLoader": "Load InstantID Model",
"InstantIDFaceAnalysis": "InstantID Face Analysis",
"ApplyInstantID": "Apply InstantID",
"ApplyInstantIDAdvanced": "Apply InstantID Advanced",
"FaceKeypointsPreprocessor": "Face Keypoints Preprocessor",
}
+37 -9
View File
@@ -4,25 +4,27 @@ Native [InstantID](https://github.com/InstantID/InstantID) support for [ComfyUI]
This extension differs from the many already available as it doesn't use *diffusers* but instead implements InstantID natively and it fully integrates with ComfyUI.
Please note this still could be considered beta stage, looking forward to your feedback.
## Important updates
- **2024/02/20:** I refactored the nodes so they are hopefully easier to use. **This is a breaking update**, the previous workflows won't work anymore.
## Basic Workflow
In the `examples` directory you'll find some basic workflows.
![workflow](examples/instantID_workflow_posed.jpg)
![workflow](examples/instantid_basic_workflow.jpg)
## Installation
**Upgrade ComfyUI to the latest version!** ComfyUI required a small update to work with InstantID that was pushed recently.
**Upgrade ComfyUI to the latest version!**
Download or `git clone` this repository into the `ComfyUI/custom_nodes/` directory. I guess the Manager will soon have this added to the list.
InstantID requires `insightface`, you need to add it to your libraries together with `onnxruntime` and `onnxruntime-gpu`.
The InsightFace model is **antelopev2** (not the classic buffalo_l). Download the models (for example from [here](https://drive.google.com/file/d/18wEUfMNohBJ4K3Ly5wpTejPfDzp-8fI8/view?usp=sharing) or [here](https://huggingface.co/MonsterMMORPG/tools/tree/main)) and place them in the `ComfyUI/models/insightface/models/antelopev2` directory.
The InsightFace model is **antelopev2** (not the classic buffalo_l). Download the models (for example from [here](https://drive.google.com/file/d/18wEUfMNohBJ4K3Ly5wpTejPfDzp-8fI8/view?usp=sharing) or [here](https://huggingface.co/MonsterMMORPG/tools/tree/main)), unzip and place them in the `ComfyUI/models/insightface/models/antelopev2` directory.
The **main model** can be downloaded from [HuggingFace](https://huggingface.co/InstantX/InstantID/resolve/main/ip-adapter.bin?download=true) and should be placed into the `ComfyUI/models/instantid` directory. (Note that the model is called *ip_adapter* as it is based on the [IPAdapter](https://github.com/tencent-ailab/IP-Adapter) models).
The **main model** can be downloaded from [HuggingFace](https://huggingface.co/InstantX/InstantID/resolve/main/ip-adapter.bin?download=true) and should be placed into the `ComfyUI/models/instantid` directory. (Note that the model is called *ip_adapter* as it is based on the [IPAdapter](https://github.com/tencent-ailab/IP-Adapter)).
You also needs a [controlnet](https://huggingface.co/InstantX/InstantID/resolve/main/ControlNetModel/diffusion_pytorch_model.safetensors?download=true), place it in the ComfyUI controlnet directory.
@@ -30,15 +32,41 @@ You also needs a [controlnet](https://huggingface.co/InstantX/InstantID/resolve/
## Watermarks!
The training data is full of watermarks, to avoid them to show up in your generations use a resolution slightly different from 1024×1024 for example **1016×1016** works pretty well.
The training data is full of watermarks, to avoid them to show up in your generations use a resolution slightly different from 1024×1024 (or the standard ones) for example **1016×1016** works pretty well.
## Lower the CFG!
It's important to lower the CFG to at least 4/5 or you can use the `RescaleCFG` node.
## Face keypoints
The person is posed based on the keypoints generated from the reference image. You can use a different pose by sending an image to the `image_kps` input.
<img src="examples/daydreaming.jpg" width="386" height="386" alt="Day Dreaming" />
## Additional Controlnets
You can add more controlnets to the generation. An example workflow for depth controlnet is provided.
## Styling with IPAdapter
It's possible to style the composition with IPAdapter. An example is provided.
<img src="examples/instant_id_ipadapter.jpg" width="512" alt="IPAdapter" />
## Multi-ID
Multi-ID is supported but the workflow is a bit complicated and the generation slower. I'll check if I can find a better way of doing it. The "hackish" workflow is provided in the example directory.
<img src="examples/instantid_multi_id.jpg" width="768" alt="IPAdapter" />
## Advanced Node
There's an InstantID advanced node available, at the moment the only difference with the standard one is that you can set the weights for the instantID models and the controlnet separately. It might be helpful for finetuning.
The instantID model influences the composition of about 25%, the rest is the controlnet.
## Other notes
It works very well with SDXL Turbo. Best results with community's checkpoints.
<div style="text-align:center">
<img src="examples/daydreaming.jpg" width="386" height="386" alt="Day Dreaming" />
</div>
File diff suppressed because it is too large Load Diff
+657
View File
@@ -0,0 +1,657 @@
{
"last_node_id": 66,
"last_link_id": 220,
"nodes": [
{
"id": 11,
"type": "InstantIDModelLoader",
"pos": [
560,
70
],
"size": {
"0": 238.72393798828125,
"1": 58
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "INSTANTID",
"type": "INSTANTID",
"links": [
197
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDModelLoader"
},
"widgets_values": [
"ip-adapter.bin"
]
},
{
"id": 38,
"type": "InstantIDFaceAnalysis",
"pos": [
570,
180
],
"size": {
"0": 227.09793090820312,
"1": 58
},
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "FACEANALYSIS",
"type": "FACEANALYSIS",
"links": [
198
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InstantIDFaceAnalysis"
},
"widgets_values": [
"CPU"
]
},
{
"id": 16,
"type": "ControlNetLoader",
"pos": [
560,
290
],
"size": {
"0": 250.07241821289062,
"1": 58
},
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "CONTROL_NET",
"type": "CONTROL_NET",
"links": [
199
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ControlNetLoader"
},
"widgets_values": [
"instantid/diffusion_pytorch_model.safetensors"
]
},
{
"id": 15,
"type": "PreviewImage",
"pos": [
1670,
300
],
"size": {
"0": 584.0855712890625,
"1": 610.4592895507812
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 19
}
],
"properties": {
"Node name for S&R": "PreviewImage"
}
},
{
"id": 5,
"type": "EmptyLatentImage",
"pos": [
910,
540
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
2
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EmptyLatentImage"
},
"widgets_values": [
1016,
1016,
1
]
},
{
"id": 8,
"type": "VAEDecode",
"pos": [
1670,
210
],
"size": {
"0": 210,
"1": 46
},
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 7
},
{
"name": "vae",
"type": "VAE",
"link": 8
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
19
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 60,
"type": "ApplyInstantID",
"pos": [
910,
210
],
"size": {
"0": 315,
"1": 266
},
"flags": {},
"order": 8,
"mode": 0,
"inputs": [
{
"name": "instantid",
"type": "INSTANTID",
"link": 197
},
{
"name": "insightface",
"type": "FACEANALYSIS",
"link": 198
},
{
"name": "control_net",
"type": "CONTROL_NET",
"link": 199
},
{
"name": "image",
"type": "IMAGE",
"link": 214
},
{
"name": "model",
"type": "MODEL",
"link": 206
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 203
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 204
},
{
"name": "image_kps",
"type": "IMAGE",
"link": null
},
{
"name": "mask",
"type": "MASK",
"link": null
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
220
],
"shape": 3,
"slot_index": 0
},
{
"name": "POSITIVE",
"type": "CONDITIONING",
"links": [
200
],
"shape": 3,
"slot_index": 1
},
{
"name": "NEGATIVE",
"type": "CONDITIONING",
"links": [
201
],
"shape": 3,
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "ApplyInstantID"
},
"widgets_values": [
0.8,
0,
1
]
},
{
"id": 39,
"type": "CLIPTextEncode",
"pos": [
520,
430
],
"size": {
"0": 291.9967346191406,
"1": 128.62518310546875
},
"flags": {},
"order": 6,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 122
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
203
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"comic character. graphic illustration, comic art, graphic novel art, vibrant, highly detailed"
]
},
{
"id": 40,
"type": "CLIPTextEncode",
"pos": [
520,
620
],
"size": {
"0": 286.3603515625,
"1": 112.35245513916016
},
"flags": {},
"order": 7,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 123
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
204
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"photograph, deformed, glitch, noisy, realistic, stock photo"
]
},
{
"id": 4,
"type": "CheckpointLoaderSimple",
"pos": [
70,
520
],
"size": {
"0": 315,
"1": 98
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
206
],
"slot_index": 0
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
122,
123
],
"slot_index": 1
},
{
"name": "VAE",
"type": "VAE",
"links": [
8
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "CheckpointLoaderSimple"
},
"widgets_values": [
"sdxl/AlbedoBaseXL.safetensors"
]
},
{
"id": 3,
"type": "KSampler",
"pos": [
1300,
210
],
"size": {
"0": 315,
"1": 262
},
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 220
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 200
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 201
},
{
"name": "latent_image",
"type": "LATENT",
"link": 2
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
7
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "KSampler"
},
"widgets_values": [
1631591050,
"fixed",
30,
4.5,
"ddpm",
"karras",
1
]
},
{
"id": 13,
"type": "LoadImage",
"pos": [
290,
70
],
"size": {
"0": 210,
"1": 314
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
214
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"joseph-gonzalez-iFgRcqHznqg-unsplash.jpg",
"image"
]
}
],
"links": [
[
2,
5,
0,
3,
3,
"LATENT"
],
[
7,
3,
0,
8,
0,
"LATENT"
],
[
8,
4,
2,
8,
1,
"VAE"
],
[
19,
8,
0,
15,
0,
"IMAGE"
],
[
122,
4,
1,
39,
0,
"CLIP"
],
[
123,
4,
1,
40,
0,
"CLIP"
],
[
197,
11,
0,
60,
0,
"INSTANTID"
],
[
198,
38,
0,
60,
1,
"FACEANALYSIS"
],
[
199,
16,
0,
60,
2,
"CONTROL_NET"
],
[
200,
60,
1,
3,
1,
"CONDITIONING"
],
[
201,
60,
2,
3,
2,
"CONDITIONING"
],
[
203,
39,
0,
60,
5,
"CONDITIONING"
],
[
204,
40,
0,
60,
6,
"CONDITIONING"
],
[
206,
4,
0,
60,
4,
"MODEL"
],
[
214,
13,
0,
60,
3,
"IMAGE"
],
[
220,
60,
0,
3,
0,
"MODEL"
]
],
"groups": [],
"config": {},
"extra": {},
"version": 0.4
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 249 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 478 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 233 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 553 KiB