added advanced node

This commit is contained in:
matt3o
2024-05-13 10:17:32 +02:00
parent b88e5aae10
commit f968ca769f
4 changed files with 1052 additions and 21 deletions
+9 -1
View File
@@ -4,6 +4,10 @@
![basic workflow](examples/pulid_wf.jpg)
## Important updates
- **2024.05.12:** Added the Advanced node, allows fine tuning of the generation.
## Notes
The code can be considered beta, things may change in the coming days. In the `examples` directory you'll find some basic workflows.
@@ -18,7 +22,11 @@ Testing other models though I noticed some quality degradation. You may need to
## The 'method' parameter
`method` applies the weights in different ways. `Fidelity` is closer to the reference ID, `Style` leaves more freedom to the checkpoint. Sometimes the difference is minimal. I've added `neutral` that doesn't do any normalization so the reference is very strong and you need to lower the weight.
`method` applies the weights in different ways. `Fidelity` is closer to the reference ID, `Style` leaves more freedom to the checkpoint. Sometimes the difference is minimal. I've added `neutral` that doesn't do any normalization, if you use this option with the standard Apply node be sure to lower the weight. With the Advanced node you can simply increase the `fidelity` value.
The Advanced node has a `fidelity` slider and a `projection` option. `ortho_v2` with `fidelity: 8` is the same as `fidelity` method in the standard node. Projection `ortho` and `fidelity: 16` is the same as method `style`.
**Lower `fidelity` values grant higher resemblance to the reference image.**
## Installation
+946
View File
@@ -0,0 +1,946 @@
{
"last_node_id": 88,
"last_link_id": 248,
"nodes": [
{
"id": 5,
"type": "EmptyLatentImage",
"pos": [
350,
265
],
"size": {
"0": 315,
"1": 106
},
"flags": {},
"order": 0,
"mode": 0,
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
2
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "EmptyLatentImage"
},
"widgets_values": [
1280,
960,
1
]
},
{
"id": 33,
"type": "ApplyPulid",
"pos": [
350,
-10
],
"size": {
"0": 315,
"1": 230
},
"flags": {},
"order": 13,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 133
},
{
"name": "pulid",
"type": "PULID",
"link": 117
},
{
"name": "eva_clip",
"type": "EVA_CLIP",
"link": 81
},
{
"name": "face_analysis",
"type": "FACEANALYSIS",
"link": 82
},
{
"name": "image",
"type": "IMAGE",
"link": 114
},
{
"name": "attn_mask",
"type": "MASK",
"link": 247
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
141
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ApplyPulid"
},
"widgets_values": [
"fidelity",
0.7000000000000001,
0,
1
]
},
{
"id": 85,
"type": "SolidMask",
"pos": [
-307,
584
],
"size": [
210,
106
],
"flags": {},
"order": 1,
"mode": 0,
"outputs": [
{
"name": "MASK",
"type": "MASK",
"links": [
244
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "SolidMask"
},
"widgets_values": [
0,
1280,
960
]
},
{
"id": 49,
"type": "LoadImage",
"pos": [
407,
550
],
"size": [
248.03589794921936,
339.7795556640626
],
"flags": {},
"order": 2,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
145
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"venere.jpg",
"image"
]
},
{
"id": 48,
"type": "InvertMask",
"pos": [
526,
438
],
"size": [
140,
26
],
"flags": {},
"order": 12,
"mode": 0,
"inputs": [
{
"name": "mask",
"type": "MASK",
"link": 246
}
],
"outputs": [
{
"name": "MASK",
"type": "MASK",
"links": [
151
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "InvertMask"
}
},
{
"id": 8,
"type": "VAEDecode",
"pos": [
1575,
160
],
"size": {
"0": 140,
"1": 46
},
"flags": {},
"order": 16,
"mode": 0,
"inputs": [
{
"name": "samples",
"type": "LATENT",
"link": 7
},
{
"name": "vae",
"type": "VAE",
"link": 8
}
],
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
10
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "VAEDecode"
}
},
{
"id": 10,
"type": "PreviewImage",
"pos": [
1592,
279
],
"size": [
1370.7157657734379,
1041.8039240156252
],
"flags": {},
"order": 17,
"mode": 0,
"inputs": [
{
"name": "images",
"type": "IMAGE",
"link": 10
}
],
"properties": {
"Node name for S&R": "PreviewImage"
}
},
{
"id": 16,
"type": "PulidModelLoader",
"pos": [
-111,
-181
],
"size": {
"0": 304.0072021484375,
"1": 58
},
"flags": {},
"order": 3,
"mode": 0,
"outputs": [
{
"name": "PULID",
"type": "PULID",
"links": [
117,
136
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "PulidModelLoader"
},
"widgets_values": [
"ip-adapter_pulid_sdxl_fp16.safetensors"
]
},
{
"id": 19,
"type": "PulidEvaClipLoader",
"pos": [
54,
-69
],
"size": {
"0": 140,
"1": 26
},
"flags": {},
"order": 4,
"mode": 0,
"outputs": [
{
"name": "EVA_CLIP",
"type": "EVA_CLIP",
"links": [
81,
137
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "PulidEvaClipLoader"
}
},
{
"id": 17,
"type": "PulidInsightFaceLoader",
"pos": [
-18,
12
],
"size": {
"0": 210,
"1": 58
},
"flags": {},
"order": 5,
"mode": 0,
"outputs": [
{
"name": "FACEANALYSIS",
"type": "FACEANALYSIS",
"links": [
82,
138
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "PulidInsightFaceLoader"
},
"widgets_values": [
"CPU"
]
},
{
"id": 12,
"type": "LoadImage",
"pos": [
-34,
145
],
"size": [
261.645185990767,
346.38255171342325
],
"flags": {},
"order": 6,
"mode": 0,
"outputs": [
{
"name": "IMAGE",
"type": "IMAGE",
"links": [
114
],
"shape": 3,
"slot_index": 0
},
{
"name": "MASK",
"type": "MASK",
"links": null,
"shape": 3
}
],
"properties": {
"Node name for S&R": "LoadImage"
},
"widgets_values": [
"monalisa.png",
"image"
]
},
{
"id": 87,
"type": "MaskComposite",
"pos": [
15,
546
],
"size": [
210,
126
],
"flags": {},
"order": 9,
"mode": 0,
"inputs": [
{
"name": "destination",
"type": "MASK",
"link": 244
},
{
"name": "source",
"type": "MASK",
"link": 245
}
],
"outputs": [
{
"name": "MASK",
"type": "MASK",
"links": [
246,
247
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "MaskComposite"
},
"widgets_values": [
0,
0,
"add"
]
},
{
"id": 86,
"type": "SolidMask",
"pos": [
-304,
747
],
"size": {
"0": 210,
"1": 106
},
"flags": {},
"order": 7,
"mode": 0,
"outputs": [
{
"name": "MASK",
"type": "MASK",
"links": [
245
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "SolidMask"
},
"widgets_values": [
1,
640,
960
]
},
{
"id": 23,
"type": "CLIPTextEncode",
"pos": [
756,
-47
],
"size": [
316.32471195096673,
101.97065006593618
],
"flags": {},
"order": 10,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 94
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
34
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"blurry, malformed, low quality, worst quality, artifacts, noise, text, watermark, glitch, deformed, ugly, horror, ill"
]
},
{
"id": 47,
"type": "ApplyPulid",
"pos": [
765,
128
],
"size": {
"0": 315,
"1": 230
},
"flags": {},
"order": 14,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 141
},
{
"name": "pulid",
"type": "PULID",
"link": 136
},
{
"name": "eva_clip",
"type": "EVA_CLIP",
"link": 137
},
{
"name": "face_analysis",
"type": "FACEANALYSIS",
"link": 138
},
{
"name": "image",
"type": "IMAGE",
"link": 145
},
{
"name": "attn_mask",
"type": "MASK",
"link": 151
}
],
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
142
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "ApplyPulid"
},
"widgets_values": [
"fidelity",
0.7000000000000001,
0,
1
]
},
{
"id": 55,
"type": "CLIPTextEncode",
"pos": [
755,
-211
],
"size": {
"0": 315.23089599609375,
"1": 113.96450805664062
},
"flags": {},
"order": 11,
"mode": 0,
"inputs": [
{
"name": "clip",
"type": "CLIP",
"link": 156
}
],
"outputs": [
{
"name": "CONDITIONING",
"type": "CONDITIONING",
"links": [
160
],
"shape": 3,
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "CLIPTextEncode"
},
"widgets_values": [
"closeup two girl friends on the streets of a cyberpunk city, cinematic, hoodie, multicolored hair, highly detailed, 4k, high resolution"
]
},
{
"id": 3,
"type": "KSampler",
"pos": [
1162,
38
],
"size": {
"0": 341.2750244140625,
"1": 262
},
"flags": {},
"order": 15,
"mode": 0,
"inputs": [
{
"name": "model",
"type": "MODEL",
"link": 142
},
{
"name": "positive",
"type": "CONDITIONING",
"link": 160
},
{
"name": "negative",
"type": "CONDITIONING",
"link": 34
},
{
"name": "latent_image",
"type": "LATENT",
"link": 2
}
],
"outputs": [
{
"name": "LATENT",
"type": "LATENT",
"links": [
7
],
"slot_index": 0
}
],
"properties": {
"Node name for S&R": "KSampler"
},
"widgets_values": [
70,
"fixed",
30,
6,
"dpmpp_2m",
"karras",
1
]
},
{
"id": 4,
"type": "CheckpointLoaderSimple",
"pos": [
-131,
-342
],
"size": {
"0": 319.03692626953125,
"1": 101.3391342163086
},
"flags": {},
"order": 8,
"mode": 0,
"outputs": [
{
"name": "MODEL",
"type": "MODEL",
"links": [
133
],
"slot_index": 0
},
{
"name": "CLIP",
"type": "CLIP",
"links": [
94,
156
],
"slot_index": 1
},
{
"name": "VAE",
"type": "VAE",
"links": [
8
],
"slot_index": 2
}
],
"properties": {
"Node name for S&R": "CheckpointLoaderSimple"
},
"widgets_values": [
"sdxl/AlbedoBaseXL.safetensors"
]
}
],
"links": [
[
2,
5,
0,
3,
3,
"LATENT"
],
[
7,
3,
0,
8,
0,
"LATENT"
],
[
8,
4,
2,
8,
1,
"VAE"
],
[
10,
8,
0,
10,
0,
"IMAGE"
],
[
34,
23,
0,
3,
2,
"CONDITIONING"
],
[
81,
19,
0,
33,
2,
"EVA_CLIP"
],
[
82,
17,
0,
33,
3,
"FACEANALYSIS"
],
[
94,
4,
1,
23,
0,
"CLIP"
],
[
114,
12,
0,
33,
4,
"IMAGE"
],
[
117,
16,
0,
33,
1,
"PULID"
],
[
133,
4,
0,
33,
0,
"MODEL"
],
[
136,
16,
0,
47,
1,
"PULID"
],
[
137,
19,
0,
47,
2,
"EVA_CLIP"
],
[
138,
17,
0,
47,
3,
"FACEANALYSIS"
],
[
141,
33,
0,
47,
0,
"MODEL"
],
[
142,
47,
0,
3,
0,
"MODEL"
],
[
145,
49,
0,
47,
4,
"IMAGE"
],
[
151,
48,
0,
47,
5,
"MASK"
],
[
156,
4,
1,
55,
0,
"CLIP"
],
[
160,
55,
0,
3,
1,
"CONDITIONING"
],
[
244,
85,
0,
87,
0,
"MASK"
],
[
245,
86,
0,
87,
1,
"MASK"
],
[
246,
87,
0,
48,
0,
"MASK"
],
[
247,
87,
0,
33,
5,
"MASK"
]
],
"groups": [],
"config": {},
"extra": {},
"version": 0.4
}
+94 -17
View File
@@ -1,7 +1,9 @@
import torch
from torch import nn
import torchvision.transforms as T
import torch.nn.functional as F
import os
import math
import folder_paths
import comfy.utils
from insightface.app import FaceAnalysis
@@ -69,6 +71,8 @@ def tensor_to_size(source, dest_size):
elif source_size > dest_size:
source = source[:dest_size]
return source
def set_model_patch_replace(model, patch_kwargs, key):
to = model.model_options["transformer_options"].copy()
if "patches_replace" not in to:
@@ -110,14 +114,16 @@ class Attn2Replace:
return out.to(dtype=dtype)
def pulid_attention(out, q, k, v, extra_options, module_key='', pulid=None, cond=None, uncond=None, weight=1.0, num_zero=8, ortho=False, ortho_v2=False, **kwargs):
def pulid_attention(out, q, k, v, extra_options, module_key='', pulid=None, cond=None, uncond=None, weight=1.0, ortho=False, ortho_v2=False, mask=None, **kwargs):
k_key = module_key + "_to_k_ip"
v_key = module_key + "_to_v_ip"
dtype = q.dtype
seq_len = q.shape[1]
cond_or_uncond = extra_options["cond_or_uncond"]
b = q.shape[0]
batch_prompt = b // len(cond_or_uncond)
_, _, oh, ow = extra_options["original_shape"]
#conds = torch.cat([uncond.repeat(batch_prompt, 1, 1), cond.repeat(batch_prompt, 1, 1)], dim=0)
#zero_tensor = torch.zeros((conds.size(0), num_zero, conds.size(-1)), dtype=conds.dtype, device=conds.device)
@@ -125,10 +131,6 @@ def pulid_attention(out, q, k, v, extra_options, module_key='', pulid=None, cond
#ip_k = pulid.ip_layers.to_kvs[k_key](conds)
#ip_v = pulid.ip_layers.to_kvs[v_key](conds)
if num_zero > 0:
zero_tensor = torch.zeros((cond.size(0), num_zero, cond.size(-1)), dtype=cond.dtype, device=cond.device)
cond = torch.cat([cond, zero_tensor], dim=1)
uncond = torch.cat([uncond, zero_tensor], dim=1)
k_cond = pulid.ip_layers.to_kvs[k_key](cond).repeat(batch_prompt, 1, 1)
k_uncond = pulid.ip_layers.to_kvs[k_key](uncond).repeat(batch_prompt, 1, 1)
v_cond = pulid.ip_layers.to_kvs[v_key](cond).repeat(batch_prompt, 1, 1)
@@ -143,7 +145,7 @@ def pulid_attention(out, q, k, v, extra_options, module_key='', pulid=None, cond
out_ip = out_ip.to(dtype=torch.float32)
projection = (torch.sum((out * out_ip), dim=-2, keepdim=True) / torch.sum((out * out), dim=-2, keepdim=True) * out)
orthogonal = out_ip - projection
out = weight * orthogonal
out_ip = weight * orthogonal
elif ortho_v2:
out = out.to(dtype=torch.float32)
out_ip = out_ip.to(dtype=torch.float32)
@@ -152,11 +154,35 @@ def pulid_attention(out, q, k, v, extra_options, module_key='', pulid=None, cond
attn_mean = attn_mean[:, :, :5].sum(dim=-1, keepdim=True)
projection = (torch.sum((out * out_ip), dim=-2, keepdim=True) / torch.sum((out * out), dim=-2, keepdim=True) * out)
orthogonal = out_ip + (attn_mean - 1) * projection
out = weight * orthogonal
out_ip = weight * orthogonal
else:
out = out_ip * weight
out_ip = out_ip * weight
return out.to(dtype=dtype)
if mask is not None:
mask_h = oh / math.sqrt(oh * ow / seq_len)
mask_h = int(mask_h) + int((seq_len % int(mask_h)) != 0)
mask_w = seq_len // mask_h
mask = F.interpolate(mask.unsqueeze(1), size=(mask_h, mask_w), mode="bilinear").squeeze(1)
mask = tensor_to_size(mask, batch_prompt)
mask = mask.repeat(len(cond_or_uncond), 1, 1)
mask = mask.view(mask.shape[0], -1, 1).repeat(1, 1, out.shape[2])
# covers cases where extreme aspect ratios can cause the mask to have a wrong size
mask_len = mask_h * mask_w
if mask_len < seq_len:
pad_len = seq_len - mask_len
pad1 = pad_len // 2
pad2 = pad_len - pad1
mask = F.pad(mask, (0, 0, pad1, pad2), value=0.0)
elif mask_len > seq_len:
crop_start = (mask_len - seq_len) // 2
mask = mask[:, crop_start:crop_start+seq_len, :]
out_ip = out_ip * mask
return out_ip.to(dtype=dtype)
def to_gray(img):
x = 0.299 * img[:, 0:1] + 0.587 * img[:, 1:2] + 0.114 * img[:, 2:3]
@@ -256,13 +282,16 @@ class ApplyPulid:
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }),
},
"optional": {
"attn_mask": ("MASK", ),
},
}
RETURN_TYPES = ("MODEL",)
FUNCTION = "apply_pulid"
CATEGORY = "pulid"
def apply_pulid(self, model, pulid, eva_clip, face_analysis, image, method, weight, start_at, end_at):
def apply_pulid(self, model, pulid, eva_clip, face_analysis, image, weight, start_at, end_at, method=None, noise=0.0, fidelity=None, projection=None, attn_mask=None):
work_model = model.clone()
device = comfy.model_management.get_torch_device()
@@ -273,11 +302,18 @@ class ApplyPulid:
eva_clip.to(device, dtype=dtype)
pulid_model = PulidModel(pulid).to(device, dtype=dtype)
if method == "fidelity":
if attn_mask is not None:
if attn_mask.dim() > 3:
attn_mask = attn_mask.squeeze(-1)
elif attn_mask.dim() < 3:
attn_mask = attn_mask.unsqueeze(0)
attn_mask = attn_mask.to(device, dtype=dtype)
if method == "fidelity" or projection == "ortho_v2":
num_zero = 8
ortho = False
ortho_v2 = True
elif method == "style":
elif method == "style" or projection == "ortho":
num_zero = 16
ortho = True
ortho_v2 = False
@@ -286,6 +322,9 @@ class ApplyPulid:
ortho = False
ortho_v2 = False
if fidelity is not None:
num_zero = fidelity
#face_analysis.det_model.input_size = (640,640)
image = tensor_to_image(image)
@@ -346,10 +385,16 @@ class ApplyPulid:
# combine embeddings
id_cond = torch.cat([iface_embeds, id_cond_vit], dim=-1)
if noise == 0:
id_uncond = torch.zeros_like(id_cond)
else:
id_uncond = torch.rand_like(id_cond) * noise
id_vit_hidden_uncond = []
for idx in range(len(id_vit_hidden)):
if noise == 0:
id_vit_hidden_uncond.append(torch.zeros_like(id_vit_hidden[idx]))
else:
id_vit_hidden_uncond.append(torch.rand_like(id_vit_hidden[idx]) * noise)
cond.append(pulid_model.get_image_embeds(id_cond, id_vit_hidden))
uncond.append(pulid_model.get_image_embeds(id_uncond, id_vit_hidden_uncond))
@@ -361,6 +406,14 @@ class ApplyPulid:
cond = torch.mean(cond, dim=0, keepdim=True)
uncond = torch.mean(uncond, dim=0, keepdim=True)
if num_zero > 0:
if noise == 0:
zero_tensor = torch.zeros((cond.size(0), num_zero, cond.size(-1)), dtype=dtype, device=device)
else:
zero_tensor = torch.rand((cond.size(0), num_zero, cond.size(-1)), dtype=dtype, device=device) * noise
cond = torch.cat([cond, zero_tensor], dim=1)
uncond = torch.cat([uncond, zero_tensor], dim=1)
sigma_start = work_model.get_model_object("model_sampling").percent_to_sigma(start_at)
sigma_end = work_model.get_model_object("model_sampling").percent_to_sigma(end_at)
@@ -371,9 +424,9 @@ class ApplyPulid:
"uncond": uncond,
"sigma_start": sigma_start,
"sigma_end": sigma_end,
"num_zero": num_zero,
"ortho": ortho,
"ortho_v2": ortho_v2,
"mask": attn_mask,
}
number = 0
@@ -396,16 +449,40 @@ class ApplyPulid:
return (work_model,)
class ApplyPulidAdvanced(ApplyPulid):
@classmethod
def INPUT_TYPES(s):
return {
"required": {
"model": ("MODEL", ),
"pulid": ("PULID", ),
"eva_clip": ("EVA_CLIP", ),
"face_analysis": ("FACEANALYSIS", ),
"image": ("IMAGE", ),
"weight": ("FLOAT", {"default": 1.0, "min": -1.0, "max": 5.0, "step": 0.05 }),
"projection": (["ortho_v2", "ortho", "none"],),
"fidelity": ("INT", {"default": 8, "min": 0, "max": 32, "step": 1 }),
"noise": ("FLOAT", {"default": 0.0, "min": -1.0, "max": 1.0, "step": 0.1 }),
"start_at": ("FLOAT", {"default": 0.0, "min": 0.0, "max": 1.0, "step": 0.001 }),
"end_at": ("FLOAT", {"default": 1.0, "min": 0.0, "max": 1.0, "step": 0.001 }),
},
"optional": {
"attn_mask": ("MASK", ),
},
}
NODE_CLASS_MAPPINGS = {
"PulidModelLoader": PulidModelLoader,
"PulidInsightFaceLoader": PulidInsightFaceLoader,
"PulidEvaClipLoader": PulidEvaClipLoader,
"ApplyPulid": ApplyPulid,
"ApplyPulidAdvanced": ApplyPulidAdvanced,
}
NODE_DISPLAY_NAME_MAPPINGS = {
"PulidModelLoader": "Load Pulid Model",
"PulidInsightFaceLoader": "Load InsightFace",
"PulidEvaClipLoader": "Load Eva Clip",
"ApplyPulid": "Apply Pulid",
"PulidModelLoader": "Load PuLID Model",
"PulidInsightFaceLoader": "Load InsightFace (PuLID)",
"PulidEvaClipLoader": "Load Eva Clip (PuLID)",
"ApplyPulid": "Apply PuLID",
"ApplyPulidAdvanced": "Apply PuLID Advanced",
}