PixArt LoRA support
Supports models from the example training code. Weight loading will probably have to be changed for native formats.
This commit is contained in:
+76
-26
@@ -58,38 +58,88 @@ for depth in range(28):
|
||||
(f"blocks.{depth}.cross_attn.proj.bias" ,f"transformer_blocks.{depth}.attn2.to_out.0.bias"),
|
||||
]
|
||||
|
||||
def convert_pixart_state_dict(unet_state_dict):
|
||||
if "adaln_single.emb.resolution_embedder.linear_1.weight" in unet_state_dict.keys():
|
||||
def find_prefix(state_dict, target_key):
|
||||
prefix = ""
|
||||
for k in state_dict.keys():
|
||||
if k.endswith(target_key):
|
||||
prefix = k.split(target_key)[0]
|
||||
break
|
||||
return prefix
|
||||
|
||||
def convert_state_dict(state_dict):
|
||||
if "adaln_single.emb.resolution_embedder.linear_1.weight" in state_dict.keys():
|
||||
cmap = conversion_map + conversion_map_ms
|
||||
else:
|
||||
cmap = conversion_map
|
||||
|
||||
new_state_dict = {k: unet_state_dict.pop(v) for k,v in cmap}
|
||||
new_state_dict = {k: state_dict[v] for k,v in cmap}
|
||||
matched = list(v for k,v in cmap if v in state_dict.keys())
|
||||
|
||||
for depth in range(28):
|
||||
# Self Attention
|
||||
q = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_q.weight")
|
||||
k = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_k.weight")
|
||||
v = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_v.weight")
|
||||
new_state_dict[f"blocks.{depth}.attn.qkv.weight"] = torch.cat((q,k,v), dim=0)
|
||||
qb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_q.bias")
|
||||
kb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_k.bias")
|
||||
vb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_v.bias")
|
||||
new_state_dict[f"blocks.{depth}.attn.qkv.bias"] = torch.cat((qb,kb,vb), dim=0)
|
||||
for wb in ["weight", "bias"]:
|
||||
# Self Attention
|
||||
key = lambda a: f"transformer_blocks.{depth}.attn1.to_{a}.{wb}"
|
||||
new_state_dict[f"blocks.{depth}.attn.qkv.{wb}"] = torch.cat((
|
||||
state_dict[key('q')], state_dict[key('k')], state_dict[key('v')]
|
||||
), dim=0)
|
||||
matched += [key('q'), key('k'), key('v')]
|
||||
|
||||
# Cross-attention (linear)
|
||||
q = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_q.weight")
|
||||
k = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_k.weight")
|
||||
v = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_v.weight")
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.q_linear.weight"] = q
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.weight"] = torch.cat((k,v), dim=0)
|
||||
qb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_q.bias")
|
||||
kb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_k.bias")
|
||||
vb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_v.bias")
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.q_linear.bias"] = qb
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.bias"] = torch.cat((kb,vb), dim=0)
|
||||
# Cross-attention (linear)
|
||||
key = lambda a: f"transformer_blocks.{depth}.attn2.to_{a}.{wb}"
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.q_linear.{wb}"] = state_dict[key('q')]
|
||||
new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.{wb}"] = torch.cat((
|
||||
state_dict[key('k')], state_dict[key('v')]
|
||||
), dim=0)
|
||||
matched += [key('q'), key('k'), key('v')]
|
||||
|
||||
if len(matched) < len(state_dict):
|
||||
print(f"PixArt: UNET conversion has leftover keys! ({len(matched)} vs {len(state_dict)})")
|
||||
print(list( set(state_dict.keys()) - set(matched) ))
|
||||
|
||||
return new_state_dict
|
||||
|
||||
# Same as above but for LoRA weights:
|
||||
def convert_lora_state_dict(state_dict):
|
||||
# peft
|
||||
rep_ap = lambda x: x.replace(".weight", ".lora_A.weight")
|
||||
rep_bp = lambda x: x.replace(".weight", ".lora_B.weight")
|
||||
# koyha
|
||||
rep_ak = lambda x: x.replace(".weight", ".lora_down.weight")
|
||||
rep_bk = lambda x: x.replace(".weight", ".lora_up.weight")
|
||||
|
||||
prefix = find_prefix(state_dict, "adaln_single.linear.lora_A.weight")
|
||||
state_dict = {k[len(prefix):]:v for k,v in state_dict.items()}
|
||||
|
||||
cmap = []
|
||||
cmap_unet = conversion_map + conversion_map_ms # todo: 512 model
|
||||
for k, v in cmap_unet:
|
||||
if not v.endswith(".weight"):
|
||||
continue
|
||||
cmap.append((rep_ak(k), rep_ap(v)))
|
||||
cmap.append((rep_bk(k), rep_bp(v)))
|
||||
|
||||
new_state_dict = {k: state_dict[v] for k,v in cmap}
|
||||
matched = list(v for k,v in cmap if v in state_dict.keys())
|
||||
|
||||
for fp, fk in ((rep_ap, rep_ak),(rep_bp, rep_bk)):
|
||||
for depth in range(28):
|
||||
# Self Attention
|
||||
key = lambda a: fp(f"transformer_blocks.{depth}.attn1.to_{a}.weight")
|
||||
new_state_dict[fk(f"blocks.{depth}.attn.qkv.weight")] = torch.cat((
|
||||
state_dict[key('q')], state_dict[key('k')], state_dict[key('v')]
|
||||
), dim=0)
|
||||
matched += [key('q'), key('k'), key('v')]
|
||||
|
||||
# Cross-attention (linear)
|
||||
key = lambda a: fp(f"transformer_blocks.{depth}.attn2.to_{a}.weight")
|
||||
new_state_dict[fk(f"blocks.{depth}.cross_attn.q_linear.weight")] = state_dict[key('q')]
|
||||
new_state_dict[fk(f"blocks.{depth}.cross_attn.kv_linear.weight")] = torch.cat((
|
||||
state_dict[key('k')], state_dict[key('v')]
|
||||
), dim=0)
|
||||
matched += [key('q'), key('k'), key('v')]
|
||||
|
||||
if len(matched) < len(state_dict):
|
||||
print(f"PixArt: LoRA conversion has leftover keys! ({len(matched)} vs {len(state_dict)})")
|
||||
print(list( set(state_dict.keys()) - set(matched) ))
|
||||
|
||||
if len(unet_state_dict.keys()) > 0:
|
||||
print(f"PixArt: UNET conversion has leftover keys!:\n{unet_state_dict.keys()}")
|
||||
|
||||
return new_state_dict
|
||||
|
||||
+11
-3
@@ -5,7 +5,7 @@ import comfy.model_base
|
||||
import comfy.utils
|
||||
import torch
|
||||
from comfy import model_management
|
||||
from .diffusers_convert import convert_pixart_state_dict
|
||||
from .diffusers_convert import convert_state_dict
|
||||
|
||||
class EXM_PixArt(comfy.supported_models_base.BASE):
|
||||
unet_config = {}
|
||||
@@ -26,8 +26,16 @@ class EXM_PixArt(comfy.supported_models_base.BASE):
|
||||
def load_pixart(model_path, model_conf):
|
||||
state_dict = comfy.utils.load_torch_file(model_path)
|
||||
state_dict = state_dict.get("model", state_dict)
|
||||
if "caption_projection.y_embedding" in state_dict:
|
||||
state_dict = convert_pixart_state_dict(state_dict) # Diffusers
|
||||
|
||||
# prefix
|
||||
for prefix in ["model.diffusion_model.",]:
|
||||
if any(True for x in state_dict if x.startswith(prefix)):
|
||||
state_dict = {k[len(prefix):]:v for k,v in state_dict.items()}
|
||||
|
||||
# diffusers
|
||||
if "adaln_single.linear.weight" in state_dict:
|
||||
state_dict = convert_state_dict(state_dict) # Diffusers
|
||||
|
||||
parameters = comfy.utils.calculate_parameters(state_dict)
|
||||
unet_dtype = model_management.unet_dtype(model_params=parameters)
|
||||
|
||||
|
||||
+146
@@ -0,0 +1,146 @@
|
||||
import os
|
||||
import copy
|
||||
import json
|
||||
import torch
|
||||
import comfy.lora
|
||||
import comfy.model_management
|
||||
from comfy.model_patcher import ModelPatcher
|
||||
from .diffusers_convert import convert_lora_state_dict
|
||||
|
||||
class EXM_PixArt_ModelPatcher(ModelPatcher):
|
||||
def calculate_weight(self, patches, weight, key):
|
||||
"""
|
||||
This is almost the same as the comfy function, but stripped down to just the LoRA patch code.
|
||||
The problem with the original code is the q/k/v keys being combined into one for the attention.
|
||||
In the diffusers code, they're treated as separate keys, but in the reference code they're recombined (q+kv|qkv).
|
||||
This means, for example, that the [1152,1152] weights become [3456,1152] in the state dict.
|
||||
The issue with this is that the LoRA weights are [128,1152],[1152,128] and become [384,1162],[3456,128] instead.
|
||||
|
||||
This is the best thing I could think of that would fix that, but it's very fragile.
|
||||
- Check key shape to determine if it needs the fallback logic
|
||||
- Cut the input into parts based on the shape (undoing the torch.cat)
|
||||
- Do the matrix multiplication logic
|
||||
- Recombine them to match the expected shape
|
||||
"""
|
||||
for p in patches:
|
||||
alpha = p[0]
|
||||
v = p[1]
|
||||
strength_model = p[2]
|
||||
if strength_model != 1.0:
|
||||
weight *= strength_model
|
||||
|
||||
if isinstance(v, list):
|
||||
v = (self.calculate_weight(v[1:], v[0].clone(), key), )
|
||||
|
||||
if len(v) == 2:
|
||||
patch_type = v[0]
|
||||
v = v[1]
|
||||
|
||||
if patch_type == "lora":
|
||||
mat1 = comfy.model_management.cast_to_device(v[0], weight.device, torch.float32)
|
||||
mat2 = comfy.model_management.cast_to_device(v[1], weight.device, torch.float32)
|
||||
if v[2] is not None:
|
||||
alpha *= v[2] / mat2.shape[0]
|
||||
try:
|
||||
mat1 = mat1.flatten(start_dim=1)
|
||||
mat2 = mat2.flatten(start_dim=1)
|
||||
|
||||
ch1 = mat1.shape[0] // mat2.shape[1]
|
||||
ch2 = mat2.shape[0] // mat1.shape[1]
|
||||
### Fallback logic for shape mismatch ###
|
||||
if mat1.shape[0] != mat2.shape[1] and ch1 == ch2 and (mat1.shape[0]/mat2.shape[1])%1 == 0:
|
||||
mat1 = mat1.chunk(ch1, dim=0)
|
||||
mat2 = mat2.chunk(ch1, dim=0)
|
||||
weight += torch.cat(
|
||||
[alpha * torch.mm(mat1[x], mat2[x]) for x in range(ch1)],
|
||||
dim=0,
|
||||
).reshape(weight.shape).type(weight.dtype)
|
||||
else:
|
||||
weight += (alpha * torch.mm(mat1, mat2)).reshape(weight.shape).type(weight.dtype)
|
||||
except Exception as e:
|
||||
print("ERROR", key, e)
|
||||
return weight
|
||||
|
||||
def clone(self):
|
||||
n = EXM_PixArt_ModelPatcher(self.model, self.load_device, self.offload_device, self.size, self.current_device, weight_inplace_update=self.weight_inplace_update)
|
||||
n.patches = {}
|
||||
for k in self.patches:
|
||||
n.patches[k] = self.patches[k][:]
|
||||
|
||||
n.object_patches = self.object_patches.copy()
|
||||
n.model_options = copy.deepcopy(self.model_options)
|
||||
n.model_keys = self.model_keys
|
||||
return n
|
||||
|
||||
def replace_model_patcher(model):
|
||||
n = EXM_PixArt_ModelPatcher(
|
||||
model = model.model,
|
||||
size = model.size,
|
||||
load_device = model.load_device,
|
||||
offload_device = model.offload_device,
|
||||
current_device = model.current_device,
|
||||
weight_inplace_update = model.weight_inplace_update,
|
||||
)
|
||||
n.patches = {}
|
||||
for k in model.patches:
|
||||
n.patches[k] = model.patches[k][:]
|
||||
|
||||
n.object_patches = model.object_patches.copy()
|
||||
n.model_options = copy.deepcopy(model.model_options)
|
||||
n.model_keys = model.model_keys
|
||||
return n
|
||||
|
||||
def find_peft_alpha(path):
|
||||
def load_json(json_path):
|
||||
with open(json_path) as f:
|
||||
data = json.load(f)
|
||||
alpha = data.get("lora_alpha")
|
||||
alpha = alpha or data.get("alpha")
|
||||
if not alpha:
|
||||
print(" Found config but `lora_alpha` is missing!")
|
||||
else:
|
||||
print(f" Found config at {json_path} [alpha:{alpha}]")
|
||||
return alpha
|
||||
|
||||
# For some weird reason peft doesn't include the alpha in the actual model
|
||||
print("PixArt: Warning! This is a PEFT LoRA. Trying to find config...")
|
||||
files = [
|
||||
f"{os.path.splitext(path)[0]}.json",
|
||||
f"{os.path.splitext(path)[0]}.config.json",
|
||||
os.path.join(os.path.dirname(path),"adapter_config.json"),
|
||||
]
|
||||
for file in files:
|
||||
if os.path.isfile(file):
|
||||
return load_json(file)
|
||||
|
||||
print(" Missing config/alpha! assuming alpha of 8. Consider converting it/adding a config json to it.")
|
||||
return 8.0
|
||||
|
||||
def load_pixart_lora(model, lora, lora_path, strength):
|
||||
k_back = lambda x: x.replace(".lora_up.weight", "")
|
||||
# need to convert the actual weights for this to work.
|
||||
if any(True for x in lora.keys() if x.endswith("adaln_single.linear.lora_A.weight")):
|
||||
lora = convert_lora_state_dict(lora)
|
||||
alpha = find_peft_alpha(lora_path)
|
||||
lora.update({f"{k_back(x)}.alpha":torch.tensor(alpha) for x in lora.keys() if "lora_up" in x})
|
||||
|
||||
key_map = {k_back(x):f"diffusion_model.{k_back(x)}.weight" for x in lora.keys() if "lora_up" in x} # fake
|
||||
|
||||
loaded = comfy.lora.load_lora(lora, key_map)
|
||||
if model is not None:
|
||||
# switch to custom model patcher when using LoRAs
|
||||
if isinstance(model, EXM_PixArt_ModelPatcher):
|
||||
new_modelpatcher = model.clone()
|
||||
else:
|
||||
new_modelpatcher = replace_model_patcher(model)
|
||||
k = new_modelpatcher.add_patches(loaded, strength)
|
||||
else:
|
||||
k = ()
|
||||
new_modelpatcher = None
|
||||
|
||||
k = set(k)
|
||||
for x in loaded:
|
||||
if (x not in k):
|
||||
print("NOT LOADED", x)
|
||||
|
||||
return new_modelpatcher
|
||||
@@ -3,7 +3,9 @@ import json
|
||||
import torch
|
||||
import folder_paths
|
||||
|
||||
from comfy import utils
|
||||
from .conf import pixart_conf, pixart_res
|
||||
from .lora import load_pixart_lora
|
||||
from .loader import load_pixart
|
||||
from .sampler import sample_pixart
|
||||
|
||||
@@ -51,6 +53,45 @@ class PixArtResolutionSelect():
|
||||
width, height = pixart_res[model][ratio]
|
||||
return (width,height)
|
||||
|
||||
class PixArtLoraLoader:
|
||||
def __init__(self):
|
||||
self.loaded_lora = None
|
||||
|
||||
@classmethod
|
||||
def INPUT_TYPES(s):
|
||||
return {
|
||||
"required": {
|
||||
"model": ("MODEL",),
|
||||
"lora_name": (folder_paths.get_filename_list("loras"), ),
|
||||
"strength": ("FLOAT", {"default": 1.0, "min": -20.0, "max": 20.0, "step": 0.01}),
|
||||
}
|
||||
}
|
||||
RETURN_TYPES = ("MODEL",)
|
||||
FUNCTION = "load_lora"
|
||||
CATEGORY = "ExtraModels/PixArt"
|
||||
TITLE = "PixArt Load LoRA"
|
||||
|
||||
def load_lora(self, model, lora_name, strength,):
|
||||
if strength == 0:
|
||||
return (model)
|
||||
|
||||
lora_path = folder_paths.get_full_path("loras", lora_name)
|
||||
lora = None
|
||||
if self.loaded_lora is not None:
|
||||
if self.loaded_lora[0] == lora_path:
|
||||
lora = self.loaded_lora[1]
|
||||
else:
|
||||
temp = self.loaded_lora
|
||||
self.loaded_lora = None
|
||||
del temp
|
||||
|
||||
if lora is None:
|
||||
lora = utils.load_torch_file(lora_path, safe_load=True)
|
||||
self.loaded_lora = (lora_path, lora)
|
||||
|
||||
model_lora = load_pixart_lora(model, lora, lora_path, strength,)
|
||||
return (model_lora,)
|
||||
|
||||
class PixArtDPMSampler:
|
||||
"""
|
||||
The sampler from the reference code.
|
||||
@@ -145,6 +186,7 @@ class PixArtT5TextEncode:
|
||||
NODE_CLASS_MAPPINGS = {
|
||||
"PixArtCheckpointLoader" : PixArtCheckpointLoader,
|
||||
"PixArtResolutionSelect" : PixArtResolutionSelect,
|
||||
"PixArtLoraLoader" : PixArtLoraLoader,
|
||||
"PixArtDPMSampler" : PixArtDPMSampler,
|
||||
"PixArtT5TextEncode" : PixArtT5TextEncode,
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user