diff --git a/PixArt/diffusers_convert.py b/PixArt/diffusers_convert.py index d2bf810..7151773 100644 --- a/PixArt/diffusers_convert.py +++ b/PixArt/diffusers_convert.py @@ -58,38 +58,88 @@ for depth in range(28): (f"blocks.{depth}.cross_attn.proj.bias" ,f"transformer_blocks.{depth}.attn2.to_out.0.bias"), ] -def convert_pixart_state_dict(unet_state_dict): - if "adaln_single.emb.resolution_embedder.linear_1.weight" in unet_state_dict.keys(): +def find_prefix(state_dict, target_key): + prefix = "" + for k in state_dict.keys(): + if k.endswith(target_key): + prefix = k.split(target_key)[0] + break + return prefix + +def convert_state_dict(state_dict): + if "adaln_single.emb.resolution_embedder.linear_1.weight" in state_dict.keys(): cmap = conversion_map + conversion_map_ms else: cmap = conversion_map - new_state_dict = {k: unet_state_dict.pop(v) for k,v in cmap} + new_state_dict = {k: state_dict[v] for k,v in cmap} + matched = list(v for k,v in cmap if v in state_dict.keys()) for depth in range(28): - # Self Attention - q = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_q.weight") - k = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_k.weight") - v = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_v.weight") - new_state_dict[f"blocks.{depth}.attn.qkv.weight"] = torch.cat((q,k,v), dim=0) - qb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_q.bias") - kb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_k.bias") - vb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn1.to_v.bias") - new_state_dict[f"blocks.{depth}.attn.qkv.bias"] = torch.cat((qb,kb,vb), dim=0) + for wb in ["weight", "bias"]: + # Self Attention + key = lambda a: f"transformer_blocks.{depth}.attn1.to_{a}.{wb}" + new_state_dict[f"blocks.{depth}.attn.qkv.{wb}"] = torch.cat(( + state_dict[key('q')], state_dict[key('k')], state_dict[key('v')] + ), dim=0) + matched += [key('q'), key('k'), key('v')] - # Cross-attention (linear) - q = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_q.weight") - k = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_k.weight") - v = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_v.weight") - new_state_dict[f"blocks.{depth}.cross_attn.q_linear.weight"] = q - new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.weight"] = torch.cat((k,v), dim=0) - qb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_q.bias") - kb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_k.bias") - vb = unet_state_dict.pop(f"transformer_blocks.{depth}.attn2.to_v.bias") - new_state_dict[f"blocks.{depth}.cross_attn.q_linear.bias"] = qb - new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.bias"] = torch.cat((kb,vb), dim=0) + # Cross-attention (linear) + key = lambda a: f"transformer_blocks.{depth}.attn2.to_{a}.{wb}" + new_state_dict[f"blocks.{depth}.cross_attn.q_linear.{wb}"] = state_dict[key('q')] + new_state_dict[f"blocks.{depth}.cross_attn.kv_linear.{wb}"] = torch.cat(( + state_dict[key('k')], state_dict[key('v')] + ), dim=0) + matched += [key('q'), key('k'), key('v')] + + if len(matched) < len(state_dict): + print(f"PixArt: UNET conversion has leftover keys! ({len(matched)} vs {len(state_dict)})") + print(list( set(state_dict.keys()) - set(matched) )) + + return new_state_dict + +# Same as above but for LoRA weights: +def convert_lora_state_dict(state_dict): + # peft + rep_ap = lambda x: x.replace(".weight", ".lora_A.weight") + rep_bp = lambda x: x.replace(".weight", ".lora_B.weight") + # koyha + rep_ak = lambda x: x.replace(".weight", ".lora_down.weight") + rep_bk = lambda x: x.replace(".weight", ".lora_up.weight") + + prefix = find_prefix(state_dict, "adaln_single.linear.lora_A.weight") + state_dict = {k[len(prefix):]:v for k,v in state_dict.items()} + + cmap = [] + cmap_unet = conversion_map + conversion_map_ms # todo: 512 model + for k, v in cmap_unet: + if not v.endswith(".weight"): + continue + cmap.append((rep_ak(k), rep_ap(v))) + cmap.append((rep_bk(k), rep_bp(v))) + + new_state_dict = {k: state_dict[v] for k,v in cmap} + matched = list(v for k,v in cmap if v in state_dict.keys()) + + for fp, fk in ((rep_ap, rep_ak),(rep_bp, rep_bk)): + for depth in range(28): + # Self Attention + key = lambda a: fp(f"transformer_blocks.{depth}.attn1.to_{a}.weight") + new_state_dict[fk(f"blocks.{depth}.attn.qkv.weight")] = torch.cat(( + state_dict[key('q')], state_dict[key('k')], state_dict[key('v')] + ), dim=0) + matched += [key('q'), key('k'), key('v')] + + # Cross-attention (linear) + key = lambda a: fp(f"transformer_blocks.{depth}.attn2.to_{a}.weight") + new_state_dict[fk(f"blocks.{depth}.cross_attn.q_linear.weight")] = state_dict[key('q')] + new_state_dict[fk(f"blocks.{depth}.cross_attn.kv_linear.weight")] = torch.cat(( + state_dict[key('k')], state_dict[key('v')] + ), dim=0) + matched += [key('q'), key('k'), key('v')] + + if len(matched) < len(state_dict): + print(f"PixArt: LoRA conversion has leftover keys! ({len(matched)} vs {len(state_dict)})") + print(list( set(state_dict.keys()) - set(matched) )) - if len(unet_state_dict.keys()) > 0: - print(f"PixArt: UNET conversion has leftover keys!:\n{unet_state_dict.keys()}") - return new_state_dict diff --git a/PixArt/loader.py b/PixArt/loader.py index 1757a79..e84fe14 100644 --- a/PixArt/loader.py +++ b/PixArt/loader.py @@ -5,7 +5,7 @@ import comfy.model_base import comfy.utils import torch from comfy import model_management -from .diffusers_convert import convert_pixart_state_dict +from .diffusers_convert import convert_state_dict class EXM_PixArt(comfy.supported_models_base.BASE): unet_config = {} @@ -26,8 +26,16 @@ class EXM_PixArt(comfy.supported_models_base.BASE): def load_pixart(model_path, model_conf): state_dict = comfy.utils.load_torch_file(model_path) state_dict = state_dict.get("model", state_dict) - if "caption_projection.y_embedding" in state_dict: - state_dict = convert_pixart_state_dict(state_dict) # Diffusers + + # prefix + for prefix in ["model.diffusion_model.",]: + if any(True for x in state_dict if x.startswith(prefix)): + state_dict = {k[len(prefix):]:v for k,v in state_dict.items()} + + # diffusers + if "adaln_single.linear.weight" in state_dict: + state_dict = convert_state_dict(state_dict) # Diffusers + parameters = comfy.utils.calculate_parameters(state_dict) unet_dtype = model_management.unet_dtype(model_params=parameters) diff --git a/PixArt/lora.py b/PixArt/lora.py new file mode 100644 index 0000000..285fba6 --- /dev/null +++ b/PixArt/lora.py @@ -0,0 +1,146 @@ +import os +import copy +import json +import torch +import comfy.lora +import comfy.model_management +from comfy.model_patcher import ModelPatcher +from .diffusers_convert import convert_lora_state_dict + +class EXM_PixArt_ModelPatcher(ModelPatcher): + def calculate_weight(self, patches, weight, key): + """ + This is almost the same as the comfy function, but stripped down to just the LoRA patch code. + The problem with the original code is the q/k/v keys being combined into one for the attention. + In the diffusers code, they're treated as separate keys, but in the reference code they're recombined (q+kv|qkv). + This means, for example, that the [1152,1152] weights become [3456,1152] in the state dict. + The issue with this is that the LoRA weights are [128,1152],[1152,128] and become [384,1162],[3456,128] instead. + + This is the best thing I could think of that would fix that, but it's very fragile. + - Check key shape to determine if it needs the fallback logic + - Cut the input into parts based on the shape (undoing the torch.cat) + - Do the matrix multiplication logic + - Recombine them to match the expected shape + """ + for p in patches: + alpha = p[0] + v = p[1] + strength_model = p[2] + if strength_model != 1.0: + weight *= strength_model + + if isinstance(v, list): + v = (self.calculate_weight(v[1:], v[0].clone(), key), ) + + if len(v) == 2: + patch_type = v[0] + v = v[1] + + if patch_type == "lora": + mat1 = comfy.model_management.cast_to_device(v[0], weight.device, torch.float32) + mat2 = comfy.model_management.cast_to_device(v[1], weight.device, torch.float32) + if v[2] is not None: + alpha *= v[2] / mat2.shape[0] + try: + mat1 = mat1.flatten(start_dim=1) + mat2 = mat2.flatten(start_dim=1) + + ch1 = mat1.shape[0] // mat2.shape[1] + ch2 = mat2.shape[0] // mat1.shape[1] + ### Fallback logic for shape mismatch ### + if mat1.shape[0] != mat2.shape[1] and ch1 == ch2 and (mat1.shape[0]/mat2.shape[1])%1 == 0: + mat1 = mat1.chunk(ch1, dim=0) + mat2 = mat2.chunk(ch1, dim=0) + weight += torch.cat( + [alpha * torch.mm(mat1[x], mat2[x]) for x in range(ch1)], + dim=0, + ).reshape(weight.shape).type(weight.dtype) + else: + weight += (alpha * torch.mm(mat1, mat2)).reshape(weight.shape).type(weight.dtype) + except Exception as e: + print("ERROR", key, e) + return weight + + def clone(self): + n = EXM_PixArt_ModelPatcher(self.model, self.load_device, self.offload_device, self.size, self.current_device, weight_inplace_update=self.weight_inplace_update) + n.patches = {} + for k in self.patches: + n.patches[k] = self.patches[k][:] + + n.object_patches = self.object_patches.copy() + n.model_options = copy.deepcopy(self.model_options) + n.model_keys = self.model_keys + return n + +def replace_model_patcher(model): + n = EXM_PixArt_ModelPatcher( + model = model.model, + size = model.size, + load_device = model.load_device, + offload_device = model.offload_device, + current_device = model.current_device, + weight_inplace_update = model.weight_inplace_update, + ) + n.patches = {} + for k in model.patches: + n.patches[k] = model.patches[k][:] + + n.object_patches = model.object_patches.copy() + n.model_options = copy.deepcopy(model.model_options) + n.model_keys = model.model_keys + return n + +def find_peft_alpha(path): + def load_json(json_path): + with open(json_path) as f: + data = json.load(f) + alpha = data.get("lora_alpha") + alpha = alpha or data.get("alpha") + if not alpha: + print(" Found config but `lora_alpha` is missing!") + else: + print(f" Found config at {json_path} [alpha:{alpha}]") + return alpha + + # For some weird reason peft doesn't include the alpha in the actual model + print("PixArt: Warning! This is a PEFT LoRA. Trying to find config...") + files = [ + f"{os.path.splitext(path)[0]}.json", + f"{os.path.splitext(path)[0]}.config.json", + os.path.join(os.path.dirname(path),"adapter_config.json"), + ] + for file in files: + if os.path.isfile(file): + return load_json(file) + + print(" Missing config/alpha! assuming alpha of 8. Consider converting it/adding a config json to it.") + return 8.0 + +def load_pixart_lora(model, lora, lora_path, strength): + k_back = lambda x: x.replace(".lora_up.weight", "") + # need to convert the actual weights for this to work. + if any(True for x in lora.keys() if x.endswith("adaln_single.linear.lora_A.weight")): + lora = convert_lora_state_dict(lora) + alpha = find_peft_alpha(lora_path) + lora.update({f"{k_back(x)}.alpha":torch.tensor(alpha) for x in lora.keys() if "lora_up" in x}) + + key_map = {k_back(x):f"diffusion_model.{k_back(x)}.weight" for x in lora.keys() if "lora_up" in x} # fake + + loaded = comfy.lora.load_lora(lora, key_map) + if model is not None: + # switch to custom model patcher when using LoRAs + if isinstance(model, EXM_PixArt_ModelPatcher): + new_modelpatcher = model.clone() + else: + new_modelpatcher = replace_model_patcher(model) + k = new_modelpatcher.add_patches(loaded, strength) + else: + k = () + new_modelpatcher = None + + k = set(k) + for x in loaded: + if (x not in k): + print("NOT LOADED", x) + + return new_modelpatcher diff --git a/PixArt/nodes.py b/PixArt/nodes.py index 7292e8f..184e6c3 100644 --- a/PixArt/nodes.py +++ b/PixArt/nodes.py @@ -3,7 +3,9 @@ import json import torch import folder_paths +from comfy import utils from .conf import pixart_conf, pixart_res +from .lora import load_pixart_lora from .loader import load_pixart from .sampler import sample_pixart @@ -51,6 +53,45 @@ class PixArtResolutionSelect(): width, height = pixart_res[model][ratio] return (width,height) +class PixArtLoraLoader: + def __init__(self): + self.loaded_lora = None + + @classmethod + def INPUT_TYPES(s): + return { + "required": { + "model": ("MODEL",), + "lora_name": (folder_paths.get_filename_list("loras"), ), + "strength": ("FLOAT", {"default": 1.0, "min": -20.0, "max": 20.0, "step": 0.01}), + } + } + RETURN_TYPES = ("MODEL",) + FUNCTION = "load_lora" + CATEGORY = "ExtraModels/PixArt" + TITLE = "PixArt Load LoRA" + + def load_lora(self, model, lora_name, strength,): + if strength == 0: + return (model) + + lora_path = folder_paths.get_full_path("loras", lora_name) + lora = None + if self.loaded_lora is not None: + if self.loaded_lora[0] == lora_path: + lora = self.loaded_lora[1] + else: + temp = self.loaded_lora + self.loaded_lora = None + del temp + + if lora is None: + lora = utils.load_torch_file(lora_path, safe_load=True) + self.loaded_lora = (lora_path, lora) + + model_lora = load_pixart_lora(model, lora, lora_path, strength,) + return (model_lora,) + class PixArtDPMSampler: """ The sampler from the reference code. @@ -145,6 +186,7 @@ class PixArtT5TextEncode: NODE_CLASS_MAPPINGS = { "PixArtCheckpointLoader" : PixArtCheckpointLoader, "PixArtResolutionSelect" : PixArtResolutionSelect, + "PixArtLoraLoader" : PixArtLoraLoader, "PixArtDPMSampler" : PixArtDPMSampler, "PixArtT5TextEncode" : PixArtT5TextEncode, }