From a0cf329fdaabeacb3a152fec0b12325693f9527b Mon Sep 17 00:00:00 2001 From: Zhaolun Zou Date: Tue, 17 Dec 2024 12:05:15 +0800 Subject: [PATCH 1/3] Added code mapping for DCAE for model: https://huggingface.co/mit-han-lab/dc-ae-f32c32-sana-1.0/blob/main/model.safetensors --- VAE/loader.py | 2 + VAE/models/dcae_key_mapping.py | 350 +++++++++++++++++++++++++++++++++ 2 files changed, 352 insertions(+) create mode 100644 VAE/models/dcae_key_mapping.py diff --git a/VAE/loader.py b/VAE/loader.py index 748e3ab..d3809ed 100644 --- a/VAE/loader.py +++ b/VAE/loader.py @@ -34,6 +34,8 @@ class EXVAE(comfy.sd.VAE): model = MoVQ(model_conf) elif model_conf["type"] == "DCAE": from .models.dcae import DCAE + from .models import dcae_key_mapping + sd = dcae_key_mapping.convert_sd(sd) model = DCAE(**model_conf) else: raise NotImplementedError(f"Unknown VAE type '{model_conf['type']}'") diff --git a/VAE/models/dcae_key_mapping.py b/VAE/models/dcae_key_mapping.py new file mode 100644 index 0000000..52ad998 --- /dev/null +++ b/VAE/models/dcae_key_mapping.py @@ -0,0 +1,350 @@ +def convert_sd(sd, cpu=False): + sd_converted = {} + mapping = get_mapping() + for k, v in sd.items(): + sd_converted[mapping[k]] = v.cpu() if cpu else v + return sd_converted + + +def get_mapping(): + return { + "encoder.project_in.conv.bias": "encoder.project_in.bias", + "encoder.project_in.conv.weight": "encoder.project_in.weight", + "encoder.stages.0.op_list.0.main.conv1.conv.bias": "encoder.stages.0.0.conv1.conv.bias", + "encoder.stages.0.op_list.0.main.conv1.conv.weight": "encoder.stages.0.0.conv1.conv.weight", + "encoder.stages.0.op_list.0.main.conv2.conv.weight": "encoder.stages.0.0.conv2.conv.weight", + "encoder.stages.0.op_list.0.main.conv2.norm.bias": "encoder.stages.0.0.conv2.norm.bias", + "encoder.stages.0.op_list.0.main.conv2.norm.weight": "encoder.stages.0.0.conv2.norm.weight", + "encoder.stages.0.op_list.1.main.conv1.conv.bias": "encoder.stages.0.1.conv1.conv.bias", + "encoder.stages.0.op_list.1.main.conv1.conv.weight": "encoder.stages.0.1.conv1.conv.weight", + "encoder.stages.0.op_list.1.main.conv2.conv.weight": "encoder.stages.0.1.conv2.conv.weight", + "encoder.stages.0.op_list.1.main.conv2.norm.bias": "encoder.stages.0.1.conv2.norm.bias", + "encoder.stages.0.op_list.1.main.conv2.norm.weight": "encoder.stages.0.1.conv2.norm.weight", + "encoder.stages.0.op_list.2.main.conv.bias": "encoder.stages.0.2.main.bias", + "encoder.stages.0.op_list.2.main.conv.weight": "encoder.stages.0.2.main.weight", + "encoder.stages.1.op_list.0.main.conv1.conv.bias": "encoder.stages.1.0.conv1.conv.bias", + "encoder.stages.1.op_list.0.main.conv1.conv.weight": "encoder.stages.1.0.conv1.conv.weight", + "encoder.stages.1.op_list.0.main.conv2.conv.weight": "encoder.stages.1.0.conv2.conv.weight", + "encoder.stages.1.op_list.0.main.conv2.norm.bias": "encoder.stages.1.0.conv2.norm.bias", + "encoder.stages.1.op_list.0.main.conv2.norm.weight": "encoder.stages.1.0.conv2.norm.weight", + "encoder.stages.1.op_list.1.main.conv1.conv.bias": "encoder.stages.1.1.conv1.conv.bias", + "encoder.stages.1.op_list.1.main.conv1.conv.weight": "encoder.stages.1.1.conv1.conv.weight", + "encoder.stages.1.op_list.1.main.conv2.conv.weight": "encoder.stages.1.1.conv2.conv.weight", + "encoder.stages.1.op_list.1.main.conv2.norm.bias": "encoder.stages.1.1.conv2.norm.bias", + "encoder.stages.1.op_list.1.main.conv2.norm.weight": "encoder.stages.1.1.conv2.norm.weight", + "encoder.stages.1.op_list.2.main.conv.bias": "encoder.stages.1.2.main.bias", + "encoder.stages.1.op_list.2.main.conv.weight": "encoder.stages.1.2.main.weight", + "encoder.stages.2.op_list.0.main.conv1.conv.bias": "encoder.stages.2.0.conv1.conv.bias", + "encoder.stages.2.op_list.0.main.conv1.conv.weight": "encoder.stages.2.0.conv1.conv.weight", + "encoder.stages.2.op_list.0.main.conv2.conv.weight": "encoder.stages.2.0.conv2.conv.weight", + "encoder.stages.2.op_list.0.main.conv2.norm.bias": "encoder.stages.2.0.conv2.norm.bias", + "encoder.stages.2.op_list.0.main.conv2.norm.weight": "encoder.stages.2.0.conv2.norm.weight", + "encoder.stages.2.op_list.1.main.conv1.conv.bias": "encoder.stages.2.1.conv1.conv.bias", + "encoder.stages.2.op_list.1.main.conv1.conv.weight": "encoder.stages.2.1.conv1.conv.weight", + "encoder.stages.2.op_list.1.main.conv2.conv.weight": "encoder.stages.2.1.conv2.conv.weight", + "encoder.stages.2.op_list.1.main.conv2.norm.bias": "encoder.stages.2.1.conv2.norm.bias", + "encoder.stages.2.op_list.1.main.conv2.norm.weight": "encoder.stages.2.1.conv2.norm.weight", + "encoder.stages.2.op_list.2.main.conv.bias": "encoder.stages.2.2.main.bias", + "encoder.stages.2.op_list.2.main.conv.weight": "encoder.stages.2.2.main.weight", + "encoder.stages.3.op_list.0.context_module.main.aggreg.0.0.weight": "encoder.stages.3.0.context_module.aggreg.0.0.weight", + "encoder.stages.3.op_list.0.context_module.main.aggreg.0.1.weight": "encoder.stages.3.0.context_module.aggreg.0.1.weight", + "encoder.stages.3.op_list.0.context_module.main.proj.conv.weight": "encoder.stages.3.0.context_module.proj.0.weight", + "encoder.stages.3.op_list.0.context_module.main.proj.norm.bias": "encoder.stages.3.0.context_module.proj.1.bias", + "encoder.stages.3.op_list.0.context_module.main.proj.norm.weight": "encoder.stages.3.0.context_module.proj.1.weight", + "encoder.stages.3.op_list.0.context_module.main.qkv.conv.weight": "encoder.stages.3.0.context_module.qkv.0.weight", + "encoder.stages.3.op_list.0.local_module.main.depth_conv.conv.bias": "encoder.stages.3.0.local_module.depth_conv.conv.bias", + "encoder.stages.3.op_list.0.local_module.main.depth_conv.conv.weight": "encoder.stages.3.0.local_module.depth_conv.conv.weight", + "encoder.stages.3.op_list.0.local_module.main.inverted_conv.conv.bias": "encoder.stages.3.0.local_module.inverted_conv.conv.bias", + "encoder.stages.3.op_list.0.local_module.main.inverted_conv.conv.weight": "encoder.stages.3.0.local_module.inverted_conv.conv.weight", + "encoder.stages.3.op_list.0.local_module.main.point_conv.conv.weight": "encoder.stages.3.0.local_module.point_conv.conv.weight", + "encoder.stages.3.op_list.0.local_module.main.point_conv.norm.bias": "encoder.stages.3.0.local_module.point_conv.norm.bias", + "encoder.stages.3.op_list.0.local_module.main.point_conv.norm.weight": "encoder.stages.3.0.local_module.point_conv.norm.weight", + "encoder.stages.3.op_list.1.context_module.main.aggreg.0.0.weight": "encoder.stages.3.1.context_module.aggreg.0.0.weight", + "encoder.stages.3.op_list.1.context_module.main.aggreg.0.1.weight": "encoder.stages.3.1.context_module.aggreg.0.1.weight", + "encoder.stages.3.op_list.1.context_module.main.proj.conv.weight": "encoder.stages.3.1.context_module.proj.0.weight", + "encoder.stages.3.op_list.1.context_module.main.proj.norm.bias": "encoder.stages.3.1.context_module.proj.1.bias", + "encoder.stages.3.op_list.1.context_module.main.proj.norm.weight": "encoder.stages.3.1.context_module.proj.1.weight", + "encoder.stages.3.op_list.1.context_module.main.qkv.conv.weight": "encoder.stages.3.1.context_module.qkv.0.weight", + "encoder.stages.3.op_list.1.local_module.main.depth_conv.conv.bias": "encoder.stages.3.1.local_module.depth_conv.conv.bias", + "encoder.stages.3.op_list.1.local_module.main.depth_conv.conv.weight": "encoder.stages.3.1.local_module.depth_conv.conv.weight", + "encoder.stages.3.op_list.1.local_module.main.inverted_conv.conv.bias": "encoder.stages.3.1.local_module.inverted_conv.conv.bias", + "encoder.stages.3.op_list.1.local_module.main.inverted_conv.conv.weight": "encoder.stages.3.1.local_module.inverted_conv.conv.weight", + "encoder.stages.3.op_list.1.local_module.main.point_conv.conv.weight": "encoder.stages.3.1.local_module.point_conv.conv.weight", + "encoder.stages.3.op_list.1.local_module.main.point_conv.norm.bias": "encoder.stages.3.1.local_module.point_conv.norm.bias", + "encoder.stages.3.op_list.1.local_module.main.point_conv.norm.weight": "encoder.stages.3.1.local_module.point_conv.norm.weight", + "encoder.stages.3.op_list.2.context_module.main.aggreg.0.0.weight": "encoder.stages.3.2.context_module.aggreg.0.0.weight", + "encoder.stages.3.op_list.2.context_module.main.aggreg.0.1.weight": "encoder.stages.3.2.context_module.aggreg.0.1.weight", + "encoder.stages.3.op_list.2.context_module.main.proj.conv.weight": "encoder.stages.3.2.context_module.proj.0.weight", + "encoder.stages.3.op_list.2.context_module.main.proj.norm.bias": "encoder.stages.3.2.context_module.proj.1.bias", + "encoder.stages.3.op_list.2.context_module.main.proj.norm.weight": "encoder.stages.3.2.context_module.proj.1.weight", + "encoder.stages.3.op_list.2.context_module.main.qkv.conv.weight": "encoder.stages.3.2.context_module.qkv.0.weight", + "encoder.stages.3.op_list.2.local_module.main.depth_conv.conv.bias": "encoder.stages.3.2.local_module.depth_conv.conv.bias", + "encoder.stages.3.op_list.2.local_module.main.depth_conv.conv.weight": "encoder.stages.3.2.local_module.depth_conv.conv.weight", + "encoder.stages.3.op_list.2.local_module.main.inverted_conv.conv.bias": "encoder.stages.3.2.local_module.inverted_conv.conv.bias", + "encoder.stages.3.op_list.2.local_module.main.inverted_conv.conv.weight": "encoder.stages.3.2.local_module.inverted_conv.conv.weight", + "encoder.stages.3.op_list.2.local_module.main.point_conv.conv.weight": "encoder.stages.3.2.local_module.point_conv.conv.weight", + "encoder.stages.3.op_list.2.local_module.main.point_conv.norm.bias": "encoder.stages.3.2.local_module.point_conv.norm.bias", + "encoder.stages.3.op_list.2.local_module.main.point_conv.norm.weight": "encoder.stages.3.2.local_module.point_conv.norm.weight", + "encoder.stages.3.op_list.3.main.conv.bias": "encoder.stages.3.3.main.bias", + "encoder.stages.3.op_list.3.main.conv.weight": "encoder.stages.3.3.main.weight", + "encoder.stages.4.op_list.0.context_module.main.aggreg.0.0.weight": "encoder.stages.4.0.context_module.aggreg.0.0.weight", + "encoder.stages.4.op_list.0.context_module.main.aggreg.0.1.weight": "encoder.stages.4.0.context_module.aggreg.0.1.weight", + "encoder.stages.4.op_list.0.context_module.main.proj.conv.weight": "encoder.stages.4.0.context_module.proj.0.weight", + "encoder.stages.4.op_list.0.context_module.main.proj.norm.bias": "encoder.stages.4.0.context_module.proj.1.bias", + "encoder.stages.4.op_list.0.context_module.main.proj.norm.weight": "encoder.stages.4.0.context_module.proj.1.weight", + "encoder.stages.4.op_list.0.context_module.main.qkv.conv.weight": "encoder.stages.4.0.context_module.qkv.0.weight", + "encoder.stages.4.op_list.0.local_module.main.depth_conv.conv.bias": "encoder.stages.4.0.local_module.depth_conv.conv.bias", + "encoder.stages.4.op_list.0.local_module.main.depth_conv.conv.weight": "encoder.stages.4.0.local_module.depth_conv.conv.weight", + "encoder.stages.4.op_list.0.local_module.main.inverted_conv.conv.bias": "encoder.stages.4.0.local_module.inverted_conv.conv.bias", + "encoder.stages.4.op_list.0.local_module.main.inverted_conv.conv.weight": "encoder.stages.4.0.local_module.inverted_conv.conv.weight", + "encoder.stages.4.op_list.0.local_module.main.point_conv.conv.weight": "encoder.stages.4.0.local_module.point_conv.conv.weight", + "encoder.stages.4.op_list.0.local_module.main.point_conv.norm.bias": "encoder.stages.4.0.local_module.point_conv.norm.bias", + "encoder.stages.4.op_list.0.local_module.main.point_conv.norm.weight": "encoder.stages.4.0.local_module.point_conv.norm.weight", + "encoder.stages.4.op_list.1.context_module.main.aggreg.0.0.weight": "encoder.stages.4.1.context_module.aggreg.0.0.weight", + "encoder.stages.4.op_list.1.context_module.main.aggreg.0.1.weight": "encoder.stages.4.1.context_module.aggreg.0.1.weight", + "encoder.stages.4.op_list.1.context_module.main.proj.conv.weight": "encoder.stages.4.1.context_module.proj.0.weight", + "encoder.stages.4.op_list.1.context_module.main.proj.norm.bias": "encoder.stages.4.1.context_module.proj.1.bias", + "encoder.stages.4.op_list.1.context_module.main.proj.norm.weight": "encoder.stages.4.1.context_module.proj.1.weight", + "encoder.stages.4.op_list.1.context_module.main.qkv.conv.weight": "encoder.stages.4.1.context_module.qkv.0.weight", + "encoder.stages.4.op_list.1.local_module.main.depth_conv.conv.bias": "encoder.stages.4.1.local_module.depth_conv.conv.bias", + "encoder.stages.4.op_list.1.local_module.main.depth_conv.conv.weight": "encoder.stages.4.1.local_module.depth_conv.conv.weight", + "encoder.stages.4.op_list.1.local_module.main.inverted_conv.conv.bias": "encoder.stages.4.1.local_module.inverted_conv.conv.bias", + "encoder.stages.4.op_list.1.local_module.main.inverted_conv.conv.weight": "encoder.stages.4.1.local_module.inverted_conv.conv.weight", + "encoder.stages.4.op_list.1.local_module.main.point_conv.conv.weight": "encoder.stages.4.1.local_module.point_conv.conv.weight", + "encoder.stages.4.op_list.1.local_module.main.point_conv.norm.bias": "encoder.stages.4.1.local_module.point_conv.norm.bias", + "encoder.stages.4.op_list.1.local_module.main.point_conv.norm.weight": "encoder.stages.4.1.local_module.point_conv.norm.weight", + "encoder.stages.4.op_list.2.context_module.main.aggreg.0.0.weight": "encoder.stages.4.2.context_module.aggreg.0.0.weight", + "encoder.stages.4.op_list.2.context_module.main.aggreg.0.1.weight": "encoder.stages.4.2.context_module.aggreg.0.1.weight", + "encoder.stages.4.op_list.2.context_module.main.proj.conv.weight": "encoder.stages.4.2.context_module.proj.0.weight", + "encoder.stages.4.op_list.2.context_module.main.proj.norm.bias": "encoder.stages.4.2.context_module.proj.1.bias", + "encoder.stages.4.op_list.2.context_module.main.proj.norm.weight": "encoder.stages.4.2.context_module.proj.1.weight", + "encoder.stages.4.op_list.2.context_module.main.qkv.conv.weight": "encoder.stages.4.2.context_module.qkv.0.weight", + "encoder.stages.4.op_list.2.local_module.main.depth_conv.conv.bias": "encoder.stages.4.2.local_module.depth_conv.conv.bias", + "encoder.stages.4.op_list.2.local_module.main.depth_conv.conv.weight": "encoder.stages.4.2.local_module.depth_conv.conv.weight", + "encoder.stages.4.op_list.2.local_module.main.inverted_conv.conv.bias": "encoder.stages.4.2.local_module.inverted_conv.conv.bias", + "encoder.stages.4.op_list.2.local_module.main.inverted_conv.conv.weight": "encoder.stages.4.2.local_module.inverted_conv.conv.weight", + "encoder.stages.4.op_list.2.local_module.main.point_conv.conv.weight": "encoder.stages.4.2.local_module.point_conv.conv.weight", + "encoder.stages.4.op_list.2.local_module.main.point_conv.norm.bias": "encoder.stages.4.2.local_module.point_conv.norm.bias", + "encoder.stages.4.op_list.2.local_module.main.point_conv.norm.weight": "encoder.stages.4.2.local_module.point_conv.norm.weight", + "encoder.stages.4.op_list.3.main.conv.bias": "encoder.stages.4.3.main.bias", + "encoder.stages.4.op_list.3.main.conv.weight": "encoder.stages.4.3.main.weight", + "encoder.stages.5.op_list.0.context_module.main.aggreg.0.0.weight": "encoder.stages.5.0.context_module.aggreg.0.0.weight", + "encoder.stages.5.op_list.0.context_module.main.aggreg.0.1.weight": "encoder.stages.5.0.context_module.aggreg.0.1.weight", + "encoder.stages.5.op_list.0.context_module.main.proj.conv.weight": "encoder.stages.5.0.context_module.proj.0.weight", + "encoder.stages.5.op_list.0.context_module.main.proj.norm.bias": "encoder.stages.5.0.context_module.proj.1.bias", + "encoder.stages.5.op_list.0.context_module.main.proj.norm.weight": "encoder.stages.5.0.context_module.proj.1.weight", + "encoder.stages.5.op_list.0.context_module.main.qkv.conv.weight": "encoder.stages.5.0.context_module.qkv.0.weight", + "encoder.stages.5.op_list.0.local_module.main.depth_conv.conv.bias": "encoder.stages.5.0.local_module.depth_conv.conv.bias", + "encoder.stages.5.op_list.0.local_module.main.depth_conv.conv.weight": "encoder.stages.5.0.local_module.depth_conv.conv.weight", + "encoder.stages.5.op_list.0.local_module.main.inverted_conv.conv.bias": "encoder.stages.5.0.local_module.inverted_conv.conv.bias", + "encoder.stages.5.op_list.0.local_module.main.inverted_conv.conv.weight": "encoder.stages.5.0.local_module.inverted_conv.conv.weight", + "encoder.stages.5.op_list.0.local_module.main.point_conv.conv.weight": "encoder.stages.5.0.local_module.point_conv.conv.weight", + "encoder.stages.5.op_list.0.local_module.main.point_conv.norm.bias": "encoder.stages.5.0.local_module.point_conv.norm.bias", + "encoder.stages.5.op_list.0.local_module.main.point_conv.norm.weight": "encoder.stages.5.0.local_module.point_conv.norm.weight", + "encoder.stages.5.op_list.1.context_module.main.aggreg.0.0.weight": "encoder.stages.5.1.context_module.aggreg.0.0.weight", + "encoder.stages.5.op_list.1.context_module.main.aggreg.0.1.weight": "encoder.stages.5.1.context_module.aggreg.0.1.weight", + "encoder.stages.5.op_list.1.context_module.main.proj.conv.weight": "encoder.stages.5.1.context_module.proj.0.weight", + "encoder.stages.5.op_list.1.context_module.main.proj.norm.bias": "encoder.stages.5.1.context_module.proj.1.bias", + "encoder.stages.5.op_list.1.context_module.main.proj.norm.weight": "encoder.stages.5.1.context_module.proj.1.weight", + "encoder.stages.5.op_list.1.context_module.main.qkv.conv.weight": "encoder.stages.5.1.context_module.qkv.0.weight", + "encoder.stages.5.op_list.1.local_module.main.depth_conv.conv.bias": "encoder.stages.5.1.local_module.depth_conv.conv.bias", + "encoder.stages.5.op_list.1.local_module.main.depth_conv.conv.weight": "encoder.stages.5.1.local_module.depth_conv.conv.weight", + "encoder.stages.5.op_list.1.local_module.main.inverted_conv.conv.bias": "encoder.stages.5.1.local_module.inverted_conv.conv.bias", + "encoder.stages.5.op_list.1.local_module.main.inverted_conv.conv.weight": "encoder.stages.5.1.local_module.inverted_conv.conv.weight", + "encoder.stages.5.op_list.1.local_module.main.point_conv.conv.weight": "encoder.stages.5.1.local_module.point_conv.conv.weight", + "encoder.stages.5.op_list.1.local_module.main.point_conv.norm.bias": "encoder.stages.5.1.local_module.point_conv.norm.bias", + "encoder.stages.5.op_list.1.local_module.main.point_conv.norm.weight": "encoder.stages.5.1.local_module.point_conv.norm.weight", + "encoder.stages.5.op_list.2.context_module.main.aggreg.0.0.weight": "encoder.stages.5.2.context_module.aggreg.0.0.weight", + "encoder.stages.5.op_list.2.context_module.main.aggreg.0.1.weight": "encoder.stages.5.2.context_module.aggreg.0.1.weight", + "encoder.stages.5.op_list.2.context_module.main.proj.conv.weight": "encoder.stages.5.2.context_module.proj.0.weight", + "encoder.stages.5.op_list.2.context_module.main.proj.norm.bias": "encoder.stages.5.2.context_module.proj.1.bias", + "encoder.stages.5.op_list.2.context_module.main.proj.norm.weight": "encoder.stages.5.2.context_module.proj.1.weight", + "encoder.stages.5.op_list.2.context_module.main.qkv.conv.weight": "encoder.stages.5.2.context_module.qkv.0.weight", + "encoder.stages.5.op_list.2.local_module.main.depth_conv.conv.bias": "encoder.stages.5.2.local_module.depth_conv.conv.bias", + "encoder.stages.5.op_list.2.local_module.main.depth_conv.conv.weight": "encoder.stages.5.2.local_module.depth_conv.conv.weight", + "encoder.stages.5.op_list.2.local_module.main.inverted_conv.conv.bias": "encoder.stages.5.2.local_module.inverted_conv.conv.bias", + "encoder.stages.5.op_list.2.local_module.main.inverted_conv.conv.weight": "encoder.stages.5.2.local_module.inverted_conv.conv.weight", + "encoder.stages.5.op_list.2.local_module.main.point_conv.conv.weight": "encoder.stages.5.2.local_module.point_conv.conv.weight", + "encoder.stages.5.op_list.2.local_module.main.point_conv.norm.bias": "encoder.stages.5.2.local_module.point_conv.norm.bias", + "encoder.stages.5.op_list.2.local_module.main.point_conv.norm.weight": "encoder.stages.5.2.local_module.point_conv.norm.weight", + "encoder.project_out.main.op_list.0.conv.bias": "encoder.project_out.main.0.conv.bias", + "encoder.project_out.main.op_list.0.conv.weight": "encoder.project_out.main.0.conv.weight", + "decoder.project_in.main.conv.bias": "decoder.project_in.main.conv.bias", + "decoder.project_in.main.conv.weight": "decoder.project_in.main.conv.weight", + "decoder.stages.0.op_list.0.main.conv.conv.bias": "decoder.stages.0.0.main.conv.bias", + "decoder.stages.0.op_list.0.main.conv.conv.weight": "decoder.stages.0.0.main.conv.weight", + "decoder.stages.0.op_list.1.main.conv1.conv.bias": "decoder.stages.0.1.conv1.conv.bias", + "decoder.stages.0.op_list.1.main.conv1.conv.weight": "decoder.stages.0.1.conv1.conv.weight", + "decoder.stages.0.op_list.1.main.conv2.conv.weight": "decoder.stages.0.1.conv2.conv.weight", + "decoder.stages.0.op_list.1.main.conv2.norm.bias": "decoder.stages.0.1.conv2.norm.bias", + "decoder.stages.0.op_list.1.main.conv2.norm.weight": "decoder.stages.0.1.conv2.norm.weight", + "decoder.stages.0.op_list.2.main.conv1.conv.bias": "decoder.stages.0.2.conv1.conv.bias", + "decoder.stages.0.op_list.2.main.conv1.conv.weight": "decoder.stages.0.2.conv1.conv.weight", + "decoder.stages.0.op_list.2.main.conv2.conv.weight": "decoder.stages.0.2.conv2.conv.weight", + "decoder.stages.0.op_list.2.main.conv2.norm.bias": "decoder.stages.0.2.conv2.norm.bias", + "decoder.stages.0.op_list.2.main.conv2.norm.weight": "decoder.stages.0.2.conv2.norm.weight", + "decoder.stages.0.op_list.3.main.conv1.conv.bias": "decoder.stages.0.3.conv1.conv.bias", + "decoder.stages.0.op_list.3.main.conv1.conv.weight": "decoder.stages.0.3.conv1.conv.weight", + "decoder.stages.0.op_list.3.main.conv2.conv.weight": "decoder.stages.0.3.conv2.conv.weight", + "decoder.stages.0.op_list.3.main.conv2.norm.bias": "decoder.stages.0.3.conv2.norm.bias", + "decoder.stages.0.op_list.3.main.conv2.norm.weight": "decoder.stages.0.3.conv2.norm.weight", + "decoder.stages.1.op_list.0.main.conv.conv.bias": "decoder.stages.1.0.main.conv.bias", + "decoder.stages.1.op_list.0.main.conv.conv.weight": "decoder.stages.1.0.main.conv.weight", + "decoder.stages.1.op_list.1.main.conv1.conv.bias": "decoder.stages.1.1.conv1.conv.bias", + "decoder.stages.1.op_list.1.main.conv1.conv.weight": "decoder.stages.1.1.conv1.conv.weight", + "decoder.stages.1.op_list.1.main.conv2.conv.weight": "decoder.stages.1.1.conv2.conv.weight", + "decoder.stages.1.op_list.1.main.conv2.norm.bias": "decoder.stages.1.1.conv2.norm.bias", + "decoder.stages.1.op_list.1.main.conv2.norm.weight": "decoder.stages.1.1.conv2.norm.weight", + "decoder.stages.1.op_list.2.main.conv1.conv.bias": "decoder.stages.1.2.conv1.conv.bias", + "decoder.stages.1.op_list.2.main.conv1.conv.weight": "decoder.stages.1.2.conv1.conv.weight", + "decoder.stages.1.op_list.2.main.conv2.conv.weight": "decoder.stages.1.2.conv2.conv.weight", + "decoder.stages.1.op_list.2.main.conv2.norm.bias": "decoder.stages.1.2.conv2.norm.bias", + "decoder.stages.1.op_list.2.main.conv2.norm.weight": "decoder.stages.1.2.conv2.norm.weight", + "decoder.stages.1.op_list.3.main.conv1.conv.bias": "decoder.stages.1.3.conv1.conv.bias", + "decoder.stages.1.op_list.3.main.conv1.conv.weight": "decoder.stages.1.3.conv1.conv.weight", + "decoder.stages.1.op_list.3.main.conv2.conv.weight": "decoder.stages.1.3.conv2.conv.weight", + "decoder.stages.1.op_list.3.main.conv2.norm.bias": "decoder.stages.1.3.conv2.norm.bias", + "decoder.stages.1.op_list.3.main.conv2.norm.weight": "decoder.stages.1.3.conv2.norm.weight", + "decoder.stages.2.op_list.0.main.conv.conv.bias": "decoder.stages.2.0.main.conv.bias", + "decoder.stages.2.op_list.0.main.conv.conv.weight": "decoder.stages.2.0.main.conv.weight", + "decoder.stages.2.op_list.1.main.conv1.conv.bias": "decoder.stages.2.1.conv1.conv.bias", + "decoder.stages.2.op_list.1.main.conv1.conv.weight": "decoder.stages.2.1.conv1.conv.weight", + "decoder.stages.2.op_list.1.main.conv2.conv.weight": "decoder.stages.2.1.conv2.conv.weight", + "decoder.stages.2.op_list.1.main.conv2.norm.bias": "decoder.stages.2.1.conv2.norm.bias", + "decoder.stages.2.op_list.1.main.conv2.norm.weight": "decoder.stages.2.1.conv2.norm.weight", + "decoder.stages.2.op_list.2.main.conv1.conv.bias": "decoder.stages.2.2.conv1.conv.bias", + "decoder.stages.2.op_list.2.main.conv1.conv.weight": "decoder.stages.2.2.conv1.conv.weight", + "decoder.stages.2.op_list.2.main.conv2.conv.weight": "decoder.stages.2.2.conv2.conv.weight", + "decoder.stages.2.op_list.2.main.conv2.norm.bias": "decoder.stages.2.2.conv2.norm.bias", + "decoder.stages.2.op_list.2.main.conv2.norm.weight": "decoder.stages.2.2.conv2.norm.weight", + "decoder.stages.2.op_list.3.main.conv1.conv.bias": "decoder.stages.2.3.conv1.conv.bias", + "decoder.stages.2.op_list.3.main.conv1.conv.weight": "decoder.stages.2.3.conv1.conv.weight", + "decoder.stages.2.op_list.3.main.conv2.conv.weight": "decoder.stages.2.3.conv2.conv.weight", + "decoder.stages.2.op_list.3.main.conv2.norm.bias": "decoder.stages.2.3.conv2.norm.bias", + "decoder.stages.2.op_list.3.main.conv2.norm.weight": "decoder.stages.2.3.conv2.norm.weight", + "decoder.stages.3.op_list.0.main.conv.conv.bias": "decoder.stages.3.0.main.conv.bias", + "decoder.stages.3.op_list.0.main.conv.conv.weight": "decoder.stages.3.0.main.conv.weight", + "decoder.stages.3.op_list.1.context_module.main.aggreg.0.0.weight": "decoder.stages.3.1.context_module.aggreg.0.0.weight", + "decoder.stages.3.op_list.1.context_module.main.aggreg.0.1.weight": "decoder.stages.3.1.context_module.aggreg.0.1.weight", + "decoder.stages.3.op_list.1.context_module.main.proj.conv.weight": "decoder.stages.3.1.context_module.proj.0.weight", + "decoder.stages.3.op_list.1.context_module.main.proj.norm.bias": "decoder.stages.3.1.context_module.proj.1.bias", + "decoder.stages.3.op_list.1.context_module.main.proj.norm.weight": "decoder.stages.3.1.context_module.proj.1.weight", + "decoder.stages.3.op_list.1.context_module.main.qkv.conv.weight": "decoder.stages.3.1.context_module.qkv.0.weight", + "decoder.stages.3.op_list.1.local_module.main.depth_conv.conv.bias": "decoder.stages.3.1.local_module.depth_conv.conv.bias", + "decoder.stages.3.op_list.1.local_module.main.depth_conv.conv.weight": "decoder.stages.3.1.local_module.depth_conv.conv.weight", + "decoder.stages.3.op_list.1.local_module.main.inverted_conv.conv.bias": "decoder.stages.3.1.local_module.inverted_conv.conv.bias", + "decoder.stages.3.op_list.1.local_module.main.inverted_conv.conv.weight": "decoder.stages.3.1.local_module.inverted_conv.conv.weight", + "decoder.stages.3.op_list.1.local_module.main.point_conv.conv.weight": "decoder.stages.3.1.local_module.point_conv.conv.weight", + "decoder.stages.3.op_list.1.local_module.main.point_conv.norm.bias": "decoder.stages.3.1.local_module.point_conv.norm.bias", + "decoder.stages.3.op_list.1.local_module.main.point_conv.norm.weight": "decoder.stages.3.1.local_module.point_conv.norm.weight", + "decoder.stages.3.op_list.2.context_module.main.aggreg.0.0.weight": "decoder.stages.3.2.context_module.aggreg.0.0.weight", + "decoder.stages.3.op_list.2.context_module.main.aggreg.0.1.weight": "decoder.stages.3.2.context_module.aggreg.0.1.weight", + "decoder.stages.3.op_list.2.context_module.main.proj.conv.weight": "decoder.stages.3.2.context_module.proj.0.weight", + "decoder.stages.3.op_list.2.context_module.main.proj.norm.bias": "decoder.stages.3.2.context_module.proj.1.bias", + "decoder.stages.3.op_list.2.context_module.main.proj.norm.weight": "decoder.stages.3.2.context_module.proj.1.weight", + "decoder.stages.3.op_list.2.context_module.main.qkv.conv.weight": "decoder.stages.3.2.context_module.qkv.0.weight", + "decoder.stages.3.op_list.2.local_module.main.depth_conv.conv.bias": "decoder.stages.3.2.local_module.depth_conv.conv.bias", + "decoder.stages.3.op_list.2.local_module.main.depth_conv.conv.weight": "decoder.stages.3.2.local_module.depth_conv.conv.weight", + "decoder.stages.3.op_list.2.local_module.main.inverted_conv.conv.bias": "decoder.stages.3.2.local_module.inverted_conv.conv.bias", + "decoder.stages.3.op_list.2.local_module.main.inverted_conv.conv.weight": "decoder.stages.3.2.local_module.inverted_conv.conv.weight", + "decoder.stages.3.op_list.2.local_module.main.point_conv.conv.weight": "decoder.stages.3.2.local_module.point_conv.conv.weight", + "decoder.stages.3.op_list.2.local_module.main.point_conv.norm.bias": "decoder.stages.3.2.local_module.point_conv.norm.bias", + "decoder.stages.3.op_list.2.local_module.main.point_conv.norm.weight": "decoder.stages.3.2.local_module.point_conv.norm.weight", + "decoder.stages.3.op_list.3.context_module.main.aggreg.0.0.weight": "decoder.stages.3.3.context_module.aggreg.0.0.weight", + "decoder.stages.3.op_list.3.context_module.main.aggreg.0.1.weight": "decoder.stages.3.3.context_module.aggreg.0.1.weight", + "decoder.stages.3.op_list.3.context_module.main.proj.conv.weight": "decoder.stages.3.3.context_module.proj.0.weight", + "decoder.stages.3.op_list.3.context_module.main.proj.norm.bias": "decoder.stages.3.3.context_module.proj.1.bias", + "decoder.stages.3.op_list.3.context_module.main.proj.norm.weight": "decoder.stages.3.3.context_module.proj.1.weight", + "decoder.stages.3.op_list.3.context_module.main.qkv.conv.weight": "decoder.stages.3.3.context_module.qkv.0.weight", + "decoder.stages.3.op_list.3.local_module.main.depth_conv.conv.bias": "decoder.stages.3.3.local_module.depth_conv.conv.bias", + "decoder.stages.3.op_list.3.local_module.main.depth_conv.conv.weight": "decoder.stages.3.3.local_module.depth_conv.conv.weight", + "decoder.stages.3.op_list.3.local_module.main.inverted_conv.conv.bias": "decoder.stages.3.3.local_module.inverted_conv.conv.bias", + "decoder.stages.3.op_list.3.local_module.main.inverted_conv.conv.weight": "decoder.stages.3.3.local_module.inverted_conv.conv.weight", + "decoder.stages.3.op_list.3.local_module.main.point_conv.conv.weight": "decoder.stages.3.3.local_module.point_conv.conv.weight", + "decoder.stages.3.op_list.3.local_module.main.point_conv.norm.bias": "decoder.stages.3.3.local_module.point_conv.norm.bias", + "decoder.stages.3.op_list.3.local_module.main.point_conv.norm.weight": "decoder.stages.3.3.local_module.point_conv.norm.weight", + "decoder.stages.4.op_list.0.main.conv.conv.bias": "decoder.stages.4.0.main.conv.bias", + "decoder.stages.4.op_list.0.main.conv.conv.weight": "decoder.stages.4.0.main.conv.weight", + "decoder.stages.4.op_list.1.context_module.main.aggreg.0.0.weight": "decoder.stages.4.1.context_module.aggreg.0.0.weight", + "decoder.stages.4.op_list.1.context_module.main.aggreg.0.1.weight": "decoder.stages.4.1.context_module.aggreg.0.1.weight", + "decoder.stages.4.op_list.1.context_module.main.proj.conv.weight": "decoder.stages.4.1.context_module.proj.0.weight", + "decoder.stages.4.op_list.1.context_module.main.proj.norm.bias": "decoder.stages.4.1.context_module.proj.1.bias", + "decoder.stages.4.op_list.1.context_module.main.proj.norm.weight": "decoder.stages.4.1.context_module.proj.1.weight", + "decoder.stages.4.op_list.1.context_module.main.qkv.conv.weight": "decoder.stages.4.1.context_module.qkv.0.weight", + "decoder.stages.4.op_list.1.local_module.main.depth_conv.conv.bias": "decoder.stages.4.1.local_module.depth_conv.conv.bias", + "decoder.stages.4.op_list.1.local_module.main.depth_conv.conv.weight": "decoder.stages.4.1.local_module.depth_conv.conv.weight", + "decoder.stages.4.op_list.1.local_module.main.inverted_conv.conv.bias": "decoder.stages.4.1.local_module.inverted_conv.conv.bias", + "decoder.stages.4.op_list.1.local_module.main.inverted_conv.conv.weight": "decoder.stages.4.1.local_module.inverted_conv.conv.weight", + "decoder.stages.4.op_list.1.local_module.main.point_conv.conv.weight": "decoder.stages.4.1.local_module.point_conv.conv.weight", + "decoder.stages.4.op_list.1.local_module.main.point_conv.norm.bias": "decoder.stages.4.1.local_module.point_conv.norm.bias", + "decoder.stages.4.op_list.1.local_module.main.point_conv.norm.weight": "decoder.stages.4.1.local_module.point_conv.norm.weight", + "decoder.stages.4.op_list.2.context_module.main.aggreg.0.0.weight": "decoder.stages.4.2.context_module.aggreg.0.0.weight", + "decoder.stages.4.op_list.2.context_module.main.aggreg.0.1.weight": "decoder.stages.4.2.context_module.aggreg.0.1.weight", + "decoder.stages.4.op_list.2.context_module.main.proj.conv.weight": "decoder.stages.4.2.context_module.proj.0.weight", + "decoder.stages.4.op_list.2.context_module.main.proj.norm.bias": "decoder.stages.4.2.context_module.proj.1.bias", + "decoder.stages.4.op_list.2.context_module.main.proj.norm.weight": "decoder.stages.4.2.context_module.proj.1.weight", + "decoder.stages.4.op_list.2.context_module.main.qkv.conv.weight": "decoder.stages.4.2.context_module.qkv.0.weight", + "decoder.stages.4.op_list.2.local_module.main.depth_conv.conv.bias": "decoder.stages.4.2.local_module.depth_conv.conv.bias", + "decoder.stages.4.op_list.2.local_module.main.depth_conv.conv.weight": "decoder.stages.4.2.local_module.depth_conv.conv.weight", + "decoder.stages.4.op_list.2.local_module.main.inverted_conv.conv.bias": "decoder.stages.4.2.local_module.inverted_conv.conv.bias", + "decoder.stages.4.op_list.2.local_module.main.inverted_conv.conv.weight": "decoder.stages.4.2.local_module.inverted_conv.conv.weight", + "decoder.stages.4.op_list.2.local_module.main.point_conv.conv.weight": "decoder.stages.4.2.local_module.point_conv.conv.weight", + "decoder.stages.4.op_list.2.local_module.main.point_conv.norm.bias": "decoder.stages.4.2.local_module.point_conv.norm.bias", + "decoder.stages.4.op_list.2.local_module.main.point_conv.norm.weight": "decoder.stages.4.2.local_module.point_conv.norm.weight", + "decoder.stages.4.op_list.3.context_module.main.aggreg.0.0.weight": "decoder.stages.4.3.context_module.aggreg.0.0.weight", + "decoder.stages.4.op_list.3.context_module.main.aggreg.0.1.weight": "decoder.stages.4.3.context_module.aggreg.0.1.weight", + "decoder.stages.4.op_list.3.context_module.main.proj.conv.weight": "decoder.stages.4.3.context_module.proj.0.weight", + "decoder.stages.4.op_list.3.context_module.main.proj.norm.bias": "decoder.stages.4.3.context_module.proj.1.bias", + "decoder.stages.4.op_list.3.context_module.main.proj.norm.weight": "decoder.stages.4.3.context_module.proj.1.weight", + "decoder.stages.4.op_list.3.context_module.main.qkv.conv.weight": "decoder.stages.4.3.context_module.qkv.0.weight", + "decoder.stages.4.op_list.3.local_module.main.depth_conv.conv.bias": "decoder.stages.4.3.local_module.depth_conv.conv.bias", + "decoder.stages.4.op_list.3.local_module.main.depth_conv.conv.weight": "decoder.stages.4.3.local_module.depth_conv.conv.weight", + "decoder.stages.4.op_list.3.local_module.main.inverted_conv.conv.bias": "decoder.stages.4.3.local_module.inverted_conv.conv.bias", + "decoder.stages.4.op_list.3.local_module.main.inverted_conv.conv.weight": "decoder.stages.4.3.local_module.inverted_conv.conv.weight", + "decoder.stages.4.op_list.3.local_module.main.point_conv.conv.weight": "decoder.stages.4.3.local_module.point_conv.conv.weight", + "decoder.stages.4.op_list.3.local_module.main.point_conv.norm.bias": "decoder.stages.4.3.local_module.point_conv.norm.bias", + "decoder.stages.4.op_list.3.local_module.main.point_conv.norm.weight": "decoder.stages.4.3.local_module.point_conv.norm.weight", + "decoder.stages.5.op_list.0.context_module.main.aggreg.0.0.weight": "decoder.stages.5.0.context_module.aggreg.0.0.weight", + "decoder.stages.5.op_list.0.context_module.main.aggreg.0.1.weight": "decoder.stages.5.0.context_module.aggreg.0.1.weight", + "decoder.stages.5.op_list.0.context_module.main.proj.conv.weight": "decoder.stages.5.0.context_module.proj.0.weight", + "decoder.stages.5.op_list.0.context_module.main.proj.norm.bias": "decoder.stages.5.0.context_module.proj.1.bias", + "decoder.stages.5.op_list.0.context_module.main.proj.norm.weight": "decoder.stages.5.0.context_module.proj.1.weight", + "decoder.stages.5.op_list.0.context_module.main.qkv.conv.weight": "decoder.stages.5.0.context_module.qkv.0.weight", + "decoder.stages.5.op_list.0.local_module.main.depth_conv.conv.bias": "decoder.stages.5.0.local_module.depth_conv.conv.bias", + "decoder.stages.5.op_list.0.local_module.main.depth_conv.conv.weight": "decoder.stages.5.0.local_module.depth_conv.conv.weight", + "decoder.stages.5.op_list.0.local_module.main.inverted_conv.conv.bias": "decoder.stages.5.0.local_module.inverted_conv.conv.bias", + "decoder.stages.5.op_list.0.local_module.main.inverted_conv.conv.weight": "decoder.stages.5.0.local_module.inverted_conv.conv.weight", + "decoder.stages.5.op_list.0.local_module.main.point_conv.conv.weight": "decoder.stages.5.0.local_module.point_conv.conv.weight", + "decoder.stages.5.op_list.0.local_module.main.point_conv.norm.bias": "decoder.stages.5.0.local_module.point_conv.norm.bias", + "decoder.stages.5.op_list.0.local_module.main.point_conv.norm.weight": "decoder.stages.5.0.local_module.point_conv.norm.weight", + "decoder.stages.5.op_list.1.context_module.main.aggreg.0.0.weight": "decoder.stages.5.1.context_module.aggreg.0.0.weight", + "decoder.stages.5.op_list.1.context_module.main.aggreg.0.1.weight": "decoder.stages.5.1.context_module.aggreg.0.1.weight", + "decoder.stages.5.op_list.1.context_module.main.proj.conv.weight": "decoder.stages.5.1.context_module.proj.0.weight", + "decoder.stages.5.op_list.1.context_module.main.proj.norm.bias": "decoder.stages.5.1.context_module.proj.1.bias", + "decoder.stages.5.op_list.1.context_module.main.proj.norm.weight": "decoder.stages.5.1.context_module.proj.1.weight", + "decoder.stages.5.op_list.1.context_module.main.qkv.conv.weight": "decoder.stages.5.1.context_module.qkv.0.weight", + "decoder.stages.5.op_list.1.local_module.main.depth_conv.conv.bias": "decoder.stages.5.1.local_module.depth_conv.conv.bias", + "decoder.stages.5.op_list.1.local_module.main.depth_conv.conv.weight": "decoder.stages.5.1.local_module.depth_conv.conv.weight", + "decoder.stages.5.op_list.1.local_module.main.inverted_conv.conv.bias": "decoder.stages.5.1.local_module.inverted_conv.conv.bias", + "decoder.stages.5.op_list.1.local_module.main.inverted_conv.conv.weight": "decoder.stages.5.1.local_module.inverted_conv.conv.weight", + "decoder.stages.5.op_list.1.local_module.main.point_conv.conv.weight": "decoder.stages.5.1.local_module.point_conv.conv.weight", + "decoder.stages.5.op_list.1.local_module.main.point_conv.norm.bias": "decoder.stages.5.1.local_module.point_conv.norm.bias", + "decoder.stages.5.op_list.1.local_module.main.point_conv.norm.weight": "decoder.stages.5.1.local_module.point_conv.norm.weight", + "decoder.stages.5.op_list.2.context_module.main.aggreg.0.0.weight": "decoder.stages.5.2.context_module.aggreg.0.0.weight", + "decoder.stages.5.op_list.2.context_module.main.aggreg.0.1.weight": "decoder.stages.5.2.context_module.aggreg.0.1.weight", + "decoder.stages.5.op_list.2.context_module.main.proj.conv.weight": "decoder.stages.5.2.context_module.proj.0.weight", + "decoder.stages.5.op_list.2.context_module.main.proj.norm.bias": "decoder.stages.5.2.context_module.proj.1.bias", + "decoder.stages.5.op_list.2.context_module.main.proj.norm.weight": "decoder.stages.5.2.context_module.proj.1.weight", + "decoder.stages.5.op_list.2.context_module.main.qkv.conv.weight": "decoder.stages.5.2.context_module.qkv.0.weight", + "decoder.stages.5.op_list.2.local_module.main.depth_conv.conv.bias": "decoder.stages.5.2.local_module.depth_conv.conv.bias", + "decoder.stages.5.op_list.2.local_module.main.depth_conv.conv.weight": "decoder.stages.5.2.local_module.depth_conv.conv.weight", + "decoder.stages.5.op_list.2.local_module.main.inverted_conv.conv.bias": "decoder.stages.5.2.local_module.inverted_conv.conv.bias", + "decoder.stages.5.op_list.2.local_module.main.inverted_conv.conv.weight": "decoder.stages.5.2.local_module.inverted_conv.conv.weight", + "decoder.stages.5.op_list.2.local_module.main.point_conv.conv.weight": "decoder.stages.5.2.local_module.point_conv.conv.weight", + "decoder.stages.5.op_list.2.local_module.main.point_conv.norm.bias": "decoder.stages.5.2.local_module.point_conv.norm.bias", + "decoder.stages.5.op_list.2.local_module.main.point_conv.norm.weight": "decoder.stages.5.2.local_module.point_conv.norm.weight", + "decoder.project_out.op_list.0.bias": "decoder.project_out.0.bias", + "decoder.project_out.op_list.0.weight": "decoder.project_out.0.weight", + "decoder.project_out.op_list.2.conv.bias": "decoder.project_out.2.conv.bias", + "decoder.project_out.op_list.2.conv.weight": "decoder.project_out.2.conv.weight", + } From d96f63d5ed1798462d07a0a0ec2ed63c021c5f0b Mon Sep 17 00:00:00 2001 From: Zhaolun Zou Date: Tue, 17 Dec 2024 12:10:07 +0800 Subject: [PATCH 2/3] Modified VAE (DCAE) weight download address in README. --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index fb36c09..ab4a667 100644 --- a/README.md +++ b/README.md @@ -59,7 +59,7 @@ https://github.com/Efficient-Large-Model/ComfyUI_ExtraModels 2. Place them in your checkpoints folder 3. Load them with the correct PixArt checkpoint loader 4. Use the "Gemma Loader" node - it should automatically download the requested model from Huggingface - Recommended to use the 4bit quantized model on CPU when low on memory. -5. Download the VAE from [here](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px_diffusers/blob/main/vae/diffusion_pytorch_model.safetensors) and place it in your VAE folder after renaming it. +5. Download the VAE from [here](https://huggingface.co/mit-han-lab/dc-ae-f32c32-sana-1.0/blob/main/model.safetensors) and place it in your VAE folder after renaming it. 6. Use either the "Empty Sana Latent Image" or "Empty DCAE Latent Image" node for the latent input when doing txt2img. [Sample workflow](https://github.com/user-attachments/files/18027854/SanaV1.json) From 99bff66e3983c64e2390773d06a33175799630db Mon Sep 17 00:00:00 2001 From: Zhaolun Zou Date: Tue, 17 Dec 2024 13:34:10 +0800 Subject: [PATCH 3/3] Add key check before dcae key mapping to prevent breaking existing weights; Added both weight addresses to README.md --- README.md | 2 +- VAE/loader.py | 5 +++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index ab4a667..28cccbe 100644 --- a/README.md +++ b/README.md @@ -59,7 +59,7 @@ https://github.com/Efficient-Large-Model/ComfyUI_ExtraModels 2. Place them in your checkpoints folder 3. Load them with the correct PixArt checkpoint loader 4. Use the "Gemma Loader" node - it should automatically download the requested model from Huggingface - Recommended to use the 4bit quantized model on CPU when low on memory. -5. Download the VAE from [here](https://huggingface.co/mit-han-lab/dc-ae-f32c32-sana-1.0/blob/main/model.safetensors) and place it in your VAE folder after renaming it. +5. Download the VAE from [here](https://huggingface.co/Efficient-Large-Model/Sana_1600M_1024px_diffusers/blob/main/vae/diffusion_pytorch_model.safetensors) or [here](https://huggingface.co/mit-han-lab/dc-ae-f32c32-sana-1.0/blob/main/model.safetensors) and place it in your VAE folder after renaming it. 6. Use either the "Empty Sana Latent Image" or "Empty DCAE Latent Image" node for the latent input when doing txt2img. [Sample workflow](https://github.com/user-attachments/files/18027854/SanaV1.json) diff --git a/VAE/loader.py b/VAE/loader.py index d3809ed..874f224 100644 --- a/VAE/loader.py +++ b/VAE/loader.py @@ -34,8 +34,9 @@ class EXVAE(comfy.sd.VAE): model = MoVQ(model_conf) elif model_conf["type"] == "DCAE": from .models.dcae import DCAE - from .models import dcae_key_mapping - sd = dcae_key_mapping.convert_sd(sd) + if 'decoder.project_out.op_list.0.bias' in sd: + from .models import dcae_key_mapping + sd = dcae_key_mapping.convert_sd(sd) model = DCAE(**model_conf) else: raise NotImplementedError(f"Unknown VAE type '{model_conf['type']}'")