Updated Out of Date Files
God bless city96. Now the clip nodes can load qwen3 gguf models and probably a bunch of other cool fixes. 👍
This commit is contained in:
+54
-1
@@ -23,7 +23,7 @@ def dequantize_tensor(tensor, dtype=None, dequant_dtype=None):
|
||||
return dequantize(tensor.data, qtype, oshape, dtype=dequant_dtype).to(dtype)
|
||||
else:
|
||||
# this is incredibly slow
|
||||
tqdm.write(f"Falling back to numpy dequant for qtype: {qtype}")
|
||||
tqdm.write(f"Falling back to numpy dequant for qtype: {getattr(qtype, 'name', repr(qtype))}")
|
||||
new = gguf.quants.dequantize(tensor.cpu().numpy(), qtype)
|
||||
return torch.from_numpy(new).to(tensor.device, dtype=dtype)
|
||||
|
||||
@@ -48,6 +48,10 @@ def to_uint32(x):
|
||||
x = x.view(torch.uint8).to(torch.int32)
|
||||
return (x[:, 0] | x[:, 1] << 8 | x[:, 2] << 16 | x[:, 3] << 24).unsqueeze(1)
|
||||
|
||||
def to_uint16(x):
|
||||
x = x.view(torch.uint8).to(torch.int32)
|
||||
return (x[:, 0] | x[:, 1] << 8).unsqueeze(1)
|
||||
|
||||
def split_block_dims(blocks, *args):
|
||||
n_max = blocks.shape[1]
|
||||
dims = list(args) + [n_max - sum(args)]
|
||||
@@ -233,6 +237,53 @@ def dequantize_blocks_Q2_K(blocks, block_size, type_size, dtype=None):
|
||||
|
||||
return qs.reshape((n_blocks, -1))
|
||||
|
||||
# IQ quants
|
||||
KVALUES = torch.tensor([-127, -104, -83, -65, -49, -35, -22, -10, 1, 13, 25, 38, 53, 69, 89, 113], dtype=torch.int8)
|
||||
|
||||
def dequantize_blocks_IQ4_NL(blocks, block_size, type_size, dtype=None):
|
||||
n_blocks = blocks.shape[0]
|
||||
|
||||
d, qs = split_block_dims(blocks, 2)
|
||||
d = d.view(torch.float16).to(dtype)
|
||||
|
||||
qs = qs.reshape((n_blocks, -1, 1, block_size//2)) >> torch.tensor([0, 4], device=d.device, dtype=torch.uint8).reshape((1, 1, 2, 1))
|
||||
qs = (qs & 0x0F).reshape((n_blocks, -1, 1)).to(torch.int32)
|
||||
|
||||
kvalues = KVALUES.to(qs.device).expand(*qs.shape[:-1], 16)
|
||||
qs = torch.gather(kvalues, dim=-1, index=qs).reshape((n_blocks, -1))
|
||||
del kvalues # should still be view, but just to be safe
|
||||
|
||||
return (d * qs)
|
||||
|
||||
def dequantize_blocks_IQ4_XS(blocks, block_size, type_size, dtype=None):
|
||||
n_blocks = blocks.shape[0]
|
||||
d, scales_h, scales_l, qs = split_block_dims(blocks, 2, 2, QK_K // 64)
|
||||
d = d.view(torch.float16).to(dtype)
|
||||
scales_h = to_uint16(scales_h)
|
||||
|
||||
shift_a = torch.tensor([0, 4], device=d.device, dtype=torch.uint8).reshape((1, 1, 2))
|
||||
shift_b = torch.tensor([2 * i for i in range(QK_K // 32)], device=d.device, dtype=torch.uint8).reshape((1, -1, 1))
|
||||
|
||||
scales_l = scales_l.reshape((n_blocks, -1, 1)) >> shift_a.reshape((1, 1, 2))
|
||||
scales_h = scales_h.reshape((n_blocks, -1, 1)) >> shift_b.reshape((1, -1, 1))
|
||||
|
||||
scales_l = scales_l.reshape((n_blocks, -1)) & 0x0F
|
||||
scales_h = scales_h.reshape((n_blocks, -1)).to(torch.uint8) & 0x03
|
||||
|
||||
scales = (scales_l | (scales_h << 4)).to(torch.int8) - 32
|
||||
dl = (d * scales.to(dtype)).reshape((n_blocks, -1, 1))
|
||||
|
||||
qs = qs.reshape((n_blocks, -1, 1, 16)) >> shift_a.reshape((1, 1, 2, 1))
|
||||
qs = qs.reshape((n_blocks, -1, 32, 1)) & 0x0F
|
||||
|
||||
kvalues = KVALUES.to(qs.device).expand(*qs.shape[:-1], 16)
|
||||
qs = torch.gather(kvalues, dim=-1, index=qs.to(torch.int32)).reshape((n_blocks, -1, 32))
|
||||
del kvalues # see IQ4_NL
|
||||
del shift_a
|
||||
del shift_b
|
||||
|
||||
return (dl * qs).reshape((n_blocks, -1))
|
||||
|
||||
dequantize_functions = {
|
||||
gguf.GGMLQuantizationType.BF16: dequantize_blocks_BF16,
|
||||
gguf.GGMLQuantizationType.Q8_0: dequantize_blocks_Q8_0,
|
||||
@@ -245,4 +296,6 @@ dequantize_functions = {
|
||||
gguf.GGMLQuantizationType.Q4_K: dequantize_blocks_Q4_K,
|
||||
gguf.GGMLQuantizationType.Q3_K: dequantize_blocks_Q3_K,
|
||||
gguf.GGMLQuantizationType.Q2_K: dequantize_blocks_Q2_K,
|
||||
gguf.GGMLQuantizationType.IQ4_NL: dequantize_blocks_IQ4_NL,
|
||||
gguf.GGMLQuantizationType.IQ4_XS: dequantize_blocks_IQ4_XS,
|
||||
}
|
||||
|
||||
@@ -3,12 +3,15 @@ import warnings
|
||||
import logging
|
||||
import torch
|
||||
import gguf
|
||||
import re
|
||||
import os
|
||||
|
||||
from .ops import GGMLTensor
|
||||
from .dequant import is_quantized, dequantize_tensor
|
||||
|
||||
IMG_ARCH_LIST = {"flux", "sd1", "sdxl", "sd3", "aura", "hidream", "cosmos", "ltxv", "hyvid", "wan"}
|
||||
TXT_ARCH_LIST = {"t5", "t5encoder", "llama"}
|
||||
IMG_ARCH_LIST = {"flux", "sd1", "sdxl", "sd3", "aura", "hidream", "cosmos", "ltxv", "hyvid", "wan", "lumina2", "qwen_image"}
|
||||
TXT_ARCH_LIST = {"t5", "t5encoder", "llama", "qwen2vl", "qwen3", "qwen3vl"}
|
||||
VIS_TYPE_LIST = {"clip-vision", "mmproj"}
|
||||
|
||||
def get_orig_shape(reader, tensor_name):
|
||||
field_key = f"comfy.gguf.orig_shape.{tensor_name}"
|
||||
@@ -70,9 +73,10 @@ def gguf_sd_loader(path, handle_prefix="model.diffusion_model.", return_arch=Fal
|
||||
# detect and verify architecture
|
||||
compat = None
|
||||
arch_str = get_field(reader, "general.architecture", str)
|
||||
if arch_str in [None, "pig"]:
|
||||
type_str = get_field(reader, "general.type", str)
|
||||
if arch_str in [None, "pig", "cow"]:
|
||||
if is_text_model:
|
||||
raise ValueError(f"This text model is incompatible with llama.cpp!\nConsider using the safetensors version\n({path})")
|
||||
raise ValueError(f"This gguf file is incompatible with llama.cpp!\nConsider using safetensors or a compatible gguf file\n({path})")
|
||||
compat = "sd.cpp" if arch_str is None else arch_str
|
||||
# import here to avoid changes to convert.py breaking regular models
|
||||
from .tools.convert import detect_arch
|
||||
@@ -81,7 +85,8 @@ def gguf_sd_loader(path, handle_prefix="model.diffusion_model.", return_arch=Fal
|
||||
except Exception as e:
|
||||
raise ValueError(f"This model is not currently supported - ({e})")
|
||||
elif arch_str not in TXT_ARCH_LIST and is_text_model:
|
||||
raise ValueError(f"Unexpected text model architecture type in GGUF file: {arch_str!r}")
|
||||
if type_str not in VIS_TYPE_LIST:
|
||||
raise ValueError(f"Unexpected text model architecture type in GGUF file: {arch_str!r}")
|
||||
elif arch_str not in IMG_ARCH_LIST and not is_text_model:
|
||||
raise ValueError(f"Unexpected architecture type in GGUF file: {arch_str!r}")
|
||||
|
||||
@@ -152,6 +157,9 @@ T5_SD_MAP = {
|
||||
LLAMA_SD_MAP = {
|
||||
"blk.": "model.layers.",
|
||||
"attn_norm": "input_layernorm",
|
||||
"attn_q_norm.": "self_attn.q_norm.",
|
||||
"attn_k_norm.": "self_attn.k_norm.",
|
||||
"attn_v_norm.": "self_attn.v_norm.",
|
||||
"attn_q": "self_attn.q_proj",
|
||||
"attn_k": "self_attn.k_proj",
|
||||
"attn_v": "self_attn.v_proj",
|
||||
@@ -165,6 +173,19 @@ LLAMA_SD_MAP = {
|
||||
"output.weight": "lm_head.weight",
|
||||
}
|
||||
|
||||
CLIP_VISION_SD_MAP = {
|
||||
"mm.": "visual.merger.mlp.",
|
||||
"v.post_ln.": "visual.merger.ln_q.",
|
||||
"v.patch_embd": "visual.patch_embed.proj",
|
||||
"v.blk.": "visual.blocks.",
|
||||
"ffn_up": "mlp.up_proj",
|
||||
"ffn_down": "mlp.down_proj",
|
||||
"ffn_gate": "mlp.gate_proj",
|
||||
"attn_out.": "attn.proj.",
|
||||
"ln1.": "norm1.",
|
||||
"ln2.": "norm2.",
|
||||
}
|
||||
|
||||
def sd_map_replace(raw_sd, key_map):
|
||||
sd = {}
|
||||
for k,v in raw_sd.items():
|
||||
@@ -185,6 +206,79 @@ def llama_permute(raw_sd, n_head, n_head_kv):
|
||||
sd[k] = v
|
||||
return sd
|
||||
|
||||
def strip_quant_suffix(name):
|
||||
pattern = r"[-_]?(?:ud-)?i?q[0-9]_[a-z0-9_\-]{1,8}$"
|
||||
match = re.search(pattern, name, re.IGNORECASE)
|
||||
if match:
|
||||
name = name[:match.start()]
|
||||
return name
|
||||
|
||||
def gguf_mmproj_loader(path):
|
||||
# Reverse version of Qwen2VLVisionModel.modify_tensors
|
||||
logging.info("Attenpting to find mmproj file for text encoder...")
|
||||
|
||||
# get name to match w/o quant suffix
|
||||
tenc_fname = os.path.basename(path)
|
||||
tenc = os.path.splitext(tenc_fname)[0].lower()
|
||||
tenc = strip_quant_suffix(tenc)
|
||||
|
||||
# try and find matching mmproj
|
||||
target = []
|
||||
root = os.path.dirname(path)
|
||||
for fname in os.listdir(root):
|
||||
name, ext = os.path.splitext(fname)
|
||||
if ext.lower() != ".gguf":
|
||||
continue
|
||||
if "mmproj" not in name.lower():
|
||||
continue
|
||||
if tenc in name.lower():
|
||||
target.append(fname)
|
||||
|
||||
if len(target) == 0:
|
||||
logging.error(f"Error: Can't find mmproj file for '{tenc_fname}' (matching:'{tenc}')! Qwen-Image-Edit will be broken!")
|
||||
return {}
|
||||
if len(target) > 1:
|
||||
logging.error(f"Ambiguous mmproj for text encoder '{tenc_fname}', will use first match.")
|
||||
|
||||
logging.info(f"Using mmproj '{target[0]}' for text encoder '{tenc_fname}'.")
|
||||
target = os.path.join(root, target[0])
|
||||
vsd = gguf_sd_loader(target, is_text_model=True)
|
||||
|
||||
# concat 4D to 5D
|
||||
if "v.patch_embd.weight.1" in vsd:
|
||||
w1 = dequantize_tensor(vsd.pop("v.patch_embd.weight"), dtype=torch.float32)
|
||||
w2 = dequantize_tensor(vsd.pop("v.patch_embd.weight.1"), dtype=torch.float32)
|
||||
vsd["v.patch_embd.weight"] = torch.stack([w1, w2], dim=2)
|
||||
|
||||
# run main replacement
|
||||
vsd = sd_map_replace(vsd, CLIP_VISION_SD_MAP)
|
||||
|
||||
# handle split Q/K/V
|
||||
if "visual.blocks.0.attn_q.weight" in vsd:
|
||||
attns = {}
|
||||
# filter out attentions + group
|
||||
for k,v in vsd.items():
|
||||
if any(x in k for x in ["attn_q", "attn_k", "attn_v"]):
|
||||
k_attn, k_name = k.rsplit(".attn_", 1)
|
||||
k_attn += ".attn.qkv." + k_name.split(".")[-1]
|
||||
if k_attn not in attns:
|
||||
attns[k_attn] = {}
|
||||
attns[k_attn][k_name] = dequantize_tensor(
|
||||
v, dtype=(torch.bfloat16 if is_quantized(v) else torch.float16)
|
||||
)
|
||||
|
||||
# recombine
|
||||
for k,v in attns.items():
|
||||
suffix = k.split(".")[-1]
|
||||
vsd[k] = torch.cat([
|
||||
v[f"q.{suffix}"],
|
||||
v[f"k.{suffix}"],
|
||||
v[f"v.{suffix}"],
|
||||
], dim=0)
|
||||
del attns
|
||||
|
||||
return vsd
|
||||
|
||||
def gguf_tokenizer_loader(path, temb_shape):
|
||||
# convert gguf tokenizer to spiece
|
||||
logging.info("Attempting to recreate sentencepiece tokenizer from GGUF file metadata...")
|
||||
@@ -233,6 +327,49 @@ def gguf_tokenizer_loader(path, temb_shape):
|
||||
del reader
|
||||
return torch.ByteTensor(list(spm.SerializeToString()))
|
||||
|
||||
def gguf_tekken_tokenizer_loader(path, temb_shape):
|
||||
# convert ggml (hf) tokenizer metadata to tekken/comfy data
|
||||
logging.info("Attempting to recreate tekken tokenizer from GGUF file metadata...")
|
||||
import json
|
||||
import base64
|
||||
from transformers.convert_slow_tokenizer import bytes_to_unicode
|
||||
|
||||
reader = gguf.GGUFReader(path)
|
||||
|
||||
model_str = get_field(reader, "tokenizer.ggml.model", str)
|
||||
if model_str == "gpt2":
|
||||
if temb_shape == (131072, 5120): # probably Mistral
|
||||
data = {
|
||||
"config": {"num_vocab_tokens": 150000, "default_vocab_size": 131072},
|
||||
"vocab": [],
|
||||
"special_tokens": [],
|
||||
}
|
||||
else:
|
||||
raise NotImplementedError("Unknown model, can't set tokenizer!")
|
||||
else:
|
||||
raise NotImplementedError("Unknown model, can't set tokenizer!")
|
||||
|
||||
tokens = get_list_field(reader, "tokenizer.ggml.tokens", str)
|
||||
toktypes = get_list_field(reader, "tokenizer.ggml.token_type", int)
|
||||
|
||||
decoder = {v: k for k, v in bytes_to_unicode().items()}
|
||||
for idx, (token, toktype) in enumerate(zip(tokens, toktypes)):
|
||||
if toktype == 3:
|
||||
data["special_tokens"].append(
|
||||
{'rank': idx, 'token_str': token, 'is_control': True}
|
||||
)
|
||||
else:
|
||||
tok = bytes([decoder[char] for char in token])
|
||||
data["vocab"].append({
|
||||
"rank": len(data["vocab"]),
|
||||
"token_bytes": base64.b64encode(tok).decode("ascii"),
|
||||
"token_str": tok.decode("utf-8", errors="replace") # ?
|
||||
})
|
||||
|
||||
logging.info(f"Created tekken tokenizer with vocab size of {len(data['vocab'])} (+{len(data['special_tokens'])})")
|
||||
del reader
|
||||
return torch.ByteTensor(list(json.dumps(data).encode('utf-8')))
|
||||
|
||||
def gguf_clip_loader(path):
|
||||
sd, arch = gguf_sd_loader(path, return_arch=True, is_text_model=True)
|
||||
if arch in {"t5", "t5encoder"}:
|
||||
@@ -244,15 +381,22 @@ def gguf_clip_loader(path):
|
||||
logging.warning(f"Dequantizing {temb_key} to prevent runtime OOM.")
|
||||
sd[temb_key] = dequantize_tensor(sd[temb_key], dtype=torch.float16)
|
||||
sd = sd_map_replace(sd, T5_SD_MAP)
|
||||
elif arch in {"llama"}:
|
||||
elif arch in {"llama", "qwen2vl", "qwen3", "qwen3vl"}:
|
||||
# TODO: pass model_options["vocab_size"] to loader somehow
|
||||
temb_key = "token_embd.weight"
|
||||
if temb_key in sd and sd[temb_key].shape[0] >= (64 * 1024):
|
||||
if arch == "llama" and sd[temb_key].shape == (131072, 5120):
|
||||
# non-standard Comfy-Org tokenizer
|
||||
sd["tekken_model"] = gguf_tekken_tokenizer_loader(path, sd[temb_key].shape)
|
||||
# See note above for T5.
|
||||
logging.warning(f"Dequantizing {temb_key} to prevent runtime OOM.")
|
||||
sd[temb_key] = dequantize_tensor(sd[temb_key], dtype=torch.float16)
|
||||
sd = sd_map_replace(sd, LLAMA_SD_MAP)
|
||||
sd = llama_permute(sd, 32, 8) # L3
|
||||
if arch == "llama":
|
||||
sd = llama_permute(sd, 32, 8) # L3 / Mistral
|
||||
if arch == "qwen2vl":
|
||||
vsd = gguf_mmproj_loader(path)
|
||||
sd.update(vsd)
|
||||
else:
|
||||
pass
|
||||
return sd
|
||||
|
||||
@@ -153,7 +153,7 @@ class GGMLLayer(torch.nn.Module):
|
||||
# Take into account space required for dequantizing the largest tensor
|
||||
if self.largest_layer:
|
||||
shape = getattr(self.weight, "tensor_shape", self.weight.shape)
|
||||
dtype = self.dequant_dtype or torch.float16
|
||||
dtype = self.dequant_dtype if self.dequant_dtype and self.dequant_dtype != "target" else torch.float16
|
||||
temp = torch.empty(*shape, device=torch.device("meta"), dtype=dtype)
|
||||
destination[prefix + "temp.weight"] = temp
|
||||
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
## Converting initial model
|
||||
|
||||
To convert your initial safetensors/ckpt model to FP16/BF16 GGUF, run the following command:
|
||||
|
||||
```
|
||||
python convert.py --src E:\models\unet\flux1-dev.safetensors
|
||||
```
|
||||
Make sure `gguf>=0.13.0` is installed for this step. Optionally, specify the output gguf file with the `--dst` arg.
|
||||
|
||||
> [!NOTE]
|
||||
> Do not use the diffusers UNET format for flux, it won't work, use the default/reference checkpoint key format. This is due to q/k/v being merged into one qkv key.
|
||||
> You can convert it by loading it in ComfyUI and saving it using the built-in "ModelSave" node.
|
||||
|
||||
> [!WARNING]
|
||||
> For hunyuan video/wan 2.1, you will see a warning about 5D tensors. This means the script will save a **non functional** model to disk first, that you can quantize. I recommend saving these in a separate `raw` folder to avoid confusion.
|
||||
>
|
||||
> After quantization, you will have to run `fix_5d_tensor.py` manually to add back the missing key that was saved by the conversion code.
|
||||
|
||||
## Quantizing using custom llama.cpp
|
||||
|
||||
Depending on your git settings, you may need to run the following script first in order to make sure the patch file is valid. It will convert Windows (CRLF) line endings to Unix (LF) ones.
|
||||
|
||||
```
|
||||
python fix_lines_ending.py
|
||||
```
|
||||
|
||||
Git clone llama.cpp into the current folder:
|
||||
|
||||
```
|
||||
git clone https://github.com/ggerganov/llama.cpp
|
||||
```
|
||||
|
||||
Check out the correct branch, then apply the custom patch needed to add image model support to the repo you just cloned.
|
||||
|
||||
```
|
||||
cd llama.cpp
|
||||
git checkout tags/b3962
|
||||
git apply ..\lcpp.patch
|
||||
```
|
||||
|
||||
Compile the llama-quantize binary. This example uses cmake, on linux you can just use make.
|
||||
|
||||
### Visual Studio 2019, Linux, etc...
|
||||
|
||||
```
|
||||
mkdir build
|
||||
cmake -B build
|
||||
cmake --build build --config Debug -j10 --target llama-quantize
|
||||
cd ..
|
||||
```
|
||||
|
||||
### Visual Studio 2022
|
||||
|
||||
```
|
||||
mkdir build
|
||||
cmake -B build -DCMAKE_CXX_STANDARD=17 -DCMAKE_CXX_STANDARD_REQUIRED=ON -DCMAKE_CXX_FLAGS="-std=c++17"
|
||||
```
|
||||
|
||||
Edit the `llama.cpp\common\log.cpp` file, inserts two lines after the existing first line:
|
||||
|
||||
```
|
||||
#include "log.h"
|
||||
|
||||
#define _SILENCE_CXX23_CHRONO_DEPRECATION_WARNING
|
||||
#include <chrono>
|
||||
```
|
||||
|
||||
Then you can build the project:
|
||||
```
|
||||
cmake --build build --config Debug -j10 --target llama-quantize
|
||||
cd ..
|
||||
```
|
||||
|
||||
### Quantize your model
|
||||
|
||||
|
||||
Now you can use the newly build binary to quantize your model to the desired format:
|
||||
```
|
||||
llama.cpp\build\bin\Debug\llama-quantize.exe E:\models\unet\flux1-dev-BF16.gguf E:\models\unet\flux1-dev-Q4_K_S.gguf Q4_K_S
|
||||
```
|
||||
|
||||
You can extract the patch again with `git diff src\llama.cpp > lcpp.patch` if you wish to change something and contribute back.
|
||||
|
||||
> [!WARNING]
|
||||
> For hunyuan video/wan 2.1, you will have to run `fix_5d_tensor.py` after the quantization step is done.
|
||||
>
|
||||
> Example usage: `fix_5d_tensors.py --src E:\models\video\raw\wan2.1-t2v-1.3b-Q8_0.gguf --dst E:\models\video\wan2.1-t2v-1.3b-Q8_0.gguf`
|
||||
>
|
||||
> By default, this also saves a `fix_5d_tensors_[arch].safetensors` file in the `ComfyUI-GGUF/tools` folder, it's recommended to delete this after all models have been converted.
|
||||
|
||||
> [!NOTE]
|
||||
> Do not quantize SDXL / SD1 / other Conv2D heavy models. If you do, make sure to **extract the UNET model first**.
|
||||
>This should be obvious, but also don't use the resulting llama-quantize binary with LLMs.
|
||||
+9
-2
@@ -139,8 +139,14 @@ class ModelSD1(ModelTemplate):
|
||||
), # Non-diffusers
|
||||
]
|
||||
|
||||
# The architectures are checked in order and the first successful match terminates the search.
|
||||
arch_list = [ModelFlux, ModelSD3, ModelAura, ModelHiDream, CosmosPredict2, ModelLTXV, ModelHyVid, ModelWan, ModelSDXL, ModelSD1]
|
||||
class ModelLumina2(ModelTemplate):
|
||||
arch = "lumina2"
|
||||
keys_detect = [
|
||||
("cap_embedder.1.weight", "context_refiner.0.attention.qkv.weight")
|
||||
]
|
||||
|
||||
arch_list = [ModelFlux, ModelSD3, ModelAura, ModelHiDream, CosmosPredict2,
|
||||
ModelLTXV, ModelHyVid, ModelWan, ModelSDXL, ModelSD1, ModelLumina2]
|
||||
|
||||
def is_model_arch(model, state_dict):
|
||||
# check if model is correct
|
||||
@@ -356,3 +362,4 @@ def convert_file(path, dst_path=None, interact=True, overwrite=False):
|
||||
if __name__ == "__main__":
|
||||
args = parse_args()
|
||||
convert_file(args.src, args.dst)
|
||||
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
# (c) City96 || Apache-2.0 (apache.org/licenses/LICENSE-2.0)
|
||||
import os
|
||||
import gguf
|
||||
import torch
|
||||
import argparse
|
||||
from tqdm import tqdm
|
||||
from safetensors.torch import load_file
|
||||
|
||||
def get_args():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--src", required=True)
|
||||
parser.add_argument("--dst", required=True)
|
||||
parser.add_argument("--fix", required=False, help="Defaults to ./fix_5d_tensors_[arch].pt")
|
||||
parser.add_argument("--overwrite", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not os.path.isfile(args.src):
|
||||
parser.error(f"Invalid source file '{args.src}'")
|
||||
if not args.overwrite and os.path.exists(args.dst):
|
||||
parser.error(f"Output exists, use '--overwrite' ({args.dst})")
|
||||
|
||||
return args
|
||||
|
||||
def get_arch_str(reader):
|
||||
field = reader.get_field("general.architecture")
|
||||
return str(field.parts[field.data[-1]], encoding="utf-8")
|
||||
|
||||
def get_file_type(reader):
|
||||
field = reader.get_field("general.file_type")
|
||||
ft = int(field.parts[field.data[-1]])
|
||||
return gguf.LlamaFileType(ft)
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = get_args()
|
||||
|
||||
# read existing
|
||||
reader = gguf.GGUFReader(args.src)
|
||||
arch = get_arch_str(reader)
|
||||
file_type = get_file_type(reader)
|
||||
print(f"Detected arch: '{arch}' (ftype: {str(file_type)})")
|
||||
|
||||
# prep fix
|
||||
if args.fix is None:
|
||||
args.fix = f"./fix_5d_tensors_{arch}.safetensors"
|
||||
|
||||
if not os.path.isfile(args.fix):
|
||||
raise OSError(f"No 5D tensor fix file: {args.fix}")
|
||||
|
||||
sd5d = load_file(args.fix)
|
||||
sd5d = {k:v.numpy() for k,v in sd5d.items()}
|
||||
print("5D tensors:", sd5d.keys())
|
||||
|
||||
# prep output
|
||||
writer = gguf.GGUFWriter(path=None, arch=arch)
|
||||
writer.add_quantization_version(gguf.GGML_QUANT_VERSION)
|
||||
writer.add_file_type(file_type)
|
||||
|
||||
added = []
|
||||
def add_extra_key(writer, key, data):
|
||||
global added
|
||||
data_qtype = gguf.GGMLQuantizationType.F32
|
||||
data = gguf.quants.quantize(data, data_qtype)
|
||||
tqdm.write(f"Adding key {key} ({data.shape})")
|
||||
writer.add_tensor(key, data, raw_dtype=data_qtype)
|
||||
added.append(key)
|
||||
|
||||
# main loop to add missing 5D tensor(s)
|
||||
for tensor in tqdm(reader.tensors):
|
||||
writer.add_tensor(tensor.name, tensor.data, raw_dtype=tensor.tensor_type)
|
||||
key5d = tensor.name.replace(".bias", ".weight")
|
||||
if key5d in sd5d.keys():
|
||||
add_extra_key(writer, key5d, sd5d[key5d])
|
||||
|
||||
# brute force for any missed
|
||||
for key, data in sd5d.items():
|
||||
if key not in added:
|
||||
add_extra_key(writer, key, data)
|
||||
|
||||
writer.write_header_to_file(path=args.dst)
|
||||
writer.write_kv_data_to_file()
|
||||
writer.write_tensors_to_file(progress=True)
|
||||
writer.close()
|
||||
@@ -0,0 +1,31 @@
|
||||
import os
|
||||
|
||||
files = ["lcpp.patch", "lcpp_sd3.patch"]
|
||||
|
||||
def has_unix_line_endings(file_path):
|
||||
try:
|
||||
with open(file_path, 'rb') as file:
|
||||
content = file.read()
|
||||
return b'\r\n' not in content
|
||||
except Exception as e:
|
||||
print(f"Error checking '{file_path}': {e}")
|
||||
return False
|
||||
|
||||
def convert_to_linux_format(file_path):
|
||||
try:
|
||||
with open(file_path, 'rb') as file:
|
||||
content = file.read().replace(b'\r\n', b'\n')
|
||||
with open(file_path, 'wb') as file:
|
||||
file.write(content)
|
||||
print(f"'{file_path}' converted to Linux line endings (LF).")
|
||||
except Exception as e:
|
||||
print(f"Error processing '{file_path}': {e}")
|
||||
|
||||
for file in files:
|
||||
if os.path.exists(file):
|
||||
if has_unix_line_endings(file):
|
||||
print(f"'{file}' already has Unix line endings (LF). No conversion needed.")
|
||||
else:
|
||||
convert_to_linux_format(file)
|
||||
else:
|
||||
print(f"File '{file}' does not exist.")
|
||||
@@ -0,0 +1,451 @@
|
||||
diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h
|
||||
index de3c706f..0267c1fa 100644
|
||||
--- a/ggml/include/ggml.h
|
||||
+++ b/ggml/include/ggml.h
|
||||
@@ -223,7 +223,7 @@
|
||||
#define GGML_MAX_OP_PARAMS 64
|
||||
|
||||
#ifndef GGML_MAX_NAME
|
||||
-# define GGML_MAX_NAME 64
|
||||
+# define GGML_MAX_NAME 128
|
||||
#endif
|
||||
|
||||
#define GGML_DEFAULT_N_THREADS 4
|
||||
@@ -2449,6 +2449,7 @@ extern "C" {
|
||||
|
||||
// manage tensor info
|
||||
GGML_API void gguf_add_tensor(struct gguf_context * ctx, const struct ggml_tensor * tensor);
|
||||
+ GGML_API void gguf_set_tensor_ndim(struct gguf_context * ctx, const char * name, int n_dim);
|
||||
GGML_API void gguf_set_tensor_type(struct gguf_context * ctx, const char * name, enum ggml_type type);
|
||||
GGML_API void gguf_set_tensor_data(struct gguf_context * ctx, const char * name, const void * data, size_t size);
|
||||
|
||||
diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c
|
||||
index b16c462f..6d1568f1 100644
|
||||
--- a/ggml/src/ggml.c
|
||||
+++ b/ggml/src/ggml.c
|
||||
@@ -22960,6 +22960,14 @@ void gguf_add_tensor(
|
||||
ctx->header.n_tensors++;
|
||||
}
|
||||
|
||||
+void gguf_set_tensor_ndim(struct gguf_context * ctx, const char * name, const int n_dim) {
|
||||
+ const int idx = gguf_find_tensor(ctx, name);
|
||||
+ if (idx < 0) {
|
||||
+ GGML_ABORT("tensor not found");
|
||||
+ }
|
||||
+ ctx->infos[idx].n_dims = n_dim;
|
||||
+}
|
||||
+
|
||||
void gguf_set_tensor_type(struct gguf_context * ctx, const char * name, enum ggml_type type) {
|
||||
const int idx = gguf_find_tensor(ctx, name);
|
||||
if (idx < 0) {
|
||||
diff --git a/src/llama.cpp b/src/llama.cpp
|
||||
index 24e1f1f0..25db4c69 100644
|
||||
--- a/src/llama.cpp
|
||||
+++ b/src/llama.cpp
|
||||
@@ -205,6 +205,17 @@ enum llm_arch {
|
||||
LLM_ARCH_GRANITE,
|
||||
LLM_ARCH_GRANITE_MOE,
|
||||
LLM_ARCH_CHAMELEON,
|
||||
+ LLM_ARCH_FLUX,
|
||||
+ LLM_ARCH_SD1,
|
||||
+ LLM_ARCH_SDXL,
|
||||
+ LLM_ARCH_SD3,
|
||||
+ LLM_ARCH_AURA,
|
||||
+ LLM_ARCH_LTXV,
|
||||
+ LLM_ARCH_HYVID,
|
||||
+ LLM_ARCH_WAN,
|
||||
+ LLM_ARCH_HIDREAM,
|
||||
+ LLM_ARCH_COSMOS,
|
||||
+ LLM_ARCH_LUMINA2,
|
||||
LLM_ARCH_UNKNOWN,
|
||||
};
|
||||
|
||||
@@ -258,6 +269,17 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
|
||||
{ LLM_ARCH_GRANITE, "granite" },
|
||||
{ LLM_ARCH_GRANITE_MOE, "granitemoe" },
|
||||
{ LLM_ARCH_CHAMELEON, "chameleon" },
|
||||
+ { LLM_ARCH_FLUX, "flux" },
|
||||
+ { LLM_ARCH_SD1, "sd1" },
|
||||
+ { LLM_ARCH_SDXL, "sdxl" },
|
||||
+ { LLM_ARCH_SD3, "sd3" },
|
||||
+ { LLM_ARCH_AURA, "aura" },
|
||||
+ { LLM_ARCH_LTXV, "ltxv" },
|
||||
+ { LLM_ARCH_HYVID, "hyvid" },
|
||||
+ { LLM_ARCH_WAN, "wan" },
|
||||
+ { LLM_ARCH_HIDREAM, "hidream" },
|
||||
+ { LLM_ARCH_COSMOS, "cosmos" },
|
||||
+ { LLM_ARCH_LUMINA2, "lumina2" },
|
||||
{ LLM_ARCH_UNKNOWN, "(unknown)" },
|
||||
};
|
||||
|
||||
@@ -1531,6 +1553,17 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
|
||||
{ LLM_TENSOR_ATTN_K_NORM, "blk.%d.attn_k_norm" },
|
||||
},
|
||||
},
|
||||
+ { LLM_ARCH_FLUX, {}},
|
||||
+ { LLM_ARCH_SD1, {}},
|
||||
+ { LLM_ARCH_SDXL, {}},
|
||||
+ { LLM_ARCH_SD3, {}},
|
||||
+ { LLM_ARCH_AURA, {}},
|
||||
+ { LLM_ARCH_LTXV, {}},
|
||||
+ { LLM_ARCH_HYVID, {}},
|
||||
+ { LLM_ARCH_WAN, {}},
|
||||
+ { LLM_ARCH_HIDREAM, {}},
|
||||
+ { LLM_ARCH_COSMOS, {}},
|
||||
+ { LLM_ARCH_LUMINA2, {}},
|
||||
{
|
||||
LLM_ARCH_UNKNOWN,
|
||||
{
|
||||
@@ -5403,6 +5436,25 @@ static void llm_load_hparams(
|
||||
// get general kv
|
||||
ml.get_key(LLM_KV_GENERAL_NAME, model.name, false);
|
||||
|
||||
+ // Disable LLM metadata for image models
|
||||
+ switch (model.arch) {
|
||||
+ case LLM_ARCH_FLUX:
|
||||
+ case LLM_ARCH_SD1:
|
||||
+ case LLM_ARCH_SDXL:
|
||||
+ case LLM_ARCH_SD3:
|
||||
+ case LLM_ARCH_AURA:
|
||||
+ case LLM_ARCH_LTXV:
|
||||
+ case LLM_ARCH_HYVID:
|
||||
+ case LLM_ARCH_WAN:
|
||||
+ case LLM_ARCH_HIDREAM:
|
||||
+ case LLM_ARCH_COSMOS:
|
||||
+ case LLM_ARCH_LUMINA2:
|
||||
+ model.ftype = ml.ftype;
|
||||
+ return;
|
||||
+ default:
|
||||
+ break;
|
||||
+ }
|
||||
+
|
||||
// get hparams kv
|
||||
ml.get_key(LLM_KV_VOCAB_SIZE, hparams.n_vocab, false) || ml.get_arr_n(LLM_KV_TOKENIZER_LIST, hparams.n_vocab);
|
||||
|
||||
@@ -18016,6 +18068,134 @@ static void llama_tensor_dequantize_internal(
|
||||
workers.clear();
|
||||
}
|
||||
|
||||
+static ggml_type img_tensor_get_type(quantize_state_internal & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype) {
|
||||
+ // Special function for quantizing image model tensors
|
||||
+ const std::string name = ggml_get_name(tensor);
|
||||
+ const llm_arch arch = qs.model.arch;
|
||||
+
|
||||
+ // Sanity check
|
||||
+ if (
|
||||
+ (name.find("model.diffusion_model.") != std::string::npos) ||
|
||||
+ (name.find("first_stage_model.") != std::string::npos) ||
|
||||
+ (name.find("single_transformer_blocks.") != std::string::npos) ||
|
||||
+ (name.find("joint_transformer_blocks.") != std::string::npos)
|
||||
+ ) {
|
||||
+ throw std::runtime_error("Invalid input GGUF file. This is not a supported UNET model");
|
||||
+ }
|
||||
+
|
||||
+ // Unsupported quant types - exclude all IQ quants for now
|
||||
+ if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ1_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ3_M || ftype == LLAMA_FTYPE_MOSTLY_Q4_0_4_4 ||
|
||||
+ ftype == LLAMA_FTYPE_MOSTLY_Q4_0_4_8 || ftype == LLAMA_FTYPE_MOSTLY_Q4_0_8_8) {
|
||||
+ throw std::runtime_error("Invalid quantization type for image model (Not supported)");
|
||||
+ }
|
||||
+
|
||||
+ if ( // Rules for to_v attention
|
||||
+ (name.find("attn_v.weight") != std::string::npos) ||
|
||||
+ (name.find(".to_v.weight") != std::string::npos) ||
|
||||
+ (name.find(".v.weight") != std::string::npos) ||
|
||||
+ (name.find(".attn.w1v.weight") != std::string::npos) ||
|
||||
+ (name.find(".attn.w2v.weight") != std::string::npos) ||
|
||||
+ (name.find("_attn.v_proj.weight") != std::string::npos)
|
||||
+ ){
|
||||
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {
|
||||
+ new_type = GGML_TYPE_Q3_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
|
||||
+ new_type = qs.i_attention_wv < 2 ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
|
||||
+ new_type = GGML_TYPE_Q5_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
|
||||
+ new_type = GGML_TYPE_Q6_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && qs.i_attention_wv < 4) {
|
||||
+ new_type = GGML_TYPE_Q5_K;
|
||||
+ }
|
||||
+ ++qs.i_attention_wv;
|
||||
+ } else if ( // Rules for fused qkv attention
|
||||
+ (name.find("attn_qkv.weight") != std::string::npos) ||
|
||||
+ (name.find("attn.qkv.weight") != std::string::npos) ||
|
||||
+ (name.find("attention.qkv.weight") != std::string::npos)
|
||||
+ ) {
|
||||
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
|
||||
+ new_type = GGML_TYPE_Q4_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {
|
||||
+ new_type = GGML_TYPE_Q5_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
|
||||
+ new_type = GGML_TYPE_Q6_K;
|
||||
+ }
|
||||
+ } else if ( // Rules for ffn
|
||||
+ (name.find("ffn_down") != std::string::npos) ||
|
||||
+ ((name.find("experts.") != std::string::npos) && (name.find(".w2.weight") != std::string::npos)) ||
|
||||
+ (name.find(".ffn.2.weight") != std::string::npos) || // is this even the right way around?
|
||||
+ (name.find(".ff.net.2.weight") != std::string::npos) ||
|
||||
+ (name.find(".mlp.layer2.weight") != std::string::npos) ||
|
||||
+ (name.find(".adaln_modulation_mlp.2.weight") != std::string::npos) ||
|
||||
+ (name.find(".feed_forward.w2.weight") != std::string::npos)
|
||||
+ ) {
|
||||
+ // TODO: add back `layer_info` with some model specific logic + logic further down
|
||||
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
|
||||
+ new_type = GGML_TYPE_Q4_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
|
||||
+ new_type = GGML_TYPE_Q5_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S) {
|
||||
+ new_type = GGML_TYPE_Q5_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {
|
||||
+ new_type = GGML_TYPE_Q6_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
|
||||
+ new_type = GGML_TYPE_Q6_K;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_0) {
|
||||
+ new_type = GGML_TYPE_Q4_1;
|
||||
+ }
|
||||
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_0) {
|
||||
+ new_type = GGML_TYPE_Q5_1;
|
||||
+ }
|
||||
+ ++qs.i_ffn_down;
|
||||
+ }
|
||||
+
|
||||
+ // Sanity check for row shape
|
||||
+ bool convert_incompatible_tensor = false;
|
||||
+ if (new_type == GGML_TYPE_Q2_K || new_type == GGML_TYPE_Q3_K || new_type == GGML_TYPE_Q4_K ||
|
||||
+ new_type == GGML_TYPE_Q5_K || new_type == GGML_TYPE_Q6_K) {
|
||||
+ int nx = tensor->ne[0];
|
||||
+ int ny = tensor->ne[1];
|
||||
+ if (nx % QK_K != 0) {
|
||||
+ LLAMA_LOG_WARN("\n\n%s : tensor cols %d x %d are not divisible by %d, required for %s", __func__, nx, ny, QK_K, ggml_type_name(new_type));
|
||||
+ convert_incompatible_tensor = true;
|
||||
+ } else {
|
||||
+ ++qs.n_k_quantized;
|
||||
+ }
|
||||
+ }
|
||||
+ if (convert_incompatible_tensor) {
|
||||
+ // TODO: Possibly reenable this in the future
|
||||
+ // switch (new_type) {
|
||||
+ // case GGML_TYPE_Q2_K:
|
||||
+ // case GGML_TYPE_Q3_K:
|
||||
+ // case GGML_TYPE_Q4_K: new_type = GGML_TYPE_Q5_0; break;
|
||||
+ // case GGML_TYPE_Q5_K: new_type = GGML_TYPE_Q5_1; break;
|
||||
+ // case GGML_TYPE_Q6_K: new_type = GGML_TYPE_Q8_0; break;
|
||||
+ // default: throw std::runtime_error("\nUnsupported tensor size encountered\n");
|
||||
+ // }
|
||||
+ new_type = GGML_TYPE_F16;
|
||||
+ LLAMA_LOG_WARN(" - using fallback quantization %s\n", ggml_type_name(new_type));
|
||||
+ ++qs.n_fallback;
|
||||
+ }
|
||||
+ return new_type;
|
||||
+}
|
||||
+
|
||||
static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype) {
|
||||
const std::string name = ggml_get_name(tensor);
|
||||
|
||||
@@ -18513,7 +18693,9 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
|
||||
if (llama_model_has_encoder(&model)) {
|
||||
n_attn_layer *= 3;
|
||||
}
|
||||
- GGML_ASSERT((qs.n_attention_wv == n_attn_layer) && "n_attention_wv is unexpected");
|
||||
+ if (model.arch != LLM_ARCH_HYVID) { // TODO: Check why this fails
|
||||
+ GGML_ASSERT((qs.n_attention_wv == n_attn_layer) && "n_attention_wv is unexpected");
|
||||
+ }
|
||||
}
|
||||
|
||||
size_t total_size_org = 0;
|
||||
@@ -18547,6 +18729,51 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
|
||||
ctx_outs[i_split] = gguf_init_empty();
|
||||
}
|
||||
gguf_add_tensor(ctx_outs[i_split], tensor);
|
||||
+ // SD3 pos_embed needs special fix as first dim is 1, which gets truncated here
|
||||
+ if (model.arch == LLM_ARCH_SD3) {
|
||||
+ const std::string name = ggml_get_name(tensor);
|
||||
+ if (name == "pos_embed" && tensor->ne[2] == 1) {
|
||||
+ const int n_dim = 3;
|
||||
+ gguf_set_tensor_ndim(ctx_outs[i_split], "pos_embed", n_dim);
|
||||
+ LLAMA_LOG_INFO("\n%s: Correcting pos_embed shape for SD3: [key:%s]\n", __func__, tensor->name);
|
||||
+ }
|
||||
+ }
|
||||
+ // same goes for auraflow
|
||||
+ if (model.arch == LLM_ARCH_AURA) {
|
||||
+ const std::string name = ggml_get_name(tensor);
|
||||
+ if (name == "positional_encoding" && tensor->ne[2] == 1) {
|
||||
+ const int n_dim = 3;
|
||||
+ gguf_set_tensor_ndim(ctx_outs[i_split], "positional_encoding", n_dim);
|
||||
+ LLAMA_LOG_INFO("\n%s: Correcting positional_encoding shape for AuraFlow: [key:%s]\n", __func__, tensor->name);
|
||||
+ }
|
||||
+ if (name == "register_tokens" && tensor->ne[2] == 1) {
|
||||
+ const int n_dim = 3;
|
||||
+ gguf_set_tensor_ndim(ctx_outs[i_split], "register_tokens", n_dim);
|
||||
+ LLAMA_LOG_INFO("\n%s: Correcting register_tokens shape for AuraFlow: [key:%s]\n", __func__, tensor->name);
|
||||
+ }
|
||||
+ }
|
||||
+ // conv3d fails due to max dims - unsure what to do here as we never even reach this check
|
||||
+ if (model.arch == LLM_ARCH_HYVID) {
|
||||
+ const std::string name = ggml_get_name(tensor);
|
||||
+ if (name == "img_in.proj.weight" && tensor->ne[5] != 1 ) {
|
||||
+ throw std::runtime_error("img_in.proj.weight size failed for HyVid");
|
||||
+ }
|
||||
+ }
|
||||
+ // All the modulation layers also have dim1, and I think conv3d fails here too but we segfaul way before that...
|
||||
+ if (model.arch == LLM_ARCH_WAN) {
|
||||
+ const std::string name = ggml_get_name(tensor);
|
||||
+ if (name.find(".modulation") != std::string::npos && tensor->ne[2] == 1) {
|
||||
+ const int n_dim = 3;
|
||||
+ gguf_set_tensor_ndim(ctx_outs[i_split], tensor->name, n_dim);
|
||||
+ LLAMA_LOG_INFO("\n%s: Correcting shape for Wan: [key:%s]\n", __func__, tensor->name);
|
||||
+ }
|
||||
+ // FLF2V model only
|
||||
+ if (name == "img_emb.emb_pos") {
|
||||
+ const int n_dim = 3;
|
||||
+ gguf_set_tensor_ndim(ctx_outs[i_split], tensor->name, n_dim);
|
||||
+ LLAMA_LOG_INFO("\n%s: Correcting shape for Wan FLF2V: [key:%s]\n", __func__, tensor->name);
|
||||
+ }
|
||||
+ }
|
||||
}
|
||||
|
||||
// Set split info if needed
|
||||
@@ -18647,6 +18874,110 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
|
||||
// do not quantize relative position bias (T5)
|
||||
quantize &= name.find("attn_rel_b.weight") == std::string::npos;
|
||||
|
||||
+ // rules for image models
|
||||
+ bool image_model = false;
|
||||
+ if (model.arch == LLM_ARCH_FLUX) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("txt_in.") == std::string::npos;
|
||||
+ quantize &= name.find("img_in.") == std::string::npos;
|
||||
+ quantize &= name.find("time_in.") == std::string::npos;
|
||||
+ quantize &= name.find("vector_in.") == std::string::npos;
|
||||
+ quantize &= name.find("guidance_in.") == std::string::npos;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_SD1 || model.arch == LLM_ARCH_SDXL) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("class_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("time_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("add_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("time_embed.") == std::string::npos;
|
||||
+ quantize &= name.find("label_emb.") == std::string::npos;
|
||||
+ quantize &= name.find("conv_in.") == std::string::npos;
|
||||
+ quantize &= name.find("conv_out.") == std::string::npos;
|
||||
+ quantize &= name != "input_blocks.0.0.weight";
|
||||
+ quantize &= name != "out.2.weight";
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_SD3) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ quantize &= name.find("time_text_embed.") == std::string::npos;
|
||||
+ quantize &= name.find("context_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("t_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("y_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("x_embedder.") == std::string::npos;
|
||||
+ quantize &= name != "proj_out.weight";
|
||||
+ quantize &= name != "pos_embed";
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_AURA) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("t_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("init_x_linear.") == std::string::npos;
|
||||
+ quantize &= name != "modF.1.weight";
|
||||
+ quantize &= name != "cond_seq_linear.weight";
|
||||
+ quantize &= name != "final_linear.weight";
|
||||
+ quantize &= name != "final_linear.weight";
|
||||
+ quantize &= name != "positional_encoding";
|
||||
+ quantize &= name != "register_tokens";
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_LTXV) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("adaln_single.") == std::string::npos;
|
||||
+ quantize &= name.find("caption_projection.") == std::string::npos;
|
||||
+ quantize &= name.find("patchify_proj.") == std::string::npos;
|
||||
+ quantize &= name.find("proj_out.") == std::string::npos;
|
||||
+ quantize &= name.find("scale_shift_table") == std::string::npos; // last block too
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_HYVID) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("txt_in.") == std::string::npos;
|
||||
+ quantize &= name.find("img_in.") == std::string::npos;
|
||||
+ quantize &= name.find("time_in.") == std::string::npos;
|
||||
+ quantize &= name.find("vector_in.") == std::string::npos;
|
||||
+ quantize &= name.find("guidance_in.") == std::string::npos;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_WAN) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("modulation.") == std::string::npos;
|
||||
+ quantize &= name.find("patch_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("text_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("time_projection.") == std::string::npos;
|
||||
+ quantize &= name.find("time_embedding.") == std::string::npos;
|
||||
+ quantize &= name.find("img_emb.") == std::string::npos;
|
||||
+ quantize &= name.find("head.") == std::string::npos;
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_HIDREAM) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("p_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("t_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("x_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ quantize &= name.find(".ff_i.gate.weight") == std::string::npos;
|
||||
+ quantize &= name.find("caption_projection.") == std::string::npos;
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_COSMOS) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("p_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("t_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("t_embedding_norm.") == std::string::npos;
|
||||
+ quantize &= name.find("x_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("pos_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ }
|
||||
+ if (model.arch == LLM_ARCH_LUMINA2) {
|
||||
+ image_model = true;
|
||||
+ quantize &= name.find("t_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("x_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("final_layer.") == std::string::npos;
|
||||
+ quantize &= name.find("cap_embedder.") == std::string::npos;
|
||||
+ quantize &= name.find("context_refiner.") == std::string::npos;
|
||||
+ quantize &= name.find("noise_refiner.") == std::string::npos;
|
||||
+ }
|
||||
+ // ignore 3D/4D tensors for image models as the code was never meant to handle these
|
||||
+ if (image_model) {
|
||||
+ quantize &= ggml_n_dims(tensor) == 2;
|
||||
+ }
|
||||
+
|
||||
enum ggml_type new_type;
|
||||
void * new_data;
|
||||
size_t new_size;
|
||||
@@ -18655,6 +18986,9 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
|
||||
new_type = default_type;
|
||||
|
||||
// get more optimal quantization type based on the tensor shape, layer, etc.
|
||||
+ if (image_model) {
|
||||
+ new_type = img_tensor_get_type(qs, new_type, tensor, ftype);
|
||||
+ } else {
|
||||
if (!params->pure && ggml_is_quantized(default_type)) {
|
||||
new_type = llama_tensor_get_type(qs, new_type, tensor, ftype);
|
||||
}
|
||||
@@ -18664,6 +18998,7 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
|
||||
if (params->output_tensor_type < GGML_TYPE_COUNT && strcmp(tensor->name, "output.weight") == 0) {
|
||||
new_type = params->output_tensor_type;
|
||||
}
|
||||
+ }
|
||||
|
||||
// If we've decided to quantize to the same type the tensor is already
|
||||
// in then there's nothing to do.
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/usr/bin/python3
|
||||
import os
|
||||
import sys
|
||||
import gguf
|
||||
|
||||
def read_tensors(path):
|
||||
reader = gguf.GGUFReader(path)
|
||||
for tensor in reader.tensors:
|
||||
if tensor.tensor_type == gguf.GGMLQuantizationType.F32:
|
||||
continue
|
||||
print(f"{str(tensor.tensor_type):32}: {tensor.name}")
|
||||
|
||||
try:
|
||||
path = sys.argv[1]
|
||||
assert os.path.isfile(path), "Invalid path"
|
||||
print(f"input: {path}")
|
||||
except Exception as e:
|
||||
input(f"failed: {e}")
|
||||
else:
|
||||
read_tensors(path)
|
||||
input()
|
||||
Reference in New Issue
Block a user