Updated Out of Date Files

God bless city96. Now the clip nodes can load qwen3 gguf models and probably a bunch of other cool fixes. 👍
This commit is contained in:
Maxed-Out-99
2025-12-15 11:43:03 -08:00
parent 34e68d5d54
commit 7a2b7ff41b
9 changed files with 893 additions and 11 deletions
+54 -1
View File
@@ -23,7 +23,7 @@ def dequantize_tensor(tensor, dtype=None, dequant_dtype=None):
return dequantize(tensor.data, qtype, oshape, dtype=dequant_dtype).to(dtype)
else:
# this is incredibly slow
tqdm.write(f"Falling back to numpy dequant for qtype: {qtype}")
tqdm.write(f"Falling back to numpy dequant for qtype: {getattr(qtype, 'name', repr(qtype))}")
new = gguf.quants.dequantize(tensor.cpu().numpy(), qtype)
return torch.from_numpy(new).to(tensor.device, dtype=dtype)
@@ -48,6 +48,10 @@ def to_uint32(x):
x = x.view(torch.uint8).to(torch.int32)
return (x[:, 0] | x[:, 1] << 8 | x[:, 2] << 16 | x[:, 3] << 24).unsqueeze(1)
def to_uint16(x):
x = x.view(torch.uint8).to(torch.int32)
return (x[:, 0] | x[:, 1] << 8).unsqueeze(1)
def split_block_dims(blocks, *args):
n_max = blocks.shape[1]
dims = list(args) + [n_max - sum(args)]
@@ -233,6 +237,53 @@ def dequantize_blocks_Q2_K(blocks, block_size, type_size, dtype=None):
return qs.reshape((n_blocks, -1))
# IQ quants
KVALUES = torch.tensor([-127, -104, -83, -65, -49, -35, -22, -10, 1, 13, 25, 38, 53, 69, 89, 113], dtype=torch.int8)
def dequantize_blocks_IQ4_NL(blocks, block_size, type_size, dtype=None):
n_blocks = blocks.shape[0]
d, qs = split_block_dims(blocks, 2)
d = d.view(torch.float16).to(dtype)
qs = qs.reshape((n_blocks, -1, 1, block_size//2)) >> torch.tensor([0, 4], device=d.device, dtype=torch.uint8).reshape((1, 1, 2, 1))
qs = (qs & 0x0F).reshape((n_blocks, -1, 1)).to(torch.int32)
kvalues = KVALUES.to(qs.device).expand(*qs.shape[:-1], 16)
qs = torch.gather(kvalues, dim=-1, index=qs).reshape((n_blocks, -1))
del kvalues # should still be view, but just to be safe
return (d * qs)
def dequantize_blocks_IQ4_XS(blocks, block_size, type_size, dtype=None):
n_blocks = blocks.shape[0]
d, scales_h, scales_l, qs = split_block_dims(blocks, 2, 2, QK_K // 64)
d = d.view(torch.float16).to(dtype)
scales_h = to_uint16(scales_h)
shift_a = torch.tensor([0, 4], device=d.device, dtype=torch.uint8).reshape((1, 1, 2))
shift_b = torch.tensor([2 * i for i in range(QK_K // 32)], device=d.device, dtype=torch.uint8).reshape((1, -1, 1))
scales_l = scales_l.reshape((n_blocks, -1, 1)) >> shift_a.reshape((1, 1, 2))
scales_h = scales_h.reshape((n_blocks, -1, 1)) >> shift_b.reshape((1, -1, 1))
scales_l = scales_l.reshape((n_blocks, -1)) & 0x0F
scales_h = scales_h.reshape((n_blocks, -1)).to(torch.uint8) & 0x03
scales = (scales_l | (scales_h << 4)).to(torch.int8) - 32
dl = (d * scales.to(dtype)).reshape((n_blocks, -1, 1))
qs = qs.reshape((n_blocks, -1, 1, 16)) >> shift_a.reshape((1, 1, 2, 1))
qs = qs.reshape((n_blocks, -1, 32, 1)) & 0x0F
kvalues = KVALUES.to(qs.device).expand(*qs.shape[:-1], 16)
qs = torch.gather(kvalues, dim=-1, index=qs.to(torch.int32)).reshape((n_blocks, -1, 32))
del kvalues # see IQ4_NL
del shift_a
del shift_b
return (dl * qs).reshape((n_blocks, -1))
dequantize_functions = {
gguf.GGMLQuantizationType.BF16: dequantize_blocks_BF16,
gguf.GGMLQuantizationType.Q8_0: dequantize_blocks_Q8_0,
@@ -245,4 +296,6 @@ dequantize_functions = {
gguf.GGMLQuantizationType.Q4_K: dequantize_blocks_Q4_K,
gguf.GGMLQuantizationType.Q3_K: dequantize_blocks_Q3_K,
gguf.GGMLQuantizationType.Q2_K: dequantize_blocks_Q2_K,
gguf.GGMLQuantizationType.IQ4_NL: dequantize_blocks_IQ4_NL,
gguf.GGMLQuantizationType.IQ4_XS: dequantize_blocks_IQ4_XS,
}
+151 -7
View File
@@ -3,12 +3,15 @@ import warnings
import logging
import torch
import gguf
import re
import os
from .ops import GGMLTensor
from .dequant import is_quantized, dequantize_tensor
IMG_ARCH_LIST = {"flux", "sd1", "sdxl", "sd3", "aura", "hidream", "cosmos", "ltxv", "hyvid", "wan"}
TXT_ARCH_LIST = {"t5", "t5encoder", "llama"}
IMG_ARCH_LIST = {"flux", "sd1", "sdxl", "sd3", "aura", "hidream", "cosmos", "ltxv", "hyvid", "wan", "lumina2", "qwen_image"}
TXT_ARCH_LIST = {"t5", "t5encoder", "llama", "qwen2vl", "qwen3", "qwen3vl"}
VIS_TYPE_LIST = {"clip-vision", "mmproj"}
def get_orig_shape(reader, tensor_name):
field_key = f"comfy.gguf.orig_shape.{tensor_name}"
@@ -70,9 +73,10 @@ def gguf_sd_loader(path, handle_prefix="model.diffusion_model.", return_arch=Fal
# detect and verify architecture
compat = None
arch_str = get_field(reader, "general.architecture", str)
if arch_str in [None, "pig"]:
type_str = get_field(reader, "general.type", str)
if arch_str in [None, "pig", "cow"]:
if is_text_model:
raise ValueError(f"This text model is incompatible with llama.cpp!\nConsider using the safetensors version\n({path})")
raise ValueError(f"This gguf file is incompatible with llama.cpp!\nConsider using safetensors or a compatible gguf file\n({path})")
compat = "sd.cpp" if arch_str is None else arch_str
# import here to avoid changes to convert.py breaking regular models
from .tools.convert import detect_arch
@@ -81,7 +85,8 @@ def gguf_sd_loader(path, handle_prefix="model.diffusion_model.", return_arch=Fal
except Exception as e:
raise ValueError(f"This model is not currently supported - ({e})")
elif arch_str not in TXT_ARCH_LIST and is_text_model:
raise ValueError(f"Unexpected text model architecture type in GGUF file: {arch_str!r}")
if type_str not in VIS_TYPE_LIST:
raise ValueError(f"Unexpected text model architecture type in GGUF file: {arch_str!r}")
elif arch_str not in IMG_ARCH_LIST and not is_text_model:
raise ValueError(f"Unexpected architecture type in GGUF file: {arch_str!r}")
@@ -152,6 +157,9 @@ T5_SD_MAP = {
LLAMA_SD_MAP = {
"blk.": "model.layers.",
"attn_norm": "input_layernorm",
"attn_q_norm.": "self_attn.q_norm.",
"attn_k_norm.": "self_attn.k_norm.",
"attn_v_norm.": "self_attn.v_norm.",
"attn_q": "self_attn.q_proj",
"attn_k": "self_attn.k_proj",
"attn_v": "self_attn.v_proj",
@@ -165,6 +173,19 @@ LLAMA_SD_MAP = {
"output.weight": "lm_head.weight",
}
CLIP_VISION_SD_MAP = {
"mm.": "visual.merger.mlp.",
"v.post_ln.": "visual.merger.ln_q.",
"v.patch_embd": "visual.patch_embed.proj",
"v.blk.": "visual.blocks.",
"ffn_up": "mlp.up_proj",
"ffn_down": "mlp.down_proj",
"ffn_gate": "mlp.gate_proj",
"attn_out.": "attn.proj.",
"ln1.": "norm1.",
"ln2.": "norm2.",
}
def sd_map_replace(raw_sd, key_map):
sd = {}
for k,v in raw_sd.items():
@@ -185,6 +206,79 @@ def llama_permute(raw_sd, n_head, n_head_kv):
sd[k] = v
return sd
def strip_quant_suffix(name):
pattern = r"[-_]?(?:ud-)?i?q[0-9]_[a-z0-9_\-]{1,8}$"
match = re.search(pattern, name, re.IGNORECASE)
if match:
name = name[:match.start()]
return name
def gguf_mmproj_loader(path):
# Reverse version of Qwen2VLVisionModel.modify_tensors
logging.info("Attenpting to find mmproj file for text encoder...")
# get name to match w/o quant suffix
tenc_fname = os.path.basename(path)
tenc = os.path.splitext(tenc_fname)[0].lower()
tenc = strip_quant_suffix(tenc)
# try and find matching mmproj
target = []
root = os.path.dirname(path)
for fname in os.listdir(root):
name, ext = os.path.splitext(fname)
if ext.lower() != ".gguf":
continue
if "mmproj" not in name.lower():
continue
if tenc in name.lower():
target.append(fname)
if len(target) == 0:
logging.error(f"Error: Can't find mmproj file for '{tenc_fname}' (matching:'{tenc}')! Qwen-Image-Edit will be broken!")
return {}
if len(target) > 1:
logging.error(f"Ambiguous mmproj for text encoder '{tenc_fname}', will use first match.")
logging.info(f"Using mmproj '{target[0]}' for text encoder '{tenc_fname}'.")
target = os.path.join(root, target[0])
vsd = gguf_sd_loader(target, is_text_model=True)
# concat 4D to 5D
if "v.patch_embd.weight.1" in vsd:
w1 = dequantize_tensor(vsd.pop("v.patch_embd.weight"), dtype=torch.float32)
w2 = dequantize_tensor(vsd.pop("v.patch_embd.weight.1"), dtype=torch.float32)
vsd["v.patch_embd.weight"] = torch.stack([w1, w2], dim=2)
# run main replacement
vsd = sd_map_replace(vsd, CLIP_VISION_SD_MAP)
# handle split Q/K/V
if "visual.blocks.0.attn_q.weight" in vsd:
attns = {}
# filter out attentions + group
for k,v in vsd.items():
if any(x in k for x in ["attn_q", "attn_k", "attn_v"]):
k_attn, k_name = k.rsplit(".attn_", 1)
k_attn += ".attn.qkv." + k_name.split(".")[-1]
if k_attn not in attns:
attns[k_attn] = {}
attns[k_attn][k_name] = dequantize_tensor(
v, dtype=(torch.bfloat16 if is_quantized(v) else torch.float16)
)
# recombine
for k,v in attns.items():
suffix = k.split(".")[-1]
vsd[k] = torch.cat([
v[f"q.{suffix}"],
v[f"k.{suffix}"],
v[f"v.{suffix}"],
], dim=0)
del attns
return vsd
def gguf_tokenizer_loader(path, temb_shape):
# convert gguf tokenizer to spiece
logging.info("Attempting to recreate sentencepiece tokenizer from GGUF file metadata...")
@@ -233,6 +327,49 @@ def gguf_tokenizer_loader(path, temb_shape):
del reader
return torch.ByteTensor(list(spm.SerializeToString()))
def gguf_tekken_tokenizer_loader(path, temb_shape):
# convert ggml (hf) tokenizer metadata to tekken/comfy data
logging.info("Attempting to recreate tekken tokenizer from GGUF file metadata...")
import json
import base64
from transformers.convert_slow_tokenizer import bytes_to_unicode
reader = gguf.GGUFReader(path)
model_str = get_field(reader, "tokenizer.ggml.model", str)
if model_str == "gpt2":
if temb_shape == (131072, 5120): # probably Mistral
data = {
"config": {"num_vocab_tokens": 150000, "default_vocab_size": 131072},
"vocab": [],
"special_tokens": [],
}
else:
raise NotImplementedError("Unknown model, can't set tokenizer!")
else:
raise NotImplementedError("Unknown model, can't set tokenizer!")
tokens = get_list_field(reader, "tokenizer.ggml.tokens", str)
toktypes = get_list_field(reader, "tokenizer.ggml.token_type", int)
decoder = {v: k for k, v in bytes_to_unicode().items()}
for idx, (token, toktype) in enumerate(zip(tokens, toktypes)):
if toktype == 3:
data["special_tokens"].append(
{'rank': idx, 'token_str': token, 'is_control': True}
)
else:
tok = bytes([decoder[char] for char in token])
data["vocab"].append({
"rank": len(data["vocab"]),
"token_bytes": base64.b64encode(tok).decode("ascii"),
"token_str": tok.decode("utf-8", errors="replace") # ?
})
logging.info(f"Created tekken tokenizer with vocab size of {len(data['vocab'])} (+{len(data['special_tokens'])})")
del reader
return torch.ByteTensor(list(json.dumps(data).encode('utf-8')))
def gguf_clip_loader(path):
sd, arch = gguf_sd_loader(path, return_arch=True, is_text_model=True)
if arch in {"t5", "t5encoder"}:
@@ -244,15 +381,22 @@ def gguf_clip_loader(path):
logging.warning(f"Dequantizing {temb_key} to prevent runtime OOM.")
sd[temb_key] = dequantize_tensor(sd[temb_key], dtype=torch.float16)
sd = sd_map_replace(sd, T5_SD_MAP)
elif arch in {"llama"}:
elif arch in {"llama", "qwen2vl", "qwen3", "qwen3vl"}:
# TODO: pass model_options["vocab_size"] to loader somehow
temb_key = "token_embd.weight"
if temb_key in sd and sd[temb_key].shape[0] >= (64 * 1024):
if arch == "llama" and sd[temb_key].shape == (131072, 5120):
# non-standard Comfy-Org tokenizer
sd["tekken_model"] = gguf_tekken_tokenizer_loader(path, sd[temb_key].shape)
# See note above for T5.
logging.warning(f"Dequantizing {temb_key} to prevent runtime OOM.")
sd[temb_key] = dequantize_tensor(sd[temb_key], dtype=torch.float16)
sd = sd_map_replace(sd, LLAMA_SD_MAP)
sd = llama_permute(sd, 32, 8) # L3
if arch == "llama":
sd = llama_permute(sd, 32, 8) # L3 / Mistral
if arch == "qwen2vl":
vsd = gguf_mmproj_loader(path)
sd.update(vsd)
else:
pass
return sd
+1 -1
View File
@@ -153,7 +153,7 @@ class GGMLLayer(torch.nn.Module):
# Take into account space required for dequantizing the largest tensor
if self.largest_layer:
shape = getattr(self.weight, "tensor_shape", self.weight.shape)
dtype = self.dequant_dtype or torch.float16
dtype = self.dequant_dtype if self.dequant_dtype and self.dequant_dtype != "target" else torch.float16
temp = torch.empty(*shape, device=torch.device("meta"), dtype=dtype)
destination[prefix + "temp.weight"] = temp
+93
View File
@@ -0,0 +1,93 @@
## Converting initial model
To convert your initial safetensors/ckpt model to FP16/BF16 GGUF, run the following command:
```
python convert.py --src E:\models\unet\flux1-dev.safetensors
```
Make sure `gguf>=0.13.0` is installed for this step. Optionally, specify the output gguf file with the `--dst` arg.
> [!NOTE]
> Do not use the diffusers UNET format for flux, it won't work, use the default/reference checkpoint key format. This is due to q/k/v being merged into one qkv key.
> You can convert it by loading it in ComfyUI and saving it using the built-in "ModelSave" node.
> [!WARNING]
> For hunyuan video/wan 2.1, you will see a warning about 5D tensors. This means the script will save a **non functional** model to disk first, that you can quantize. I recommend saving these in a separate `raw` folder to avoid confusion.
>
> After quantization, you will have to run `fix_5d_tensor.py` manually to add back the missing key that was saved by the conversion code.
## Quantizing using custom llama.cpp
Depending on your git settings, you may need to run the following script first in order to make sure the patch file is valid. It will convert Windows (CRLF) line endings to Unix (LF) ones.
```
python fix_lines_ending.py
```
Git clone llama.cpp into the current folder:
```
git clone https://github.com/ggerganov/llama.cpp
```
Check out the correct branch, then apply the custom patch needed to add image model support to the repo you just cloned.
```
cd llama.cpp
git checkout tags/b3962
git apply ..\lcpp.patch
```
Compile the llama-quantize binary. This example uses cmake, on linux you can just use make.
### Visual Studio 2019, Linux, etc...
```
mkdir build
cmake -B build
cmake --build build --config Debug -j10 --target llama-quantize
cd ..
```
### Visual Studio 2022
```
mkdir build
cmake -B build -DCMAKE_CXX_STANDARD=17 -DCMAKE_CXX_STANDARD_REQUIRED=ON -DCMAKE_CXX_FLAGS="-std=c++17"
```
Edit the `llama.cpp\common\log.cpp` file, inserts two lines after the existing first line:
```
#include "log.h"
#define _SILENCE_CXX23_CHRONO_DEPRECATION_WARNING
#include <chrono>
```
Then you can build the project:
```
cmake --build build --config Debug -j10 --target llama-quantize
cd ..
```
### Quantize your model
Now you can use the newly build binary to quantize your model to the desired format:
```
llama.cpp\build\bin\Debug\llama-quantize.exe E:\models\unet\flux1-dev-BF16.gguf E:\models\unet\flux1-dev-Q4_K_S.gguf Q4_K_S
```
You can extract the patch again with `git diff src\llama.cpp > lcpp.patch` if you wish to change something and contribute back.
> [!WARNING]
> For hunyuan video/wan 2.1, you will have to run `fix_5d_tensor.py` after the quantization step is done.
>
> Example usage: `fix_5d_tensors.py --src E:\models\video\raw\wan2.1-t2v-1.3b-Q8_0.gguf --dst E:\models\video\wan2.1-t2v-1.3b-Q8_0.gguf`
>
> By default, this also saves a `fix_5d_tensors_[arch].safetensors` file in the `ComfyUI-GGUF/tools` folder, it's recommended to delete this after all models have been converted.
> [!NOTE]
> Do not quantize SDXL / SD1 / other Conv2D heavy models. If you do, make sure to **extract the UNET model first**.
>This should be obvious, but also don't use the resulting llama-quantize binary with LLMs.
+9 -2
View File
@@ -139,8 +139,14 @@ class ModelSD1(ModelTemplate):
), # Non-diffusers
]
# The architectures are checked in order and the first successful match terminates the search.
arch_list = [ModelFlux, ModelSD3, ModelAura, ModelHiDream, CosmosPredict2, ModelLTXV, ModelHyVid, ModelWan, ModelSDXL, ModelSD1]
class ModelLumina2(ModelTemplate):
arch = "lumina2"
keys_detect = [
("cap_embedder.1.weight", "context_refiner.0.attention.qkv.weight")
]
arch_list = [ModelFlux, ModelSD3, ModelAura, ModelHiDream, CosmosPredict2,
ModelLTXV, ModelHyVid, ModelWan, ModelSDXL, ModelSD1, ModelLumina2]
def is_model_arch(model, state_dict):
# check if model is correct
@@ -356,3 +362,4 @@ def convert_file(path, dst_path=None, interact=True, overwrite=False):
if __name__ == "__main__":
args = parse_args()
convert_file(args.src, args.dst)
+82
View File
@@ -0,0 +1,82 @@
# (c) City96 || Apache-2.0 (apache.org/licenses/LICENSE-2.0)
import os
import gguf
import torch
import argparse
from tqdm import tqdm
from safetensors.torch import load_file
def get_args():
parser = argparse.ArgumentParser()
parser.add_argument("--src", required=True)
parser.add_argument("--dst", required=True)
parser.add_argument("--fix", required=False, help="Defaults to ./fix_5d_tensors_[arch].pt")
parser.add_argument("--overwrite", action="store_true")
args = parser.parse_args()
if not os.path.isfile(args.src):
parser.error(f"Invalid source file '{args.src}'")
if not args.overwrite and os.path.exists(args.dst):
parser.error(f"Output exists, use '--overwrite' ({args.dst})")
return args
def get_arch_str(reader):
field = reader.get_field("general.architecture")
return str(field.parts[field.data[-1]], encoding="utf-8")
def get_file_type(reader):
field = reader.get_field("general.file_type")
ft = int(field.parts[field.data[-1]])
return gguf.LlamaFileType(ft)
if __name__ == "__main__":
args = get_args()
# read existing
reader = gguf.GGUFReader(args.src)
arch = get_arch_str(reader)
file_type = get_file_type(reader)
print(f"Detected arch: '{arch}' (ftype: {str(file_type)})")
# prep fix
if args.fix is None:
args.fix = f"./fix_5d_tensors_{arch}.safetensors"
if not os.path.isfile(args.fix):
raise OSError(f"No 5D tensor fix file: {args.fix}")
sd5d = load_file(args.fix)
sd5d = {k:v.numpy() for k,v in sd5d.items()}
print("5D tensors:", sd5d.keys())
# prep output
writer = gguf.GGUFWriter(path=None, arch=arch)
writer.add_quantization_version(gguf.GGML_QUANT_VERSION)
writer.add_file_type(file_type)
added = []
def add_extra_key(writer, key, data):
global added
data_qtype = gguf.GGMLQuantizationType.F32
data = gguf.quants.quantize(data, data_qtype)
tqdm.write(f"Adding key {key} ({data.shape})")
writer.add_tensor(key, data, raw_dtype=data_qtype)
added.append(key)
# main loop to add missing 5D tensor(s)
for tensor in tqdm(reader.tensors):
writer.add_tensor(tensor.name, tensor.data, raw_dtype=tensor.tensor_type)
key5d = tensor.name.replace(".bias", ".weight")
if key5d in sd5d.keys():
add_extra_key(writer, key5d, sd5d[key5d])
# brute force for any missed
for key, data in sd5d.items():
if key not in added:
add_extra_key(writer, key, data)
writer.write_header_to_file(path=args.dst)
writer.write_kv_data_to_file()
writer.write_tensors_to_file(progress=True)
writer.close()
+31
View File
@@ -0,0 +1,31 @@
import os
files = ["lcpp.patch", "lcpp_sd3.patch"]
def has_unix_line_endings(file_path):
try:
with open(file_path, 'rb') as file:
content = file.read()
return b'\r\n' not in content
except Exception as e:
print(f"Error checking '{file_path}': {e}")
return False
def convert_to_linux_format(file_path):
try:
with open(file_path, 'rb') as file:
content = file.read().replace(b'\r\n', b'\n')
with open(file_path, 'wb') as file:
file.write(content)
print(f"'{file_path}' converted to Linux line endings (LF).")
except Exception as e:
print(f"Error processing '{file_path}': {e}")
for file in files:
if os.path.exists(file):
if has_unix_line_endings(file):
print(f"'{file}' already has Unix line endings (LF). No conversion needed.")
else:
convert_to_linux_format(file)
else:
print(f"File '{file}' does not exist.")
+451
View File
@@ -0,0 +1,451 @@
diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h
index de3c706f..0267c1fa 100644
--- a/ggml/include/ggml.h
+++ b/ggml/include/ggml.h
@@ -223,7 +223,7 @@
#define GGML_MAX_OP_PARAMS 64
#ifndef GGML_MAX_NAME
-# define GGML_MAX_NAME 64
+# define GGML_MAX_NAME 128
#endif
#define GGML_DEFAULT_N_THREADS 4
@@ -2449,6 +2449,7 @@ extern "C" {
// manage tensor info
GGML_API void gguf_add_tensor(struct gguf_context * ctx, const struct ggml_tensor * tensor);
+ GGML_API void gguf_set_tensor_ndim(struct gguf_context * ctx, const char * name, int n_dim);
GGML_API void gguf_set_tensor_type(struct gguf_context * ctx, const char * name, enum ggml_type type);
GGML_API void gguf_set_tensor_data(struct gguf_context * ctx, const char * name, const void * data, size_t size);
diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c
index b16c462f..6d1568f1 100644
--- a/ggml/src/ggml.c
+++ b/ggml/src/ggml.c
@@ -22960,6 +22960,14 @@ void gguf_add_tensor(
ctx->header.n_tensors++;
}
+void gguf_set_tensor_ndim(struct gguf_context * ctx, const char * name, const int n_dim) {
+ const int idx = gguf_find_tensor(ctx, name);
+ if (idx < 0) {
+ GGML_ABORT("tensor not found");
+ }
+ ctx->infos[idx].n_dims = n_dim;
+}
+
void gguf_set_tensor_type(struct gguf_context * ctx, const char * name, enum ggml_type type) {
const int idx = gguf_find_tensor(ctx, name);
if (idx < 0) {
diff --git a/src/llama.cpp b/src/llama.cpp
index 24e1f1f0..25db4c69 100644
--- a/src/llama.cpp
+++ b/src/llama.cpp
@@ -205,6 +205,17 @@ enum llm_arch {
LLM_ARCH_GRANITE,
LLM_ARCH_GRANITE_MOE,
LLM_ARCH_CHAMELEON,
+ LLM_ARCH_FLUX,
+ LLM_ARCH_SD1,
+ LLM_ARCH_SDXL,
+ LLM_ARCH_SD3,
+ LLM_ARCH_AURA,
+ LLM_ARCH_LTXV,
+ LLM_ARCH_HYVID,
+ LLM_ARCH_WAN,
+ LLM_ARCH_HIDREAM,
+ LLM_ARCH_COSMOS,
+ LLM_ARCH_LUMINA2,
LLM_ARCH_UNKNOWN,
};
@@ -258,6 +269,17 @@ static const std::map<llm_arch, const char *> LLM_ARCH_NAMES = {
{ LLM_ARCH_GRANITE, "granite" },
{ LLM_ARCH_GRANITE_MOE, "granitemoe" },
{ LLM_ARCH_CHAMELEON, "chameleon" },
+ { LLM_ARCH_FLUX, "flux" },
+ { LLM_ARCH_SD1, "sd1" },
+ { LLM_ARCH_SDXL, "sdxl" },
+ { LLM_ARCH_SD3, "sd3" },
+ { LLM_ARCH_AURA, "aura" },
+ { LLM_ARCH_LTXV, "ltxv" },
+ { LLM_ARCH_HYVID, "hyvid" },
+ { LLM_ARCH_WAN, "wan" },
+ { LLM_ARCH_HIDREAM, "hidream" },
+ { LLM_ARCH_COSMOS, "cosmos" },
+ { LLM_ARCH_LUMINA2, "lumina2" },
{ LLM_ARCH_UNKNOWN, "(unknown)" },
};
@@ -1531,6 +1553,17 @@ static const std::map<llm_arch, std::map<llm_tensor, const char *>> LLM_TENSOR_N
{ LLM_TENSOR_ATTN_K_NORM, "blk.%d.attn_k_norm" },
},
},
+ { LLM_ARCH_FLUX, {}},
+ { LLM_ARCH_SD1, {}},
+ { LLM_ARCH_SDXL, {}},
+ { LLM_ARCH_SD3, {}},
+ { LLM_ARCH_AURA, {}},
+ { LLM_ARCH_LTXV, {}},
+ { LLM_ARCH_HYVID, {}},
+ { LLM_ARCH_WAN, {}},
+ { LLM_ARCH_HIDREAM, {}},
+ { LLM_ARCH_COSMOS, {}},
+ { LLM_ARCH_LUMINA2, {}},
{
LLM_ARCH_UNKNOWN,
{
@@ -5403,6 +5436,25 @@ static void llm_load_hparams(
// get general kv
ml.get_key(LLM_KV_GENERAL_NAME, model.name, false);
+ // Disable LLM metadata for image models
+ switch (model.arch) {
+ case LLM_ARCH_FLUX:
+ case LLM_ARCH_SD1:
+ case LLM_ARCH_SDXL:
+ case LLM_ARCH_SD3:
+ case LLM_ARCH_AURA:
+ case LLM_ARCH_LTXV:
+ case LLM_ARCH_HYVID:
+ case LLM_ARCH_WAN:
+ case LLM_ARCH_HIDREAM:
+ case LLM_ARCH_COSMOS:
+ case LLM_ARCH_LUMINA2:
+ model.ftype = ml.ftype;
+ return;
+ default:
+ break;
+ }
+
// get hparams kv
ml.get_key(LLM_KV_VOCAB_SIZE, hparams.n_vocab, false) || ml.get_arr_n(LLM_KV_TOKENIZER_LIST, hparams.n_vocab);
@@ -18016,6 +18068,134 @@ static void llama_tensor_dequantize_internal(
workers.clear();
}
+static ggml_type img_tensor_get_type(quantize_state_internal & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype) {
+ // Special function for quantizing image model tensors
+ const std::string name = ggml_get_name(tensor);
+ const llm_arch arch = qs.model.arch;
+
+ // Sanity check
+ if (
+ (name.find("model.diffusion_model.") != std::string::npos) ||
+ (name.find("first_stage_model.") != std::string::npos) ||
+ (name.find("single_transformer_blocks.") != std::string::npos) ||
+ (name.find("joint_transformer_blocks.") != std::string::npos)
+ ) {
+ throw std::runtime_error("Invalid input GGUF file. This is not a supported UNET model");
+ }
+
+ // Unsupported quant types - exclude all IQ quants for now
+ if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS ||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ1_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL ||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S ||
+ ftype == LLAMA_FTYPE_MOSTLY_IQ3_M || ftype == LLAMA_FTYPE_MOSTLY_Q4_0_4_4 ||
+ ftype == LLAMA_FTYPE_MOSTLY_Q4_0_4_8 || ftype == LLAMA_FTYPE_MOSTLY_Q4_0_8_8) {
+ throw std::runtime_error("Invalid quantization type for image model (Not supported)");
+ }
+
+ if ( // Rules for to_v attention
+ (name.find("attn_v.weight") != std::string::npos) ||
+ (name.find(".to_v.weight") != std::string::npos) ||
+ (name.find(".v.weight") != std::string::npos) ||
+ (name.find(".attn.w1v.weight") != std::string::npos) ||
+ (name.find(".attn.w2v.weight") != std::string::npos) ||
+ (name.find("_attn.v_proj.weight") != std::string::npos)
+ ){
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {
+ new_type = GGML_TYPE_Q3_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
+ new_type = qs.i_attention_wv < 2 ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
+ new_type = GGML_TYPE_Q5_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
+ new_type = GGML_TYPE_Q6_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && qs.i_attention_wv < 4) {
+ new_type = GGML_TYPE_Q5_K;
+ }
+ ++qs.i_attention_wv;
+ } else if ( // Rules for fused qkv attention
+ (name.find("attn_qkv.weight") != std::string::npos) ||
+ (name.find("attn.qkv.weight") != std::string::npos) ||
+ (name.find("attention.qkv.weight") != std::string::npos)
+ ) {
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
+ new_type = GGML_TYPE_Q4_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {
+ new_type = GGML_TYPE_Q5_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
+ new_type = GGML_TYPE_Q6_K;
+ }
+ } else if ( // Rules for ffn
+ (name.find("ffn_down") != std::string::npos) ||
+ ((name.find("experts.") != std::string::npos) && (name.find(".w2.weight") != std::string::npos)) ||
+ (name.find(".ffn.2.weight") != std::string::npos) || // is this even the right way around?
+ (name.find(".ff.net.2.weight") != std::string::npos) ||
+ (name.find(".mlp.layer2.weight") != std::string::npos) ||
+ (name.find(".adaln_modulation_mlp.2.weight") != std::string::npos) ||
+ (name.find(".feed_forward.w2.weight") != std::string::npos)
+ ) {
+ // TODO: add back `layer_info` with some model specific logic + logic further down
+ if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {
+ new_type = GGML_TYPE_Q4_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {
+ new_type = GGML_TYPE_Q5_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S) {
+ new_type = GGML_TYPE_Q5_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {
+ new_type = GGML_TYPE_Q6_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) {
+ new_type = GGML_TYPE_Q6_K;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_0) {
+ new_type = GGML_TYPE_Q4_1;
+ }
+ else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_0) {
+ new_type = GGML_TYPE_Q5_1;
+ }
+ ++qs.i_ffn_down;
+ }
+
+ // Sanity check for row shape
+ bool convert_incompatible_tensor = false;
+ if (new_type == GGML_TYPE_Q2_K || new_type == GGML_TYPE_Q3_K || new_type == GGML_TYPE_Q4_K ||
+ new_type == GGML_TYPE_Q5_K || new_type == GGML_TYPE_Q6_K) {
+ int nx = tensor->ne[0];
+ int ny = tensor->ne[1];
+ if (nx % QK_K != 0) {
+ LLAMA_LOG_WARN("\n\n%s : tensor cols %d x %d are not divisible by %d, required for %s", __func__, nx, ny, QK_K, ggml_type_name(new_type));
+ convert_incompatible_tensor = true;
+ } else {
+ ++qs.n_k_quantized;
+ }
+ }
+ if (convert_incompatible_tensor) {
+ // TODO: Possibly reenable this in the future
+ // switch (new_type) {
+ // case GGML_TYPE_Q2_K:
+ // case GGML_TYPE_Q3_K:
+ // case GGML_TYPE_Q4_K: new_type = GGML_TYPE_Q5_0; break;
+ // case GGML_TYPE_Q5_K: new_type = GGML_TYPE_Q5_1; break;
+ // case GGML_TYPE_Q6_K: new_type = GGML_TYPE_Q8_0; break;
+ // default: throw std::runtime_error("\nUnsupported tensor size encountered\n");
+ // }
+ new_type = GGML_TYPE_F16;
+ LLAMA_LOG_WARN(" - using fallback quantization %s\n", ggml_type_name(new_type));
+ ++qs.n_fallback;
+ }
+ return new_type;
+}
+
static ggml_type llama_tensor_get_type(quantize_state_internal & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype) {
const std::string name = ggml_get_name(tensor);
@@ -18513,7 +18693,9 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
if (llama_model_has_encoder(&model)) {
n_attn_layer *= 3;
}
- GGML_ASSERT((qs.n_attention_wv == n_attn_layer) && "n_attention_wv is unexpected");
+ if (model.arch != LLM_ARCH_HYVID) { // TODO: Check why this fails
+ GGML_ASSERT((qs.n_attention_wv == n_attn_layer) && "n_attention_wv is unexpected");
+ }
}
size_t total_size_org = 0;
@@ -18547,6 +18729,51 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
ctx_outs[i_split] = gguf_init_empty();
}
gguf_add_tensor(ctx_outs[i_split], tensor);
+ // SD3 pos_embed needs special fix as first dim is 1, which gets truncated here
+ if (model.arch == LLM_ARCH_SD3) {
+ const std::string name = ggml_get_name(tensor);
+ if (name == "pos_embed" && tensor->ne[2] == 1) {
+ const int n_dim = 3;
+ gguf_set_tensor_ndim(ctx_outs[i_split], "pos_embed", n_dim);
+ LLAMA_LOG_INFO("\n%s: Correcting pos_embed shape for SD3: [key:%s]\n", __func__, tensor->name);
+ }
+ }
+ // same goes for auraflow
+ if (model.arch == LLM_ARCH_AURA) {
+ const std::string name = ggml_get_name(tensor);
+ if (name == "positional_encoding" && tensor->ne[2] == 1) {
+ const int n_dim = 3;
+ gguf_set_tensor_ndim(ctx_outs[i_split], "positional_encoding", n_dim);
+ LLAMA_LOG_INFO("\n%s: Correcting positional_encoding shape for AuraFlow: [key:%s]\n", __func__, tensor->name);
+ }
+ if (name == "register_tokens" && tensor->ne[2] == 1) {
+ const int n_dim = 3;
+ gguf_set_tensor_ndim(ctx_outs[i_split], "register_tokens", n_dim);
+ LLAMA_LOG_INFO("\n%s: Correcting register_tokens shape for AuraFlow: [key:%s]\n", __func__, tensor->name);
+ }
+ }
+ // conv3d fails due to max dims - unsure what to do here as we never even reach this check
+ if (model.arch == LLM_ARCH_HYVID) {
+ const std::string name = ggml_get_name(tensor);
+ if (name == "img_in.proj.weight" && tensor->ne[5] != 1 ) {
+ throw std::runtime_error("img_in.proj.weight size failed for HyVid");
+ }
+ }
+ // All the modulation layers also have dim1, and I think conv3d fails here too but we segfaul way before that...
+ if (model.arch == LLM_ARCH_WAN) {
+ const std::string name = ggml_get_name(tensor);
+ if (name.find(".modulation") != std::string::npos && tensor->ne[2] == 1) {
+ const int n_dim = 3;
+ gguf_set_tensor_ndim(ctx_outs[i_split], tensor->name, n_dim);
+ LLAMA_LOG_INFO("\n%s: Correcting shape for Wan: [key:%s]\n", __func__, tensor->name);
+ }
+ // FLF2V model only
+ if (name == "img_emb.emb_pos") {
+ const int n_dim = 3;
+ gguf_set_tensor_ndim(ctx_outs[i_split], tensor->name, n_dim);
+ LLAMA_LOG_INFO("\n%s: Correcting shape for Wan FLF2V: [key:%s]\n", __func__, tensor->name);
+ }
+ }
}
// Set split info if needed
@@ -18647,6 +18874,110 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
// do not quantize relative position bias (T5)
quantize &= name.find("attn_rel_b.weight") == std::string::npos;
+ // rules for image models
+ bool image_model = false;
+ if (model.arch == LLM_ARCH_FLUX) {
+ image_model = true;
+ quantize &= name.find("txt_in.") == std::string::npos;
+ quantize &= name.find("img_in.") == std::string::npos;
+ quantize &= name.find("time_in.") == std::string::npos;
+ quantize &= name.find("vector_in.") == std::string::npos;
+ quantize &= name.find("guidance_in.") == std::string::npos;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ }
+ if (model.arch == LLM_ARCH_SD1 || model.arch == LLM_ARCH_SDXL) {
+ image_model = true;
+ quantize &= name.find("class_embedding.") == std::string::npos;
+ quantize &= name.find("time_embedding.") == std::string::npos;
+ quantize &= name.find("add_embedding.") == std::string::npos;
+ quantize &= name.find("time_embed.") == std::string::npos;
+ quantize &= name.find("label_emb.") == std::string::npos;
+ quantize &= name.find("conv_in.") == std::string::npos;
+ quantize &= name.find("conv_out.") == std::string::npos;
+ quantize &= name != "input_blocks.0.0.weight";
+ quantize &= name != "out.2.weight";
+ }
+ if (model.arch == LLM_ARCH_SD3) {
+ image_model = true;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ quantize &= name.find("time_text_embed.") == std::string::npos;
+ quantize &= name.find("context_embedder.") == std::string::npos;
+ quantize &= name.find("t_embedder.") == std::string::npos;
+ quantize &= name.find("y_embedder.") == std::string::npos;
+ quantize &= name.find("x_embedder.") == std::string::npos;
+ quantize &= name != "proj_out.weight";
+ quantize &= name != "pos_embed";
+ }
+ if (model.arch == LLM_ARCH_AURA) {
+ image_model = true;
+ quantize &= name.find("t_embedder.") == std::string::npos;
+ quantize &= name.find("init_x_linear.") == std::string::npos;
+ quantize &= name != "modF.1.weight";
+ quantize &= name != "cond_seq_linear.weight";
+ quantize &= name != "final_linear.weight";
+ quantize &= name != "final_linear.weight";
+ quantize &= name != "positional_encoding";
+ quantize &= name != "register_tokens";
+ }
+ if (model.arch == LLM_ARCH_LTXV) {
+ image_model = true;
+ quantize &= name.find("adaln_single.") == std::string::npos;
+ quantize &= name.find("caption_projection.") == std::string::npos;
+ quantize &= name.find("patchify_proj.") == std::string::npos;
+ quantize &= name.find("proj_out.") == std::string::npos;
+ quantize &= name.find("scale_shift_table") == std::string::npos; // last block too
+ }
+ if (model.arch == LLM_ARCH_HYVID) {
+ image_model = true;
+ quantize &= name.find("txt_in.") == std::string::npos;
+ quantize &= name.find("img_in.") == std::string::npos;
+ quantize &= name.find("time_in.") == std::string::npos;
+ quantize &= name.find("vector_in.") == std::string::npos;
+ quantize &= name.find("guidance_in.") == std::string::npos;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ }
+ if (model.arch == LLM_ARCH_WAN) {
+ image_model = true;
+ quantize &= name.find("modulation.") == std::string::npos;
+ quantize &= name.find("patch_embedding.") == std::string::npos;
+ quantize &= name.find("text_embedding.") == std::string::npos;
+ quantize &= name.find("time_projection.") == std::string::npos;
+ quantize &= name.find("time_embedding.") == std::string::npos;
+ quantize &= name.find("img_emb.") == std::string::npos;
+ quantize &= name.find("head.") == std::string::npos;
+ }
+ if (model.arch == LLM_ARCH_HIDREAM) {
+ image_model = true;
+ quantize &= name.find("p_embedder.") == std::string::npos;
+ quantize &= name.find("t_embedder.") == std::string::npos;
+ quantize &= name.find("x_embedder.") == std::string::npos;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ quantize &= name.find(".ff_i.gate.weight") == std::string::npos;
+ quantize &= name.find("caption_projection.") == std::string::npos;
+ }
+ if (model.arch == LLM_ARCH_COSMOS) {
+ image_model = true;
+ quantize &= name.find("p_embedder.") == std::string::npos;
+ quantize &= name.find("t_embedder.") == std::string::npos;
+ quantize &= name.find("t_embedding_norm.") == std::string::npos;
+ quantize &= name.find("x_embedder.") == std::string::npos;
+ quantize &= name.find("pos_embedder.") == std::string::npos;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ }
+ if (model.arch == LLM_ARCH_LUMINA2) {
+ image_model = true;
+ quantize &= name.find("t_embedder.") == std::string::npos;
+ quantize &= name.find("x_embedder.") == std::string::npos;
+ quantize &= name.find("final_layer.") == std::string::npos;
+ quantize &= name.find("cap_embedder.") == std::string::npos;
+ quantize &= name.find("context_refiner.") == std::string::npos;
+ quantize &= name.find("noise_refiner.") == std::string::npos;
+ }
+ // ignore 3D/4D tensors for image models as the code was never meant to handle these
+ if (image_model) {
+ quantize &= ggml_n_dims(tensor) == 2;
+ }
+
enum ggml_type new_type;
void * new_data;
size_t new_size;
@@ -18655,6 +18986,9 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
new_type = default_type;
// get more optimal quantization type based on the tensor shape, layer, etc.
+ if (image_model) {
+ new_type = img_tensor_get_type(qs, new_type, tensor, ftype);
+ } else {
if (!params->pure && ggml_is_quantized(default_type)) {
new_type = llama_tensor_get_type(qs, new_type, tensor, ftype);
}
@@ -18664,6 +18998,7 @@ static void llama_model_quantize_internal(const std::string & fname_inp, const s
if (params->output_tensor_type < GGML_TYPE_COUNT && strcmp(tensor->name, "output.weight") == 0) {
new_type = params->output_tensor_type;
}
+ }
// If we've decided to quantize to the same type the tensor is already
// in then there's nothing to do.
+21
View File
@@ -0,0 +1,21 @@
#!/usr/bin/python3
import os
import sys
import gguf
def read_tensors(path):
reader = gguf.GGUFReader(path)
for tensor in reader.tensors:
if tensor.tensor_type == gguf.GGMLQuantizationType.F32:
continue
print(f"{str(tensor.tensor_type):32}: {tensor.name}")
try:
path = sys.argv[1]
assert os.path.isfile(path), "Invalid path"
print(f"input: {path}")
except Exception as e:
input(f"failed: {e}")
else:
read_tensors(path)
input()