"""gguf-bits-per-weight.py - work out what a GGUF quantisation format actually costs per weight, straight from llama.cpp's own source, and join that against llama.cpp's published size and quality measurements. The name of a format is not its cost. Q4_K does not store four bits per weight, because every block also carries scales, minimums and a super-block delta. This script reads ggml/src/ggml-common.h, evaluates the static_assert that fixes each block struct's byte size, divides by the block's element count, and prints the exact figure. Nothing is hard-coded: change the header upstream and the numbers here change with it. It then parses two measured tables out of the llama.cpp repository: tools/quantize/README.md whole-model bits/weight and file size for Llama-3.1-8B, per quantisation type tools/perplexity/README.md perplexity, delta-perplexity, KLD and token probability error for Llama-3 8B, measured on one RTX 4090 at a stated revision and reports three things the formats documentation does not: the gap between a format's per-block cost and its whole-model cost, what an importance matrix is worth at identical file size, and which formats are dominated, meaning some other format is both smaller and more accurate. Sources are fetched at run time from raw.githubusercontent.com, so the output is always against current upstream rather than a snapshot. Pass --offline with a directory to read previously saved copies instead. Standard library only (urllib, re, json, argparse, sys). Python 3.9 or newer. python gguf-bits-per-weight.py python gguf-bits-per-weight.py --offline ./saved python gguf-bits-per-weight.py --csv gguf-bits-per-weight.csv """ from __future__ import annotations import argparse import csv import os import re import sys import urllib.request RAW = "https://raw.githubusercontent.com/ggml-org/llama.cpp/master/" FILES = { "ggml-common.h": "ggml/src/ggml-common.h", "quantize.md": "tools/quantize/README.md", "perplexity.md": "tools/perplexity/README.md", } # C type sizes, as the header assumes them on every platform llama.cpp targets. CTYPES = {"ggml_half": 2, "ggml_half2": 4, "uint32_t": 4, "uint16_t": 2, "int16_t": 2, "uint8_t": 1, "int8_t": 1, "float": 4} # Element count per block, where the header does not name it as QK. # Every K-quant and I-quant super-block holds QK_K weights. BLOCK_ELEMS = {"iq4_nl": "QK4_NL", "mxfp4": "QK_MXFP4", "nvfp4": "QK_NVFP4"} # The block types llama-quantize will actually write for model weights, in the # order the tool lists them. q8_1 and q8_K are activation-side formats and q1_0, # q2_0 are not offered as targets, so they are reported separately. WEIGHT_TYPES = ["iq1_s", "iq1_m", "iq2_xxs", "iq2_xs", "iq2_s", "q2_K", "iq3_xxs", "iq3_s", "q3_K", "iq4_xs", "iq4_nl", "q4_0", "q4_1", "q4_K", "q5_0", "q5_1", "q5_K", "q6_K", "q8_0", "tq1_0", "tq2_0", "mxfp4", "nvfp4"] def load(offline): out = {} for name, path in FILES.items(): if offline: with open(os.path.join(offline, name), encoding="utf-8") as f: out[name] = f.read() continue request = urllib.request.Request(RAW + path, headers={"User-Agent": "Mozilla/5.0"}) with urllib.request.urlopen(request, timeout=45) as response: out[name] = response.read().decode("utf-8") return out def constants(header): """Every QK*, K_SCALE_SIZE and IQ3S_N_SCALE define, evaluated in order.""" env = dict(CTYPES) env["sizeof"] = lambda x: x for match in re.finditer(r"#define\s+(QK\w*|K_SCALE_SIZE|IQ3S_N_SCALE)\s+(.+)", header): expr = re.sub(r"//.*$", "", match.group(2)).strip() try: env[match.group(1)] = eval(expr, {"__builtins__": {}}, dict(env)) except Exception: pass # a define that is not arithmetic; skip it return env def block_sizes(header, env): """Byte size and bits per weight for every block struct the header pins.""" rows = {} pattern = r'static_assert\(sizeof\(block_(\w+)\)\s*==\s*(.+?),\s*"' for match in re.finditer(pattern, header, re.S): name = match.group(1) expr = re.sub(r"\s+", " ", match.group(2)).strip() try: size = eval(expr, {"__builtins__": {}}, dict(env)) except Exception: continue key = BLOCK_ELEMS.get(name) if key is None: for candidate in ("QK" + name[1:].upper(), "QK_" + name.upper()): if candidate in env: key = candidate break elems = env.get(key) if key else None if elems is None and (name.endswith("_K") or name[:2] in ("iq", "tq")): elems = env["QK_K"] if elems is None: continue rows[name] = {"bytes": int(size), "elems": int(elems), "bpw": 8.0 * size / elems, "expr": expr} return rows def quantize_table(markdown): """Whole-model bits/weight and size from tools/quantize/README.md.""" rows = {} for block in re.findall(r"(?ms)^\| Measure\s*\|.*?(?=\n\n|\Z)", markdown): lines = [l for l in block.splitlines() if l.startswith("|")] names = [c.strip() for c in lines[0].strip("|").split("|")][1:] for line in lines[2:]: cells = [c.strip() for c in line.strip("|").split("|")] if not cells: continue label, values = cells[0], cells[1:] for name, value in zip(names, values): try: number = float(value.split()[0]) except (ValueError, IndexError): continue entry = rows.setdefault(name, {}) if label.startswith("bits/weight"): entry["bpw"] = number elif label.startswith("size"): entry["gib"] = number elif label.startswith("prompt processing"): entry["pp"] = number elif label.startswith("text generation"): entry["tg"] = number return rows def quality_table(markdown): """PPL, delta-PPL and KLD per type from tools/perplexity/README.md.""" rows = [] header = "| Quantization | imatrix | Model size [GiB] | PPL" start = markdown.find(header) if start < 0: return rows for line in markdown[start:].splitlines()[2:]: if not line.startswith("|"): break cells = [c.strip() for c in line.strip("|").split("|")] if len(cells) < 6: continue try: rows.append({"type": cells[0], "imatrix": cells[1], "gib": float(cells[2]), "ppl": float(cells[3].split("±")[0]), "dppl": float(cells[4].split("±")[0]), "kld": float(cells[5].split("±")[0])}) except ValueError: continue return rows def pareto(rows): """Rows for which nothing else is both smaller and lower perplexity.""" front = [] for row in rows: beaten = any(other["gib"] <= row["gib"] and other["ppl"] < row["ppl"] for other in rows if other is not row) if not beaten: front.append(row) return sorted(front, key=lambda r: r["gib"]) def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--offline", metavar="DIR", help="read saved copies from DIR instead of fetching") parser.add_argument("--csv", metavar="FILE", help="also write the joined table") args = parser.parse_args() src = load(args.offline) env = constants(src["ggml-common.h"]) blocks = block_sizes(src["ggml-common.h"], env) measured = quantize_table(src["quantize.md"]) quality = quality_table(src["perplexity.md"]) print(f"QK_K = {int(env['QK_K'])} block structs parsed: {len(blocks)}") print() print("EXACT COST PER WEIGHT, derived from ggml-common.h") print(f" {'format':<9} {'bytes':>6} {'weights':>8} {'bits/wt':>8} block size expression") for name in WEIGHT_TYPES: row = blocks.get(name) if not row: continue print(f" {name:<9} {row['bytes']:>6} {row['elems']:>8} " f"{row['bpw']:>8.4f} {row['expr']}") other = sorted(set(blocks) - set(WEIGHT_TYPES)) print(f" not weight targets: " + ", ".join(f"{n} {blocks[n]['bpw']:.4f}" for n in other)) print() print("PER-BLOCK COST vs WHOLE-MODEL COST (Llama-3.1-8B, quantize README)") print(f" {'measured':<9} {'block':<8} {'per-block':>10} {'whole':>8} " f"{'gap':>7} {'gap %':>7} {'GiB':>6}") joined = [] for name in sorted(measured, key=lambda n: measured[n].get("bpw", 0)): entry = measured[name] if "bpw" not in entry: continue # Only an exact name match is reported. A measured name such as Q4_K_M # or IQ3_XS is a mixture the quantize tool assembles from more than one # block type, so it has no single per-block cost to compare against. block = next((b for b in blocks if b.lower() == name.lower()), None) if block is None: print(f" {name:<9} {'mixture':<8} {'-':>10} {entry['bpw']:>8.4f}" f" {'':>7} {'':>7} {entry.get('gib', 0):>6.2f}") continue exact = blocks[block]["bpw"] gap = entry["bpw"] - exact joined.append((name, block, exact, entry["bpw"], gap, entry.get("gib"))) print(f" {name:<9} {block:<8} {exact:>10.4f} {entry['bpw']:>8.4f} " f"{gap:>+7.4f} {100 * gap / exact:>+6.1f}% {entry.get('gib', 0):>6.2f}") print() f16 = measured.get("F16") if f16 and "gib" in f16: params = f16["gib"] * (2 ** 30) * 8 / f16["bpw"] print(f"IMPLIED WEIGHT COUNT from the F16 row: " f"{f16['gib']} GiB at {f16['bpw']} bits/weight = " f"{params / 1e9:.4f} billion weights") print(f" {'format':<9} {'naive GiB':>10} {'actual GiB':>11} " f"{'extra GiB':>10} {'extra %':>8}") for name, block, exact, whole, gap, gib in joined: if gib is None: continue naive = params * exact / 8 / (2 ** 30) print(f" {name:<9} {naive:>10.2f} {gib:>11.2f} " f"{gib - naive:>+10.2f} {100 * (gib - naive) / naive:>+7.1f}%") print() print("QUALITY, measured (Llama-3 8B, perplexity README)") print(f" {'type':<9} {'imatrix':<8} {'GiB':>6} {'PPL':>10} " f"{'dPPL':>9} {'KLD':>9}") for row in quality[:16]: print(f" {row['type']:<9} {row['imatrix']:<8} {row['gib']:>6.2f} " f"{row['ppl']:>10.4f} {row['dppl']:>9.4f} {row['kld']:>9.5f}") print(f" ... {len(quality)} rows in all") print() # Only three types are measured at more than one calibration size, and the # ordering by token count is not monotonic, so comparing each type's BEST # imatrix row would be cherry-picking. Every row below uses the same size. basis = "WT 10m" print(f"WHAT AN IMPORTANCE MATRIX IS WORTH, at identical file size " f"(calibration held at {basis})") seen = {} for row in quality: seen.setdefault(row["type"], {})[row["imatrix"]] = row print(f" {'type':<9} {'no imatrix':>11} {'with imatrix':>13} " f"{'dPPL saved':>11} {'% of gap':>9}") for name, variants in seen.items(): plain = variants.get("None") withmat = variants.get(basis) if not (plain and withmat): continue saved = plain["ppl"] - withmat["ppl"] share = 100 * saved / plain["dppl"] if plain["dppl"] else 0.0 print(f" {name:<9} {plain['ppl']:>11.4f} {withmat['ppl']:>13.4f} " f"{saved:>11.4f} {share:>8.1f}%") print() print("CALIBRATION SIZE IS NOT MONOTONIC") for name, variants in seen.items(): sized = {k: v for k, v in variants.items() if k != "None"} if len(sized) < 2: continue order = sorted(sized.items(), key=lambda kv: kv[1]["ppl"]) spread = order[-1][1]["ppl"] - order[0][1]["ppl"] print(f" {name:<9} spread {spread:.6f} over {len(order)} sizes: " + ", ".join(f"{k} {v['ppl']:.4f}" for k, v in order)) print() print("DOMINATED FORMATS: something else is both smaller and more accurate") best = pareto(quality) front = {(r["type"], r["imatrix"]) for r in best} for row in sorted(quality, key=lambda r: r["gib"]): if (row["type"], row["imatrix"]) in front: continue winner = min((o for o in quality if o["gib"] <= row["gib"] and o["ppl"] < row["ppl"]), key=lambda o: o["ppl"]) print(f" {row['type']:<9} ({row['imatrix']:<7}) {row['gib']:>5.2f} GiB " f"PPL {row['ppl']:>8.4f} beaten by {winner['type']} " f"({winner['imatrix']}) at {winner['gib']:.2f} GiB " f"PPL {winner['ppl']:.4f}") print() print(f"PARETO FRONTIER: {len(best)} of {len(quality)} measured rows") for row in best: print(f" {row['gib']:>5.2f} GiB PPL {row['ppl']:>8.4f} " f"{row['type']} ({row['imatrix']})") if args.csv: with open(args.csv, "w", encoding="utf-8", newline="") as handle: writer = csv.writer(handle) writer.writerow(["format", "block_struct", "block_bytes", "block_weights", "bits_per_weight_exact", "bits_per_weight_whole_model", "gap_bits", "gap_percent", "file_gib"]) for name, block, exact, whole, gap, gib in joined: writer.writerow([name, block, blocks[block]["bytes"], blocks[block]["elems"], f"{exact:.4f}", f"{whole:.4f}", f"{gap:+.4f}", f"{100 * gap / exact:+.2f}", gib]) print(f"\nwrote {args.csv}") if __name__ == "__main__": main()