"""Re-derive the headline numbers of arXiv 2609.17346v1 from its own published tables. What it does ------------ 1. Writes datasets/where-should-a-document-live.csv: every number this article quotes, transcribed from the two papers' tables and from result sentences that quote figure values, one row per measurement. 2. Re-derives each table's printed "Avg." column from its five per-benchmark cells and reports the difference, so a reader can see whether the averages are reproducible. 3. Recomputes the method gaps the abstract states (representation-based vs parametric), the Qwen3-8B / Gemma-3-12B rank flip, and the multi-document deltas. 4. Renders content/images/where-should-a-document-live/methods.png from the same rows. Sources transcribed (read 2026-09-18): arXiv 2609.17346v1, "Where Should a Document Live: Context, Representations, or Parameters?", Tables 2-6 and the result sentences of sections 5.1, 5.2 and 6. arXiv 2608.25655v1, "Reconstructing the Right Episode", Table 2, Table 3 and the result sentences of sections 6.1, 6.4 and 6.6. No number here is computed by me except the differences this script prints; every input value appears in one of those tables or sentences. Inputs: none (the measurements are transcribed below). Outputs: datasets/where-should-a-document-live.csv, the PNG above, and a printed report. Run: python code/where-should-a-document-live.py Requires: Python 3.11+ (stdlib) and matplotlib (tested on 3.10.9) for the chart only; pass --no-chart to run on the standard library alone. """ from __future__ import annotations import argparse import csv import os HERE = os.path.dirname(os.path.abspath(__file__)) ROOT = os.path.dirname(HERE) SLUG = "where-should-a-document-live" CSV_PATH = os.path.join(ROOT, "datasets", SLUG + ".csv") IMG_PATH = os.path.join(ROOT, "content", "images", SLUG, "methods.png") P1, A1 = SLUG, "2609.17346v1" P2, A2 = "scale-qa", "2608.25655v1" # The five knowledge-intensive benchmarks and their metrics (paper Table 1). BM = [ ("LongHealth", "accuracy"), ("QuALITY", "accuracy"), ("QASPER", "token_f1"), ("FinQA", "exact_match"), ("TechQA", "judge_score"), ] # Table 2: Qwen3-8B, single oracle document, fixed adapter size. # value, standard deviation over 3 runs; last element is the printed Avg. TABLE2 = { "no context": [(37.5, 1.1), (43.6, 0.4), (19.2, 0.6), (2.9, 0.0), (21.1, 1.1), 24.9], "ICL": [(87.4, 0.8), (82.5, 0.3), (56.7, 0.4), (66.8, 2.7), (74.7, 0.9), 73.6], "Cartridge": [(81.1, 1.1), (78.6, 0.9), (54.9, 0.3), (62.7, 0.2), (75.8, 0.7), 70.6], "Compaction": [(87.7, 0.9), (82.1, 0.3), (54.8, 0.1), (66.4, 0.2), (76.0, 2.4), 73.4], "LoRA": [(75.3, 0.9), (73.7, 1.2), (50.3, 0.6), (49.0, 0.9), (74.3, 2.6), 64.5], "MLP adapters": [(74.8, 0.9), (72.3, 1.2), (47.5, 0.9), (44.1, 0.3), (72.5, 1.8), 62.2], "full fine-tuning": [(69.0, 2.4), (72.8, 0.3), (39.8, 0.3), (41.6, 2.8), (76.0, 3.4), 59.8], } # Table 4: Gemma-3-12B, same protocol. TABLE4 = { "no context": [(40.6, 1.4), (28.0, 0.7), (19.3, 0.3), (5.2, 0.0), (14.8, 1.2), 21.6], "ICL (oracle)": [(78.5, 0.4), (77.4, 0.2), (51.3, 0.3), (61.9, 0.0), (75.4, 1.9), 68.9], "Cartridge": [(49.8, 2.1), (43.0, 1.4), (23.4, 1.2), (9.5, 1.7), (60.6, 0.3), 37.3], "Compaction": [(81.4, 0.3), (74.4, 0.5), (48.0, 0.2), (61.2, 0.9), (61.0, 2.6), 65.2], "LoRA": [(56.5, 1.2), (62.4, 0.5), (34.8, 1.0), (34.7, 0.2), (65.6, 1.0), 50.8], } # Table 6: training objective for per-document LoRA, Qwen3-8B, single document. TABLE6 = { "LoRA (distillation)": [(75.2, 0.9), (73.7, 1.2), (50.3, 0.6), (49.0, 0.9), (74.8, 3.3), 64.6], "LoRA (next-token prediction)": [(72.6, 1.6), (70.7, 0.7), (46.9, 1.0), (41.3, 0.6), (73.2, 1.2), 60.9], } # Table 3: merge operators for per-document LoRA adapters, top-5 retrieved documents. BM3 = [("LongHealth", "accuracy"), ("QuALITY", "accuracy"), ("FinQA", "exact_match"), ("TechQA", "judge_score")] TABLE3 = { "mean": [(51.2, 1.9), (56.0, 0.7), (5.1, 0.1), (21.7, 0.4)], "cat": [(51.0, 2.2), (56.0, 0.7), (5.0, 0.1), (22.1, 1.0)], "TIES": [(51.2, 2.3), (56.0, 0.6), (5.1, 0.1), (21.9, 0.9)], "DARE": [(50.9, 2.0), (56.0, 0.7), (5.0, 0.1), (22.5, 0.8)], } # Table 5: control benchmarks on Gemma-3-12B, base model vs 2x Cartridge. TABLE5 = [("GSM8K", 88.7, 36.1, 23.9), ("HumanEval", 83.5, 54.4, 14.6), ("IFEval", 78.4, 39.2, 15.7), ("MMLU", 72.6, 56.9, 8.2)] # Compression-sweep values quoted in the text of section 5.1 (plotted in Figure 1). SWEEP = [ ("Compaction", "FinQA", "exact_match", "2x", 66.4), ("Compaction", "FinQA", "exact_match", "20x", 19.3), ("Compaction", "LongHealth", "accuracy", "2x", 87.7), ("Compaction", "LongHealth", "accuracy", "100x", 46.5), ("Cartridge", "LongHealth", "accuracy", "2x", 81.1), ("Cartridge", "LongHealth", "accuracy", "100x", 77.3), ("Cartridge", "QuALITY", "accuracy", "2x", 78.6), ("Cartridge", "QuALITY", "accuracy", "100x", 76.4), ("Cartridge", "TechQA", "judge_score", "2x", 75.8), ("Cartridge", "TechQA", "judge_score", "100x", 76.9), ] # Multi-document values quoted in the text of section 5.2 (plotted in Figure 2). MULTI = [ ("Cartridge", "LongHealth", "accuracy", "k=1", 70.8), ("Cartridge", "LongHealth", "accuracy", "k=10", 83.2), ("Cartridge", "TechQA", "judge_score", "k=1", 70.7), ("Cartridge", "TechQA", "judge_score", "k=10", 70.9), ("Cartridge", "QuALITY", "accuracy", "k=1", 72.5), ("Cartridge", "QuALITY", "accuracy", "k=10", 74.6), ("Compaction", "LongHealth", "accuracy", "k=1", 72.9), ("Compaction", "LongHealth", "accuracy", "k=10", 56.3), ("Compaction", "TechQA", "judge_score", "k=1", 65.4), ("Compaction", "TechQA", "judge_score", "k=10", 26.4), ("LoRA (merged)", "TechQA", "judge_score", "k=1", 57.4), ("LoRA (merged)", "TechQA", "judge_score", "k=3", 23.5), ("LoRA (merged)", "QuALITY", "accuracy", "k=1", 70.8), ("LoRA (merged)", "QuALITY", "accuracy", "k=10", 44.2), ("merged adapters", "FinQA", "exact_match", "k=1", 34.5), ("merged adapters", "FinQA", "exact_match", "k=3", 4.9), ("LoRA (joint)", "FinQA", "exact_match", "k=3", 13.0), ("LoRA (joint)", "TechQA", "judge_score", "k=3", 40.0), ("Cartridge", "LongHealth", "accuracy", "k=10, 2x", 83.2), ("Cartridge", "LongHealth", "accuracy", "k=10, 100x", 76.5), ("Cartridge", "FinQA", "exact_match", "k=10, 2x", 50.7), ("Cartridge", "FinQA", "exact_match", "k=10, 20x", 22.1), ("Compaction", "LongHealth", "accuracy", "k=10, 2x", 56.3), ("Compaction", "LongHealth", "accuracy", "k=10, 50x", 37.2), ] # SCALE-QA Table 2: accuracy, closed-loop evidence hit, rationale similarity, # context tokens, latency in seconds, per backend. SCALE_T2 = [ ("Gemma2:9b", "TSIM", 69.6, 70.7, 0.472, 1060.8, 4.20), ("Gemma2:9b", "Standard RAG", 24.4, 7.7, 0.296, 929.5, 3.35), ("Gemma2:9b", "Hybrid-RRF Chunk RAG", 31.1, 11.2, 0.254, 1193.6, 4.00), ("Gemma2:9b", "RAPTOR", 40.6, 35.3, 0.403, 790.5, 20.67), ("Gemma2:9b", "MEMGPT", 60.1, 62.6, 0.413, 2301.9, 3.88), ("Gemma2:9b", "HIPPORAG", 25.2, 12.5, 0.380, 748.9, 6.51), ("Gemini 2.5 Flash", "TSIM", 80.2, 67.9, 0.445, 1275.3, 1.46), ("Gemini 2.5 Flash", "Standard RAG", 29.8, 7.6, 0.166, 912.6, 0.98), ("Gemini 2.5 Flash", "Hybrid-RRF Chunk RAG", 32.8, 11.2, 0.183, 1193.3, 1.20), ("Gemini 2.5 Flash", "RAPTOR", 52.6, 34.8, 0.305, 790.0, 9.85), ("Gemini 2.5 Flash", "MEMGPT", 74.6, 62.3, 0.398, 2349.6, 1.19), ("Gemini 2.5 Flash", "HIPPORAG", 34.4, 12.1, 0.245, 752.7, 3.10), ("GPT-4o-mini", "TSIM", 73.8, 74.2, 0.485, 1043.6, 2.66), ("GPT-4o-mini", "Standard RAG", 27.1, 7.6, 0.238, 910.1, 1.72), ("GPT-4o-mini", "Hybrid-RRF Chunk RAG", 31.1, 11.3, 0.203, 1193.5, 1.59), ("GPT-4o-mini", "RAPTOR", 43.0, 35.1, 0.419, 789.2, 11.59), ("GPT-4o-mini", "MEMGPT", 50.2, 58.6, 0.298, 2238.4, 1.77), ("GPT-4o-mini", "HIPPORAG", 31.0, 11.9, 0.202, 753.7, 2.85), ("GPT-4o-mini", "Tuned Hybrid-Rerank Chunk RAG", 56.2, 63.5, 0.487, 3004.2, 11.91), ] # SCALE-QA context-scaling and ablation values quoted in sections 6.1, 6.3, 6.4, 6.6. SCALE_SCALING = [ ("GPT-4o-mini", "Full Context", "16k", 62.5), ("GPT-4o-mini", "Full Context", "128k", 29.8), ("GPT-4o-mini", "TSIM", "128k", 73.8), ("DeepSeek R1", "Full Context", "128k", 81.2), ("DeepSeek R1", "TSIM", "128k", 93.8), ("Gemini 2.5 Flash", "Full Context", "1M", 87.2), ("Gemini 2.5 Flash", "TSIM", "1M", 96.5), ] SCALE_ABL = [("L0: Standard RAG top-5", 26.2, 5.6), ("L1: fixed-token direct", 43.4, 35.0), ("L2: semantic-drift direct", 55.5, 52.2), ("L3: full TSIM stack", 74.2, 74.3)] SCALE_LME = [("TSIM", 71.0), ("context-matched fixed-chunk control", 61.2), ("turn-level BGE retrieval", 56.6)] COLS = ["paper", "arxiv_id", "source", "model", "setting", "method", "benchmark", "metric", "variant", "value", "sd"] def build_rows() -> list[dict]: rows: list[dict] = [] def add(paper, aid, source, model, setting, method, bench, metric, variant, value, sd=""): rows.append(dict(paper=paper, arxiv_id=aid, source=source, model=model, setting=setting, method=method, benchmark=bench, metric=metric, variant=variant, value=value, sd=sd)) for table, model, variant, data in [ ("Table 2", "Qwen3-8B", "2x / rank 64 / d=512", TABLE2), ("Table 4", "Gemma-3-12B", "2x / rank 64", TABLE4), ("Table 6", "Qwen3-8B", "rank 64", TABLE6), ]: setting = "single document, training objective" if table == "Table 6" else \ "single oracle document, fixed adapter size" for method, vals in data.items(): for (bench, metric), (value, sd) in zip(BM, vals[:5]): add(P1, A1, table, model, setting, method, bench, metric, variant, value, sd) add(P1, A1, table, model, setting, method, "Avg", "printed_average", variant, vals[5]) for method, vals in TABLE3.items(): for (bench, metric), (value, sd) in zip(BM3, vals): add(P1, A1, "Table 3", "Qwen3-8B", "multi-document, merged LoRA adapters", "LoRA merge: " + method, bench, metric, "k=5, rank 64", value, sd) for bench, base, cart, sd in TABLE5: add(P1, A1, "Table 5", "Gemma-3-12B", "control benchmark (forgetting)", "base model", bench, "accuracy", "-", base) add(P1, A1, "Table 5", "Gemma-3-12B", "control benchmark (forgetting)", "Cartridge", bench, "accuracy", "2x", cart, sd) for method, bench, metric, variant, value in SWEEP: add(P1, A1, "Figure 1 (values quoted in section 5.1)", "Qwen3-8B", "single document, compression sweep", method, bench, metric, variant, value) for method, bench, metric, variant, value in MULTI: add(P1, A1, "Figure 2 (values quoted in section 5.2)", "Qwen3-8B", "multi-document retrieval", method, bench, metric, variant, value) for backend, method, acc, hit, rat, tok, lat in SCALE_T2: for metric, value in [("accuracy", acc), ("cl_hit", hit), ("rationale_similarity", rat), ("context_tokens", tok), ("latency_s", lat)]: add(P2, A2, "Table 2", backend, "SCALE-QA, 3,000 questions at 128k", method, "SCALE-QA", metric, "128k", value) for backend, method, variant, value in SCALE_SCALING: add(P2, A2, "Figures 4-5 (values quoted in sections 6.1 and 6.4)", backend, "context scaling", method, "SCALE-QA", "accuracy", variant, value) for stage, acc, hit in SCALE_ABL: add(P2, A2, "Table 3", "GPT-4o-mini", "ablation", stage, "SCALE-QA", "accuracy", "128k", acc) add(P2, A2, "Table 3", "GPT-4o-mini", "ablation", stage, "SCALE-QA", "cl_hit", "128k", hit) for method, value in SCALE_LME: add(P2, A2, "section 6.6", "not stated", "LongMemEval-S cleaned V1, 500 questions", method, "LongMemEval-S", "judged_accuracy", "no session boundaries", value) return rows def write_csv(rows: list[dict]) -> None: os.makedirs(os.path.dirname(CSV_PATH), exist_ok=True) with open(CSV_PATH, "w", newline="", encoding="utf-8") as fh: writer = csv.DictWriter(fh, fieldnames=COLS) writer.writeheader() writer.writerows(rows) print(f"wrote {CSV_PATH} ({len(rows)} rows)") def check_averages() -> None: """Recompute each printed Avg. column from its five per-benchmark cells.""" print("\n-- printed Avg. vs the mean of the five benchmark cells --") print(f"{'table':8} {'model':13} {'method':30} {'printed':>8} {'recomputed':>11} {'diff':>6}") for table, model, data in [("Table 2", "Qwen3-8B", TABLE2), ("Table 4", "Gemma-3-12B", TABLE4), ("Table 6", "Qwen3-8B", TABLE6)]: for method, vals in data.items(): cells = [v for v, _ in vals[:5]] recomputed = sum(cells) / len(cells) printed = vals[5] print(f"{table:8} {model:13} {method:30} {printed:8.1f} {recomputed:11.2f} " f"{recomputed - printed:+6.2f}") def check_gaps() -> None: """The gaps the abstract states, recomputed from the fixed-size tables.""" avg2 = {m: v[5] for m, v in TABLE2.items()} avg4 = {m: v[5] for m, v in TABLE4.items()} rep = ["Cartridge", "Compaction"] par = ["LoRA", "MLP adapters", "full fine-tuning"] print("\n-- oracle setting, Qwen3-8B (Table 2), averages --") for m in ["ICL"] + rep + par + ["no context"]: print(f" {m:20} {avg2[m]:5.1f}") best_par = max(par, key=lambda m: avg2[m]) par_mean = sum(avg2[m] for m in par) / len(par) print(f" Cartridge - best parametric ({best_par}): {avg2['Cartridge'] - avg2[best_par]:+.1f}") print(f" Cartridge - mean of parametric methods: {avg2['Cartridge'] - par_mean:+.1f}") print(f" Cartridge - full fine-tuning: {avg2['Cartridge'] - avg2['full fine-tuning']:+.1f}") print(f" ICL - Compaction: {avg2['ICL'] - avg2['Compaction']:+.1f}") print(f" ICL - Cartridge: {avg2['ICL'] - avg2['Cartridge']:+.1f}") print(f" ICL - no context: {avg2['ICL'] - avg2['no context']:+.1f}") print("\n-- the same methods on Gemma-3-12B (Table 4) --") for m in ["ICL (oracle)", "Cartridge", "Compaction", "LoRA", "no context"]: print(f" {m:20} {avg4[m]:5.1f}") print(f" Cartridge - LoRA on Qwen3-8B: {avg2['Cartridge'] - avg2['LoRA']:+.1f}") print(f" Cartridge - LoRA on Gemma-3-12B: {avg4['Cartridge'] - avg4['LoRA']:+.1f}") print(f" Cartridge, Qwen3-8B -> Gemma-3-12B: {avg4['Cartridge'] - avg2['Cartridge']:+.1f}") print("\n-- the same LoRA configuration in Table 2 and Table 6 --") for (bench, _), (v2, s2), (v6, s6) in zip(BM, TABLE2["LoRA"][:5], TABLE6["LoRA (distillation)"][:5]): flag = "" if (v2, s2) == (v6, s6) else " <- differs" print(f" {bench:12} Table 2 {v2:5.1f}+-{s2:<4} Table 6 {v6:5.1f}+-{s6:<4}{flag}") print(f" printed averages: Table 2 {TABLE2['LoRA'][5]}, Table 6 {TABLE6['LoRA (distillation)'][5]}") print("\n-- distillation vs next-token prediction (Table 6) --") d, n = TABLE6["LoRA (distillation)"][5], TABLE6["LoRA (next-token prediction)"][5] print(f" {d} - {n} = {d - n:+.1f} points on average") print("\n-- merge operators at k=5 (Table 3), spread across the four operators --") for i, (bench, _) in enumerate(BM3): vals = [TABLE3[op][i][0] for op in TABLE3] print(f" {bench:12} min {min(vals):5.1f} max {max(vals):5.1f} spread {max(vals) - min(vals):4.1f}") print("\n-- multi-document deltas from the values quoted in section 5.2 --") idx = {(m, b, v): val for m, b, _, v, val in MULTI} for method, bench, a, b in [("Cartridge", "LongHealth", "k=1", "k=10"), ("Cartridge", "QuALITY", "k=1", "k=10"), ("Cartridge", "TechQA", "k=1", "k=10"), ("Compaction", "LongHealth", "k=1", "k=10"), ("Compaction", "TechQA", "k=1", "k=10"), ("LoRA (merged)", "QuALITY", "k=1", "k=10"), ("LoRA (merged)", "TechQA", "k=1", "k=3"), ("merged adapters", "FinQA", "k=1", "k=3")]: lo, hi = idx[(method, bench, a)], idx[(method, bench, b)] print(f" {method:16} {bench:12} {a} {lo:5.1f} -> {b} {hi:5.1f} {hi - lo:+6.1f}") print("\n-- Cartridge forgetting on Gemma-3-12B (Table 5) --") for bench, base, cart, _ in TABLE5: print(f" {bench:10} base {base:5.1f} 2x Cartridge {cart:5.1f} {cart - base:+6.1f}") def check_scale_qa() -> None: print("\n-- SCALE-QA (arXiv 2608.25655v1), accuracy per prompt token --") print(f"{'backend':18} {'method':30} {'acc':>5} {'ctx tok':>8} {'acc/1k tok':>11}") for backend, method, acc, _hit, _rat, tok, _lat in SCALE_T2: print(f"{backend:18} {method:30} {acc:5.1f} {tok:8.1f} {acc / (tok / 1000.0):11.1f}") print("\n-- long context vs reconstructed memory, quoted values --") by = {(b, m, v): val for b, m, v, val in SCALE_SCALING} print(f" GPT-4o-mini Full Context 16k {by[('GPT-4o-mini', 'Full Context', '16k')]:.1f} " f"-> 128k {by[('GPT-4o-mini', 'Full Context', '128k')]:.1f} " f"({by[('GPT-4o-mini', 'Full Context', '128k')] - by[('GPT-4o-mini', 'Full Context', '16k')]:+.1f})") for backend, variant in [("DeepSeek R1", "128k"), ("Gemini 2.5 Flash", "1M")]: fc, ts = by[(backend, "Full Context", variant)], by[(backend, "TSIM", variant)] print(f" {backend:17} at {variant:4} full context {fc:5.1f} reconstructed memory {ts:5.1f} {ts - fc:+.1f}") def draw_chart() -> None: import matplotlib matplotlib.use("Agg") import matplotlib.pyplot as plt methods = ["ICL", "Compaction", "Cartridge", "LoRA", "MLP adapters", "full fine-tuning", "no context"] qwen = [TABLE2[m][5] for m in methods] gemma_key = {"ICL": "ICL (oracle)"} gemma = [TABLE4[gemma_key.get(m, m)][5] if gemma_key.get(m, m) in TABLE4 else None for m in methods] fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(11.5, 4.6)) y = range(len(methods)) h = 0.38 ax1.barh([v + h / 2 for v in y], qwen, height=h, color="#5b4bd6", label="Qwen3-8B (Table 2)") ax1.barh([v - h / 2 for v in y], [g if g is not None else 0 for g in gemma], height=h, color="#b9b2f0", label="Gemma-3-12B (Table 4)") for i, (q, g) in enumerate(zip(qwen, gemma)): ax1.text(q + 0.8, i + h / 2, f"{q:.1f}", va="center", fontsize=8) if g is not None: ax1.text(g + 0.8, i - h / 2, f"{g:.1f}", va="center", fontsize=8) else: ax1.text(1.0, i - h / 2, "not run", va="center", fontsize=8, color="#666666") ax1.set_yticks(list(y)) ax1.set_yticklabels(methods, fontsize=9) ax1.invert_yaxis() ax1.set_xlim(0, 100) ax1.set_xlabel("average score over the five benchmarks") ax1.set_title("Single oracle document: the ranking is not stable\nacross base models", fontsize=10) ax1.legend(fontsize=8, loc="lower right") ax1.grid(axis="x", alpha=0.25) series = [("Cartridge", "LongHealth", "k=1", "k=10", "#5b4bd6"), ("Cartridge", "TechQA", "k=1", "k=10", "#7d70e0"), ("Compaction", "LongHealth", "k=1", "k=10", "#d98b3a"), ("Compaction", "TechQA", "k=1", "k=10", "#e0a868"), ("LoRA (merged)", "QuALITY", "k=1", "k=10", "#c0392b"), ("LoRA (merged)", "TechQA", "k=1", "k=3", "#d4695c")] idx = {(m, b, v): val for m, b, _, v, val in MULTI} for i, (method, bench, a, b, colour) in enumerate(series): lo, hi = idx[(method, bench, a)], idx[(method, bench, b)] ax2.plot([0, 1], [lo, hi], marker="o", color=colour, label=f"{method}, {bench} ({a} to {b})") ax2.annotate(f"{hi - lo:+.1f}", xy=(1.02, hi), fontsize=8, color=colour, va="center") ax2.set_xticks([0, 1]) ax2.set_xticklabels(["fewer retrieved documents", "more retrieved documents"], fontsize=9) ax2.set_xlim(-0.08, 1.35) ax2.set_ylim(0, 92) # headroom below the lowest series so the legend sits clear of the lines ax2.set_ylabel("score") ax2.set_title("Composing per-document artifacts:\nonly the KV caches survive extra documents", fontsize=10) ax2.legend(fontsize=7, loc="lower left") ax2.grid(axis="y", alpha=0.25) fig.suptitle("arXiv 2609.17346v1, values as published in Tables 2 and 4 and section 5.2", fontsize=9) fig.tight_layout(rect=(0, 0, 1, 0.96)) os.makedirs(os.path.dirname(IMG_PATH), exist_ok=True) fig.savefig(IMG_PATH, dpi=140) print(f"\nwrote {IMG_PATH}") def main() -> None: ap = argparse.ArgumentParser(description=__doc__) ap.add_argument("--no-chart", action="store_true", help="skip the matplotlib figure") args = ap.parse_args() rows = build_rows() write_csv(rows) check_averages() check_gaps() check_scale_qa() if not args.no_chart: draw_chart() if __name__ == "__main__": main()