benchmark_gate
benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).
scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05]
scripts/benchmark_gate.py --self-test
just benchmark writes a Markdown report whose first table is the summary — one row per arm
(baseline, rag-strict, rag-general) with the coverage columns; the kept runs under
docs/benchmarks/ have the same table, several per file. The gate reads one number from each:
rag-general coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md.
It fails when the new run's number is more than --tolerance below the best kept one, or
below the new run's own baseline — marola must still beat the plain prompt. --new/--kept
accept a file or a directory (newest file by name). Standard library only.
1#!/usr/bin/env python3 2"""benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9). 3 4 scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05] 5 scripts/benchmark_gate.py --self-test 6 7`just benchmark` writes a Markdown report whose first table is the summary — one row per arm 8(`baseline`, `rag-strict`, `rag-general`) with the coverage columns; the kept runs under 9docs/benchmarks/ have the same table, several per file. The gate reads one number from each: 10`rag-general` coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md. 11It fails when the new run's number is more than `--tolerance` below the best kept one, or 12below the new run's own `baseline` — marola must still beat the plain prompt. `--new`/`--kept` 13accept a file or a directory (newest file by name). Standard library only. 14""" 15 16import argparse 17import re 18import sys 19import tempfile 20from pathlib import Path 21 22HEADER = re.compile(r"^\|\s*arm\s*\|\s*coverage \(in-corpus\)\s*\|") 23ROW = re.compile(r"^\|\s*([a-z-]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|") 24METRIC_ARM = "rag-general" 25PLAIN_ARM = "baseline" 26 27 28def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]: 29 """Every summary table in the text: arm -> {in_corpus, general, all}.""" 30 tables: list[dict[str, dict[str, float]]] = [] 31 current: dict[str, dict[str, float]] | None = None 32 for line in markdown.splitlines(): 33 if HEADER.match(line): 34 current = {} 35 tables.append(current) 36 elif current is not None and (m := ROW.match(line)): 37 current[m.group(1)] = { 38 "in_corpus": float(m.group(2)), 39 "general": float(m.group(3)), 40 "all": float(m.group(4)), 41 } 42 elif current is not None and not line.startswith("|"): 43 current = None 44 return [t for t in tables if t] 45 46 47def newest(path: Path) -> Path: 48 if path.is_dir(): 49 files = sorted(p for p in path.glob("*.md") if p.is_file()) 50 if not files: 51 raise SystemExit(f"benchmark_gate: no *.md under {path}") 52 return files[-1] 53 return path 54 55 56def metric(table: dict[str, dict[str, float]], arm: str) -> float | None: 57 row = table.get(arm) 58 return None if row is None else row["all"] 59 60 61def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]: 62 new_tables = parse_tables(new_md) 63 kept_tables = parse_tables(kept_md) 64 lines: list[str] = [] 65 if not new_tables: 66 return False, ["no summary table in the new benchmark"] 67 if not kept_tables: 68 return False, ["no summary table in the kept benchmark"] 69 new = new_tables[0] 70 got = metric(new, METRIC_ARM) 71 plain = metric(new, PLAIN_ARM) 72 kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables) 73 if got is None or plain is None: 74 return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"] 75 ok = True 76 lines.append( 77 f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}" 78 ) 79 if got < kept_best - tolerance: 80 ok = False 81 lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}") 82 lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}") 83 if got < plain: 84 ok = False 85 lines.append( 86 f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it" 87 ) 88 lines.append("PASS" if ok else "the model is not promoted") 89 return ok, lines 90 91 92def self_test() -> int: 93 kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks" 94 kept = newest(kept_dir).read_text() 95 tables = parse_tables(kept) 96 assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}" 97 best = max(metric(t, METRIC_ARM) for t in tables) 98 assert abs(best - 0.84) < 1e-9, best # run 2 of docs/benchmarks/2026-09-05.md 99 100 good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n" 101 good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n" 102 ok, out = check(good, kept, 0.05) 103 assert ok, out 104 ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05) 105 assert not ok and any("below the kept" in line for line in out), out 106 ok, out = check( 107 good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05 108 ) 109 assert not ok and any("plain prompt" in line for line in out), out 110 ok, out = check("# nothing here\n", kept, 0.05) 111 assert not ok and "no summary table" in out[0] 112 with tempfile.TemporaryDirectory() as tmp: 113 d = Path(tmp) 114 (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |")) 115 (d / "benchmark-20260905-0950.md").write_text(good) 116 assert newest(d).name == "benchmark-20260905-0950.md" 117 print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})") 118 return 0 119 120 121def main(argv: list[str]) -> int: 122 ap = argparse.ArgumentParser( 123 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 124 ) 125 ap.add_argument("--self-test", action="store_true") 126 sub = ap.add_subparsers(dest="cmd") 127 chk = sub.add_parser("check") 128 chk.add_argument( 129 "--new", required=True, type=Path, help="a benchmark report, or the directory data/" 130 ) 131 chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path) 132 chk.add_argument("--tolerance", default=0.05, type=float) 133 args = ap.parse_args(argv) 134 if args.self_test: 135 return self_test() 136 if args.cmd != "check": 137 ap.print_help() 138 return 2 139 new_path, kept_path = newest(args.new), newest(args.kept) 140 print(f"new: {new_path}\nkept: {kept_path}") 141 ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance) 142 print("\n".join(lines)) 143 return 0 if ok else 1 144 145 146if __name__ == "__main__": 147 sys.exit(main(sys.argv[1:]))
29def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]: 30 """Every summary table in the text: arm -> {in_corpus, general, all}.""" 31 tables: list[dict[str, dict[str, float]]] = [] 32 current: dict[str, dict[str, float]] | None = None 33 for line in markdown.splitlines(): 34 if HEADER.match(line): 35 current = {} 36 tables.append(current) 37 elif current is not None and (m := ROW.match(line)): 38 current[m.group(1)] = { 39 "in_corpus": float(m.group(2)), 40 "general": float(m.group(3)), 41 "all": float(m.group(4)), 42 } 43 elif current is not None and not line.startswith("|"): 44 current = None 45 return [t for t in tables if t]
Every summary table in the text: arm -> {in_corpus, general, all}.
62def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]: 63 new_tables = parse_tables(new_md) 64 kept_tables = parse_tables(kept_md) 65 lines: list[str] = [] 66 if not new_tables: 67 return False, ["no summary table in the new benchmark"] 68 if not kept_tables: 69 return False, ["no summary table in the kept benchmark"] 70 new = new_tables[0] 71 got = metric(new, METRIC_ARM) 72 plain = metric(new, PLAIN_ARM) 73 kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables) 74 if got is None or plain is None: 75 return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"] 76 ok = True 77 lines.append( 78 f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}" 79 ) 80 if got < kept_best - tolerance: 81 ok = False 82 lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}") 83 lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}") 84 if got < plain: 85 ok = False 86 lines.append( 87 f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it" 88 ) 89 lines.append("PASS" if ok else "the model is not promoted") 90 return ok, lines
93def self_test() -> int: 94 kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks" 95 kept = newest(kept_dir).read_text() 96 tables = parse_tables(kept) 97 assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}" 98 best = max(metric(t, METRIC_ARM) for t in tables) 99 assert abs(best - 0.84) < 1e-9, best # run 2 of docs/benchmarks/2026-09-05.md 100 101 good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n" 102 good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n" 103 ok, out = check(good, kept, 0.05) 104 assert ok, out 105 ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05) 106 assert not ok and any("below the kept" in line for line in out), out 107 ok, out = check( 108 good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05 109 ) 110 assert not ok and any("plain prompt" in line for line in out), out 111 ok, out = check("# nothing here\n", kept, 0.05) 112 assert not ok and "no summary table" in out[0] 113 with tempfile.TemporaryDirectory() as tmp: 114 d = Path(tmp) 115 (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |")) 116 (d / "benchmark-20260905-0950.md").write_text(good) 117 assert newest(d).name == "benchmark-20260905-0950.md" 118 print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})") 119 return 0
122def main(argv: list[str]) -> int: 123 ap = argparse.ArgumentParser( 124 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 125 ) 126 ap.add_argument("--self-test", action="store_true") 127 sub = ap.add_subparsers(dest="cmd") 128 chk = sub.add_parser("check") 129 chk.add_argument( 130 "--new", required=True, type=Path, help="a benchmark report, or the directory data/" 131 ) 132 chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path) 133 chk.add_argument("--tolerance", default=0.05, type=float) 134 args = ap.parse_args(argv) 135 if args.self_test: 136 return self_test() 137 if args.cmd != "check": 138 ap.print_help() 139 return 2 140 new_path, kept_path = newest(args.new), newest(args.kept) 141 print(f"new: {new_path}\nkept: {kept_path}") 142 ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance) 143 print("\n".join(lines)) 144 return 0 if ok else 1