benchmark_gate

benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).

scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05]
scripts/benchmark_gate.py --self-test

just benchmark writes a Markdown report whose first table is the summary — one row per arm (baseline, rag-strict, rag-general) with the coverage columns; the kept runs under docs/benchmarks/ have the same table, several per file. The gate reads one number from each: rag-general coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md. It fails when the new run's number is more than --tolerance below the best kept one, or below the new run's own baseline — marola must still beat the plain prompt. --new/--kept accept a file or a directory (newest file by name). Standard library only.

  1#!/usr/bin/env python3
  2"""benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).
  3
  4    scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05]
  5    scripts/benchmark_gate.py --self-test
  6
  7`just benchmark` writes a Markdown report whose first table is the summary — one row per arm
  8(`baseline`, `rag-strict`, `rag-general`) with the coverage columns; the kept runs under
  9docs/benchmarks/ have the same table, several per file. The gate reads one number from each:
 10`rag-general` coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md.
 11It fails when the new run's number is more than `--tolerance` below the best kept one, or
 12below the new run's own `baseline` — marola must still beat the plain prompt. `--new`/`--kept`
 13accept a file or a directory (newest file by name). Standard library only.
 14"""
 15
 16import argparse
 17import re
 18import sys
 19import tempfile
 20from pathlib import Path
 21
 22HEADER = re.compile(r"^\|\s*arm\s*\|\s*coverage \(in-corpus\)\s*\|")
 23ROW = re.compile(r"^\|\s*([a-z-]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|")
 24METRIC_ARM = "rag-general"
 25PLAIN_ARM = "baseline"
 26
 27
 28def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
 29    """Every summary table in the text: arm -> {in_corpus, general, all}."""
 30    tables: list[dict[str, dict[str, float]]] = []
 31    current: dict[str, dict[str, float]] | None = None
 32    for line in markdown.splitlines():
 33        if HEADER.match(line):
 34            current = {}
 35            tables.append(current)
 36        elif current is not None and (m := ROW.match(line)):
 37            current[m.group(1)] = {
 38                "in_corpus": float(m.group(2)),
 39                "general": float(m.group(3)),
 40                "all": float(m.group(4)),
 41            }
 42        elif current is not None and not line.startswith("|"):
 43            current = None
 44    return [t for t in tables if t]
 45
 46
 47def newest(path: Path) -> Path:
 48    if path.is_dir():
 49        files = sorted(p for p in path.glob("*.md") if p.is_file())
 50        if not files:
 51            raise SystemExit(f"benchmark_gate: no *.md under {path}")
 52        return files[-1]
 53    return path
 54
 55
 56def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
 57    row = table.get(arm)
 58    return None if row is None else row["all"]
 59
 60
 61def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
 62    new_tables = parse_tables(new_md)
 63    kept_tables = parse_tables(kept_md)
 64    lines: list[str] = []
 65    if not new_tables:
 66        return False, ["no summary table in the new benchmark"]
 67    if not kept_tables:
 68        return False, ["no summary table in the kept benchmark"]
 69    new = new_tables[0]
 70    got = metric(new, METRIC_ARM)
 71    plain = metric(new, PLAIN_ARM)
 72    kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables)
 73    if got is None or plain is None:
 74        return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"]
 75    ok = True
 76    lines.append(
 77        f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}"
 78    )
 79    if got < kept_best - tolerance:
 80        ok = False
 81        lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}")
 82    lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}")
 83    if got < plain:
 84        ok = False
 85        lines.append(
 86            f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it"
 87        )
 88    lines.append("PASS" if ok else "the model is not promoted")
 89    return ok, lines
 90
 91
 92def self_test() -> int:
 93    kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks"
 94    kept = newest(kept_dir).read_text()
 95    tables = parse_tables(kept)
 96    assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}"
 97    best = max(metric(t, METRIC_ARM) for t in tables)
 98    assert abs(best - 0.84) < 1e-9, best  # run 2 of docs/benchmarks/2026-09-05.md
 99
100    good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n"
101    good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n"
102    ok, out = check(good, kept, 0.05)
103    assert ok, out
104    ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05)
105    assert not ok and any("below the kept" in line for line in out), out
106    ok, out = check(
107        good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05
108    )
109    assert not ok and any("plain prompt" in line for line in out), out
110    ok, out = check("# nothing here\n", kept, 0.05)
111    assert not ok and "no summary table" in out[0]
112    with tempfile.TemporaryDirectory() as tmp:
113        d = Path(tmp)
114        (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |"))
115        (d / "benchmark-20260905-0950.md").write_text(good)
116        assert newest(d).name == "benchmark-20260905-0950.md"
117    print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})")
118    return 0
119
120
121def main(argv: list[str]) -> int:
122    ap = argparse.ArgumentParser(
123        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
124    )
125    ap.add_argument("--self-test", action="store_true")
126    sub = ap.add_subparsers(dest="cmd")
127    chk = sub.add_parser("check")
128    chk.add_argument(
129        "--new", required=True, type=Path, help="a benchmark report, or the directory data/"
130    )
131    chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path)
132    chk.add_argument("--tolerance", default=0.05, type=float)
133    args = ap.parse_args(argv)
134    if args.self_test:
135        return self_test()
136    if args.cmd != "check":
137        ap.print_help()
138        return 2
139    new_path, kept_path = newest(args.new), newest(args.kept)
140    print(f"new: {new_path}\nkept: {kept_path}")
141    ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance)
142    print("\n".join(lines))
143    return 0 if ok else 1
144
145
146if __name__ == "__main__":
147    sys.exit(main(sys.argv[1:]))
ROW = re.compile('^\\|\\s*([a-z-]+)\\s*\\|\\s*([\\d.]+)\\s*\\|\\s*([\\d.]+)\\s*\\|\\s*([\\d.]+)\\s*\\|')
METRIC_ARM = 'rag-general'
PLAIN_ARM = 'baseline'
def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
29def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
30    """Every summary table in the text: arm -> {in_corpus, general, all}."""
31    tables: list[dict[str, dict[str, float]]] = []
32    current: dict[str, dict[str, float]] | None = None
33    for line in markdown.splitlines():
34        if HEADER.match(line):
35            current = {}
36            tables.append(current)
37        elif current is not None and (m := ROW.match(line)):
38            current[m.group(1)] = {
39                "in_corpus": float(m.group(2)),
40                "general": float(m.group(3)),
41                "all": float(m.group(4)),
42            }
43        elif current is not None and not line.startswith("|"):
44            current = None
45    return [t for t in tables if t]

Every summary table in the text: arm -> {in_corpus, general, all}.

def newest(path: pathlib.Path) -> pathlib.Path:
48def newest(path: Path) -> Path:
49    if path.is_dir():
50        files = sorted(p for p in path.glob("*.md") if p.is_file())
51        if not files:
52            raise SystemExit(f"benchmark_gate: no *.md under {path}")
53        return files[-1]
54    return path
def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
57def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
58    row = table.get(arm)
59    return None if row is None else row["all"]
def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
62def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
63    new_tables = parse_tables(new_md)
64    kept_tables = parse_tables(kept_md)
65    lines: list[str] = []
66    if not new_tables:
67        return False, ["no summary table in the new benchmark"]
68    if not kept_tables:
69        return False, ["no summary table in the kept benchmark"]
70    new = new_tables[0]
71    got = metric(new, METRIC_ARM)
72    plain = metric(new, PLAIN_ARM)
73    kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables)
74    if got is None or plain is None:
75        return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"]
76    ok = True
77    lines.append(
78        f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}"
79    )
80    if got < kept_best - tolerance:
81        ok = False
82        lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}")
83    lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}")
84    if got < plain:
85        ok = False
86        lines.append(
87            f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it"
88        )
89    lines.append("PASS" if ok else "the model is not promoted")
90    return ok, lines
def self_test() -> int:
 93def self_test() -> int:
 94    kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks"
 95    kept = newest(kept_dir).read_text()
 96    tables = parse_tables(kept)
 97    assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}"
 98    best = max(metric(t, METRIC_ARM) for t in tables)
 99    assert abs(best - 0.84) < 1e-9, best  # run 2 of docs/benchmarks/2026-09-05.md
100
101    good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n"
102    good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n"
103    ok, out = check(good, kept, 0.05)
104    assert ok, out
105    ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05)
106    assert not ok and any("below the kept" in line for line in out), out
107    ok, out = check(
108        good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05
109    )
110    assert not ok and any("plain prompt" in line for line in out), out
111    ok, out = check("# nothing here\n", kept, 0.05)
112    assert not ok and "no summary table" in out[0]
113    with tempfile.TemporaryDirectory() as tmp:
114        d = Path(tmp)
115        (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |"))
116        (d / "benchmark-20260905-0950.md").write_text(good)
117        assert newest(d).name == "benchmark-20260905-0950.md"
118    print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})")
119    return 0
def main(argv: list[str]) -> int:
122def main(argv: list[str]) -> int:
123    ap = argparse.ArgumentParser(
124        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
125    )
126    ap.add_argument("--self-test", action="store_true")
127    sub = ap.add_subparsers(dest="cmd")
128    chk = sub.add_parser("check")
129    chk.add_argument(
130        "--new", required=True, type=Path, help="a benchmark report, or the directory data/"
131    )
132    chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path)
133    chk.add_argument("--tolerance", default=0.05, type=float)
134    args = ap.parse_args(argv)
135    if args.self_test:
136        return self_test()
137    if args.cmd != "check":
138        ap.print_help()
139        return 2
140    new_path, kept_path = newest(args.new), newest(args.kept)
141    print(f"new: {new_path}\nkept: {kept_path}")
142    ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance)
143    print("\n".join(lines))
144    return 0 if ok else 1