benchmark_gate
benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).
scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05] [--questions .tmp/resources/benchmark_questions.json]
scripts/benchmark_gate.py --self-test
just benchmark writes a Markdown report whose first table is the summary — one row per arm
(baseline, rag-strict, rag-general) with the coverage columns; the kept runs under
docs/benchmarks/ have the same table, several per file. The gate reads one number from each:
rag-general coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md.
It fails when the new run's number is more than --tolerance below the best kept one, or
below the new run's own baseline — marola must still beat the plain prompt. --new/--kept
accept a file or a directory (newest file by name). --questions is the app's question set from the
resources tarball (MIP-0070 §5.4): the new run must have answered exactly those ids on every arm, so
an image and a resources pin that drifted apart fail here. Standard library only.
1#!/usr/bin/env python3 2"""benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9). 3 4 scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05] \ 5 [--questions .tmp/resources/benchmark_questions.json] 6 scripts/benchmark_gate.py --self-test 7 8`just benchmark` writes a Markdown report whose first table is the summary — one row per arm 9(`baseline`, `rag-strict`, `rag-general`) with the coverage columns; the kept runs under 10docs/benchmarks/ have the same table, several per file. The gate reads one number from each: 11`rag-general` coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md. 12It fails when the new run's number is more than `--tolerance` below the best kept one, or 13below the new run's own `baseline` — marola must still beat the plain prompt. `--new`/`--kept` 14accept a file or a directory (newest file by name). `--questions` is the app's question set from the 15resources tarball (MIP-0070 §5.4): the new run must have answered exactly those ids on every arm, so 16an image and a resources pin that drifted apart fail here. Standard library only. 17""" 18 19import argparse 20import json 21import re 22import sys 23import tempfile 24from pathlib import Path 25 26HEADER = re.compile(r"^\|\s*arm\s*\|\s*coverage \(in-corpus\)\s*\|") 27ROW = re.compile(r"^\|\s*([a-z-]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|") 28METRIC_ARM = "rag-general" 29PLAIN_ARM = "baseline" 30ARMS = ("baseline", "rag-strict", "rag-general") 31PER_QUESTION_HEADER = re.compile(r"^\|\s*id\s*\|\s*topic\s*\|") 32 33 34def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]: 35 """Every summary table in the text: arm -> {in_corpus, general, all}.""" 36 tables: list[dict[str, dict[str, float]]] = [] 37 current: dict[str, dict[str, float]] | None = None 38 for line in markdown.splitlines(): 39 if HEADER.match(line): 40 current = {} 41 tables.append(current) 42 elif current is not None and (m := ROW.match(line)): 43 current[m.group(1)] = { 44 "in_corpus": float(m.group(2)), 45 "general": float(m.group(3)), 46 "all": float(m.group(4)), 47 } 48 elif current is not None and not line.startswith("|"): 49 current = None 50 return [t for t in tables if t] 51 52 53def newest(path: Path) -> Path: 54 if path.is_dir(): 55 files = sorted(p for p in path.glob("*.md") if p.is_file()) 56 if not files: 57 raise SystemExit(f"benchmark_gate: no *.md under {path}") 58 return files[-1] 59 return path 60 61 62def metric(table: dict[str, dict[str, float]], arm: str) -> float | None: 63 row = table.get(arm) 64 return None if row is None else row["all"] 65 66 67def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]: 68 new_tables = parse_tables(new_md) 69 kept_tables = parse_tables(kept_md) 70 lines: list[str] = [] 71 if not new_tables: 72 return False, ["no summary table in the new benchmark"] 73 if not kept_tables: 74 return False, ["no summary table in the kept benchmark"] 75 new = new_tables[0] 76 got = metric(new, METRIC_ARM) 77 plain = metric(new, PLAIN_ARM) 78 kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables) 79 if got is None or plain is None: 80 return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"] 81 ok = True 82 lines.append( 83 f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}" 84 ) 85 if got < kept_best - tolerance: 86 ok = False 87 lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}") 88 lines.append( 89 " after a deliberate model or embedder change: commit this run's report as " 90 "docs/benchmarks/<date>.md (it becomes the newest kept run), see docs/index.md" 91 ) 92 lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}") 93 if got < plain: 94 ok = False 95 lines.append( 96 f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it" 97 ) 98 lines.append("PASS" if ok else "the model is not promoted") 99 return ok, lines 100 101 102def question_problems(new_md: str, questions: list[dict]) -> list[str]: 103 """What the new run's per-question table lacks, or has extra, against the question set.""" 104 if not questions: 105 return ["the question set is empty"] 106 seen: set[tuple[str, str]] = set() 107 in_table = False 108 for line in new_md.splitlines(): 109 if PER_QUESTION_HEADER.match(line): 110 in_table = True 111 elif in_table and line.startswith("|") and not line.startswith("|---"): 112 cells = [c.strip() for c in line.strip().strip("|").split("|")] 113 if len(cells) >= 4: 114 seen.add((cells[0], cells[3])) 115 elif in_table and not line.startswith("|"): 116 in_table = False 117 if not seen: 118 return ["no per-question table in the new benchmark"] 119 want = {(q["id"], arm) for q in questions for arm in ARMS} 120 problems = [f"missing: {qid} on {arm}" for qid, arm in sorted(want - seen)] 121 problems += [f"not in the question set: {qid} on {arm}" for qid, arm in sorted(seen - want)] 122 return problems 123 124 125def self_test() -> int: 126 kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks" 127 kept = newest(kept_dir).read_text() 128 tables = parse_tables(kept) 129 assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}" 130 best = max(metric(t, METRIC_ARM) for t in tables) 131 assert abs(best - 0.84) < 1e-9, best # run 2 of docs/benchmarks/2026-09-05.md 132 133 good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n" 134 good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n" 135 ok, out = check(good, kept, 0.05) 136 assert ok, out 137 ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05) 138 assert not ok and any("below the kept" in line for line in out), out 139 assert any("docs/benchmarks/" in line for line in out), "the failure names no way forward" 140 ok, out = check( 141 good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05 142 ) 143 assert not ok and any("plain prompt" in line for line in out), out 144 ok, out = check("# nothing here\n", kept, 0.05) 145 assert not ok and "no summary table" in out[0] 146 with tempfile.TemporaryDirectory() as tmp: 147 d = Path(tmp) 148 (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |")) 149 (d / "benchmark-20260905-0950.md").write_text(good) 150 assert newest(d).name == "benchmark-20260905-0950.md" 151 152 questions = [{"id": "q01"}, {"id": "q02"}] 153 per_q = "\n## Per question\n\n| id | topic | in corpus | arm | coverage | cited | abstained | ms | answer (first 140 chars) |\n|---|---|---|---|---|---|---|---|---|\n" 154 full = ( 155 good 156 + per_q 157 + "".join( 158 f"| {q['id']} | safety | yes | {arm} | 1.00 | | | 1 | a |\n" 159 for q in questions 160 for arm in ARMS 161 ) 162 ) 163 assert question_problems(full, questions) == [], question_problems(full, questions) 164 partial = full.replace( 165 "| q02 | safety | yes | rag-general |", "| q03 | safety | yes | rag-general |" 166 ) 167 probs = question_problems(partial, questions) 168 assert any("q02" in p for p in probs) and any("q03" in p for p in probs), probs 169 assert question_problems(good, questions), "a report with no per-question table passed" 170 assert question_problems(full, []), "an empty question set passed" 171 print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})") 172 return 0 173 174 175def main(argv: list[str]) -> int: 176 ap = argparse.ArgumentParser( 177 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 178 ) 179 ap.add_argument("--self-test", action="store_true") 180 sub = ap.add_subparsers(dest="cmd") 181 chk = sub.add_parser("check") 182 chk.add_argument( 183 "--new", required=True, type=Path, help="a benchmark report, or the directory data/" 184 ) 185 chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path) 186 chk.add_argument("--tolerance", default=0.05, type=float) 187 chk.add_argument("--questions", type=Path, help="benchmark_questions.json (resources tarball)") 188 args = ap.parse_args(argv) 189 if args.self_test: 190 return self_test() 191 if args.cmd != "check": 192 ap.print_help() 193 return 2 194 new_path, kept_path = newest(args.new), newest(args.kept) 195 print(f"new: {new_path}\nkept: {kept_path}") 196 ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance) 197 if args.questions is not None: 198 problems = question_problems(new_path.read_text(), json.loads(args.questions.read_text())) 199 if problems: 200 ok = False 201 lines = lines[:-1] + [f"FAIL: {p} ({args.questions})" for p in problems] 202 lines.append("the model is not promoted") 203 print("\n".join(lines)) 204 return 0 if ok else 1 205 206 207if __name__ == "__main__": 208 sys.exit(main(sys.argv[1:]))
35def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]: 36 """Every summary table in the text: arm -> {in_corpus, general, all}.""" 37 tables: list[dict[str, dict[str, float]]] = [] 38 current: dict[str, dict[str, float]] | None = None 39 for line in markdown.splitlines(): 40 if HEADER.match(line): 41 current = {} 42 tables.append(current) 43 elif current is not None and (m := ROW.match(line)): 44 current[m.group(1)] = { 45 "in_corpus": float(m.group(2)), 46 "general": float(m.group(3)), 47 "all": float(m.group(4)), 48 } 49 elif current is not None and not line.startswith("|"): 50 current = None 51 return [t for t in tables if t]
Every summary table in the text: arm -> {in_corpus, general, all}.
68def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]: 69 new_tables = parse_tables(new_md) 70 kept_tables = parse_tables(kept_md) 71 lines: list[str] = [] 72 if not new_tables: 73 return False, ["no summary table in the new benchmark"] 74 if not kept_tables: 75 return False, ["no summary table in the kept benchmark"] 76 new = new_tables[0] 77 got = metric(new, METRIC_ARM) 78 plain = metric(new, PLAIN_ARM) 79 kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables) 80 if got is None or plain is None: 81 return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"] 82 ok = True 83 lines.append( 84 f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}" 85 ) 86 if got < kept_best - tolerance: 87 ok = False 88 lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}") 89 lines.append( 90 " after a deliberate model or embedder change: commit this run's report as " 91 "docs/benchmarks/<date>.md (it becomes the newest kept run), see docs/index.md" 92 ) 93 lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}") 94 if got < plain: 95 ok = False 96 lines.append( 97 f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it" 98 ) 99 lines.append("PASS" if ok else "the model is not promoted") 100 return ok, lines
103def question_problems(new_md: str, questions: list[dict]) -> list[str]: 104 """What the new run's per-question table lacks, or has extra, against the question set.""" 105 if not questions: 106 return ["the question set is empty"] 107 seen: set[tuple[str, str]] = set() 108 in_table = False 109 for line in new_md.splitlines(): 110 if PER_QUESTION_HEADER.match(line): 111 in_table = True 112 elif in_table and line.startswith("|") and not line.startswith("|---"): 113 cells = [c.strip() for c in line.strip().strip("|").split("|")] 114 if len(cells) >= 4: 115 seen.add((cells[0], cells[3])) 116 elif in_table and not line.startswith("|"): 117 in_table = False 118 if not seen: 119 return ["no per-question table in the new benchmark"] 120 want = {(q["id"], arm) for q in questions for arm in ARMS} 121 problems = [f"missing: {qid} on {arm}" for qid, arm in sorted(want - seen)] 122 problems += [f"not in the question set: {qid} on {arm}" for qid, arm in sorted(seen - want)] 123 return problems
What the new run's per-question table lacks, or has extra, against the question set.
126def self_test() -> int: 127 kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks" 128 kept = newest(kept_dir).read_text() 129 tables = parse_tables(kept) 130 assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}" 131 best = max(metric(t, METRIC_ARM) for t in tables) 132 assert abs(best - 0.84) < 1e-9, best # run 2 of docs/benchmarks/2026-09-05.md 133 134 good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n" 135 good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n" 136 ok, out = check(good, kept, 0.05) 137 assert ok, out 138 ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05) 139 assert not ok and any("below the kept" in line for line in out), out 140 assert any("docs/benchmarks/" in line for line in out), "the failure names no way forward" 141 ok, out = check( 142 good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05 143 ) 144 assert not ok and any("plain prompt" in line for line in out), out 145 ok, out = check("# nothing here\n", kept, 0.05) 146 assert not ok and "no summary table" in out[0] 147 with tempfile.TemporaryDirectory() as tmp: 148 d = Path(tmp) 149 (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |")) 150 (d / "benchmark-20260905-0950.md").write_text(good) 151 assert newest(d).name == "benchmark-20260905-0950.md" 152 153 questions = [{"id": "q01"}, {"id": "q02"}] 154 per_q = "\n## Per question\n\n| id | topic | in corpus | arm | coverage | cited | abstained | ms | answer (first 140 chars) |\n|---|---|---|---|---|---|---|---|---|\n" 155 full = ( 156 good 157 + per_q 158 + "".join( 159 f"| {q['id']} | safety | yes | {arm} | 1.00 | | | 1 | a |\n" 160 for q in questions 161 for arm in ARMS 162 ) 163 ) 164 assert question_problems(full, questions) == [], question_problems(full, questions) 165 partial = full.replace( 166 "| q02 | safety | yes | rag-general |", "| q03 | safety | yes | rag-general |" 167 ) 168 probs = question_problems(partial, questions) 169 assert any("q02" in p for p in probs) and any("q03" in p for p in probs), probs 170 assert question_problems(good, questions), "a report with no per-question table passed" 171 assert question_problems(full, []), "an empty question set passed" 172 print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})") 173 return 0
176def main(argv: list[str]) -> int: 177 ap = argparse.ArgumentParser( 178 description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter 179 ) 180 ap.add_argument("--self-test", action="store_true") 181 sub = ap.add_subparsers(dest="cmd") 182 chk = sub.add_parser("check") 183 chk.add_argument( 184 "--new", required=True, type=Path, help="a benchmark report, or the directory data/" 185 ) 186 chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path) 187 chk.add_argument("--tolerance", default=0.05, type=float) 188 chk.add_argument("--questions", type=Path, help="benchmark_questions.json (resources tarball)") 189 args = ap.parse_args(argv) 190 if args.self_test: 191 return self_test() 192 if args.cmd != "check": 193 ap.print_help() 194 return 2 195 new_path, kept_path = newest(args.new), newest(args.kept) 196 print(f"new: {new_path}\nkept: {kept_path}") 197 ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance) 198 if args.questions is not None: 199 problems = question_problems(new_path.read_text(), json.loads(args.questions.read_text())) 200 if problems: 201 ok = False 202 lines = lines[:-1] + [f"FAIL: {p} ({args.questions})" for p in problems] 203 lines.append("the model is not promoted") 204 print("\n".join(lines)) 205 return 0 if ok else 1