benchmark_gate

benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).

scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05]         [--questions .tmp/resources/benchmark_questions.json]
scripts/benchmark_gate.py --self-test

just benchmark writes a Markdown report whose first table is the summary — one row per arm (baseline, rag-strict, rag-general) with the coverage columns; the kept runs under docs/benchmarks/ have the same table, several per file. The gate reads one number from each: rag-general coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md. It fails when the new run's number is more than --tolerance below the best kept one, or below the new run's own baseline — marola must still beat the plain prompt. --new/--kept accept a file or a directory (newest file by name). --questions is the app's question set from the resources tarball (MIP-0070 §5.4): the new run must have answered exactly those ids on every arm, so an image and a resources pin that drifted apart fail here. Standard library only.

  1#!/usr/bin/env python3
  2"""benchmark_gate — the promotion gate for the marola-local image (MIP-0008 §5.3, tasks decision 9).
  3
  4    scripts/benchmark_gate.py check --new data --kept docs/benchmarks [--tolerance 0.05] \
  5        [--questions .tmp/resources/benchmark_questions.json]
  6    scripts/benchmark_gate.py --self-test
  7
  8`just benchmark` writes a Markdown report whose first table is the summary — one row per arm
  9(`baseline`, `rag-strict`, `rag-general`) with the coverage columns; the kept runs under
 10docs/benchmarks/ have the same table, several per file. The gate reads one number from each:
 11`rag-general` coverage (all), "the result that matters" in docs/benchmarks/2026-09-05.md.
 12It fails when the new run's number is more than `--tolerance` below the best kept one, or
 13below the new run's own `baseline` — marola must still beat the plain prompt. `--new`/`--kept`
 14accept a file or a directory (newest file by name). `--questions` is the app's question set from the
 15resources tarball (MIP-0070 §5.4): the new run must have answered exactly those ids on every arm, so
 16an image and a resources pin that drifted apart fail here. Standard library only.
 17"""
 18
 19import argparse
 20import json
 21import re
 22import sys
 23import tempfile
 24from pathlib import Path
 25
 26HEADER = re.compile(r"^\|\s*arm\s*\|\s*coverage \(in-corpus\)\s*\|")
 27ROW = re.compile(r"^\|\s*([a-z-]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|\s*([\d.]+)\s*\|")
 28METRIC_ARM = "rag-general"
 29PLAIN_ARM = "baseline"
 30ARMS = ("baseline", "rag-strict", "rag-general")
 31PER_QUESTION_HEADER = re.compile(r"^\|\s*id\s*\|\s*topic\s*\|")
 32
 33
 34def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
 35    """Every summary table in the text: arm -> {in_corpus, general, all}."""
 36    tables: list[dict[str, dict[str, float]]] = []
 37    current: dict[str, dict[str, float]] | None = None
 38    for line in markdown.splitlines():
 39        if HEADER.match(line):
 40            current = {}
 41            tables.append(current)
 42        elif current is not None and (m := ROW.match(line)):
 43            current[m.group(1)] = {
 44                "in_corpus": float(m.group(2)),
 45                "general": float(m.group(3)),
 46                "all": float(m.group(4)),
 47            }
 48        elif current is not None and not line.startswith("|"):
 49            current = None
 50    return [t for t in tables if t]
 51
 52
 53def newest(path: Path) -> Path:
 54    if path.is_dir():
 55        files = sorted(p for p in path.glob("*.md") if p.is_file())
 56        if not files:
 57            raise SystemExit(f"benchmark_gate: no *.md under {path}")
 58        return files[-1]
 59    return path
 60
 61
 62def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
 63    row = table.get(arm)
 64    return None if row is None else row["all"]
 65
 66
 67def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
 68    new_tables = parse_tables(new_md)
 69    kept_tables = parse_tables(kept_md)
 70    lines: list[str] = []
 71    if not new_tables:
 72        return False, ["no summary table in the new benchmark"]
 73    if not kept_tables:
 74        return False, ["no summary table in the kept benchmark"]
 75    new = new_tables[0]
 76    got = metric(new, METRIC_ARM)
 77    plain = metric(new, PLAIN_ARM)
 78    kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables)
 79    if got is None or plain is None:
 80        return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"]
 81    ok = True
 82    lines.append(
 83        f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}"
 84    )
 85    if got < kept_best - tolerance:
 86        ok = False
 87        lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}")
 88        lines.append(
 89            "  after a deliberate model or embedder change: commit this run's report as "
 90            "docs/benchmarks/<date>.md (it becomes the newest kept run), see docs/index.md"
 91        )
 92    lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}")
 93    if got < plain:
 94        ok = False
 95        lines.append(
 96            f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it"
 97        )
 98    lines.append("PASS" if ok else "the model is not promoted")
 99    return ok, lines
100
101
102def question_problems(new_md: str, questions: list[dict]) -> list[str]:
103    """What the new run's per-question table lacks, or has extra, against the question set."""
104    if not questions:
105        return ["the question set is empty"]
106    seen: set[tuple[str, str]] = set()
107    in_table = False
108    for line in new_md.splitlines():
109        if PER_QUESTION_HEADER.match(line):
110            in_table = True
111        elif in_table and line.startswith("|") and not line.startswith("|---"):
112            cells = [c.strip() for c in line.strip().strip("|").split("|")]
113            if len(cells) >= 4:
114                seen.add((cells[0], cells[3]))
115        elif in_table and not line.startswith("|"):
116            in_table = False
117    if not seen:
118        return ["no per-question table in the new benchmark"]
119    want = {(q["id"], arm) for q in questions for arm in ARMS}
120    problems = [f"missing: {qid} on {arm}" for qid, arm in sorted(want - seen)]
121    problems += [f"not in the question set: {qid} on {arm}" for qid, arm in sorted(seen - want)]
122    return problems
123
124
125def self_test() -> int:
126    kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks"
127    kept = newest(kept_dir).read_text()
128    tables = parse_tables(kept)
129    assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}"
130    best = max(metric(t, METRIC_ARM) for t in tables)
131    assert abs(best - 0.84) < 1e-9, best  # run 2 of docs/benchmarks/2026-09-05.md
132
133    good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n"
134    good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n"
135    ok, out = check(good, kept, 0.05)
136    assert ok, out
137    ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05)
138    assert not ok and any("below the kept" in line for line in out), out
139    assert any("docs/benchmarks/" in line for line in out), "the failure names no way forward"
140    ok, out = check(
141        good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05
142    )
143    assert not ok and any("plain prompt" in line for line in out), out
144    ok, out = check("# nothing here\n", kept, 0.05)
145    assert not ok and "no summary table" in out[0]
146    with tempfile.TemporaryDirectory() as tmp:
147        d = Path(tmp)
148        (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |"))
149        (d / "benchmark-20260905-0950.md").write_text(good)
150        assert newest(d).name == "benchmark-20260905-0950.md"
151
152    questions = [{"id": "q01"}, {"id": "q02"}]
153    per_q = "\n## Per question\n\n| id | topic | in corpus | arm | coverage | cited | abstained | ms | answer (first 140 chars) |\n|---|---|---|---|---|---|---|---|---|\n"
154    full = (
155        good
156        + per_q
157        + "".join(
158            f"| {q['id']} | safety | yes | {arm} | 1.00 | | | 1 | a |\n"
159            for q in questions
160            for arm in ARMS
161        )
162    )
163    assert question_problems(full, questions) == [], question_problems(full, questions)
164    partial = full.replace(
165        "| q02 | safety | yes | rag-general |", "| q03 | safety | yes | rag-general |"
166    )
167    probs = question_problems(partial, questions)
168    assert any("q02" in p for p in probs) and any("q03" in p for p in probs), probs
169    assert question_problems(good, questions), "a report with no per-question table passed"
170    assert question_problems(full, []), "an empty question set passed"
171    print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})")
172    return 0
173
174
175def main(argv: list[str]) -> int:
176    ap = argparse.ArgumentParser(
177        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
178    )
179    ap.add_argument("--self-test", action="store_true")
180    sub = ap.add_subparsers(dest="cmd")
181    chk = sub.add_parser("check")
182    chk.add_argument(
183        "--new", required=True, type=Path, help="a benchmark report, or the directory data/"
184    )
185    chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path)
186    chk.add_argument("--tolerance", default=0.05, type=float)
187    chk.add_argument("--questions", type=Path, help="benchmark_questions.json (resources tarball)")
188    args = ap.parse_args(argv)
189    if args.self_test:
190        return self_test()
191    if args.cmd != "check":
192        ap.print_help()
193        return 2
194    new_path, kept_path = newest(args.new), newest(args.kept)
195    print(f"new: {new_path}\nkept: {kept_path}")
196    ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance)
197    if args.questions is not None:
198        problems = question_problems(new_path.read_text(), json.loads(args.questions.read_text()))
199        if problems:
200            ok = False
201            lines = lines[:-1] + [f"FAIL: {p} ({args.questions})" for p in problems]
202            lines.append("the model is not promoted")
203    print("\n".join(lines))
204    return 0 if ok else 1
205
206
207if __name__ == "__main__":
208    sys.exit(main(sys.argv[1:]))
ROW = re.compile('^\\|\\s*([a-z-]+)\\s*\\|\\s*([\\d.]+)\\s*\\|\\s*([\\d.]+)\\s*\\|\\s*([\\d.]+)\\s*\\|')
METRIC_ARM = 'rag-general'
PLAIN_ARM = 'baseline'
ARMS = ('baseline', 'rag-strict', 'rag-general')
PER_QUESTION_HEADER = re.compile('^\\|\\s*id\\s*\\|\\s*topic\\s*\\|')
def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
35def parse_tables(markdown: str) -> list[dict[str, dict[str, float]]]:
36    """Every summary table in the text: arm -> {in_corpus, general, all}."""
37    tables: list[dict[str, dict[str, float]]] = []
38    current: dict[str, dict[str, float]] | None = None
39    for line in markdown.splitlines():
40        if HEADER.match(line):
41            current = {}
42            tables.append(current)
43        elif current is not None and (m := ROW.match(line)):
44            current[m.group(1)] = {
45                "in_corpus": float(m.group(2)),
46                "general": float(m.group(3)),
47                "all": float(m.group(4)),
48            }
49        elif current is not None and not line.startswith("|"):
50            current = None
51    return [t for t in tables if t]

Every summary table in the text: arm -> {in_corpus, general, all}.

def newest(path: pathlib.Path) -> pathlib.Path:
54def newest(path: Path) -> Path:
55    if path.is_dir():
56        files = sorted(p for p in path.glob("*.md") if p.is_file())
57        if not files:
58            raise SystemExit(f"benchmark_gate: no *.md under {path}")
59        return files[-1]
60    return path
def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
63def metric(table: dict[str, dict[str, float]], arm: str) -> float | None:
64    row = table.get(arm)
65    return None if row is None else row["all"]
def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
 68def check(new_md: str, kept_md: str, tolerance: float) -> tuple[bool, list[str]]:
 69    new_tables = parse_tables(new_md)
 70    kept_tables = parse_tables(kept_md)
 71    lines: list[str] = []
 72    if not new_tables:
 73        return False, ["no summary table in the new benchmark"]
 74    if not kept_tables:
 75        return False, ["no summary table in the kept benchmark"]
 76    new = new_tables[0]
 77    got = metric(new, METRIC_ARM)
 78    plain = metric(new, PLAIN_ARM)
 79    kept_best = max((metric(t, METRIC_ARM) or 0.0) for t in kept_tables)
 80    if got is None or plain is None:
 81        return False, [f"the new benchmark lacks a {METRIC_ARM} or {PLAIN_ARM} row"]
 82    ok = True
 83    lines.append(
 84        f"{METRIC_ARM} coverage (all): new {got:.2f}, best kept {kept_best:.2f}, tolerance {tolerance:.2f}"
 85    )
 86    if got < kept_best - tolerance:
 87        ok = False
 88        lines.append(f"FAIL: {got:.2f} is more than {tolerance:.2f} below the kept {kept_best:.2f}")
 89        lines.append(
 90            "  after a deliberate model or embedder change: commit this run's report as "
 91            "docs/benchmarks/<date>.md (it becomes the newest kept run), see docs/index.md"
 92        )
 93    lines.append(f"{PLAIN_ARM} coverage (all) in the new run: {plain:.2f}")
 94    if got < plain:
 95        ok = False
 96        lines.append(
 97            f"FAIL: {METRIC_ARM} {got:.2f} is below the plain prompt {plain:.2f} — marola no longer beats it"
 98        )
 99    lines.append("PASS" if ok else "the model is not promoted")
100    return ok, lines
def question_problems(new_md: str, questions: list[dict]) -> list[str]:
103def question_problems(new_md: str, questions: list[dict]) -> list[str]:
104    """What the new run's per-question table lacks, or has extra, against the question set."""
105    if not questions:
106        return ["the question set is empty"]
107    seen: set[tuple[str, str]] = set()
108    in_table = False
109    for line in new_md.splitlines():
110        if PER_QUESTION_HEADER.match(line):
111            in_table = True
112        elif in_table and line.startswith("|") and not line.startswith("|---"):
113            cells = [c.strip() for c in line.strip().strip("|").split("|")]
114            if len(cells) >= 4:
115                seen.add((cells[0], cells[3]))
116        elif in_table and not line.startswith("|"):
117            in_table = False
118    if not seen:
119        return ["no per-question table in the new benchmark"]
120    want = {(q["id"], arm) for q in questions for arm in ARMS}
121    problems = [f"missing: {qid} on {arm}" for qid, arm in sorted(want - seen)]
122    problems += [f"not in the question set: {qid} on {arm}" for qid, arm in sorted(seen - want)]
123    return problems

What the new run's per-question table lacks, or has extra, against the question set.

def self_test() -> int:
126def self_test() -> int:
127    kept_dir = Path(__file__).resolve().parent.parent / "docs" / "benchmarks"
128    kept = newest(kept_dir).read_text()
129    tables = parse_tables(kept)
130    assert len(tables) >= 2, f"expected several runs in {kept_dir}, parsed {len(tables)}"
131    best = max(metric(t, METRIC_ARM) for t in tables)
132    assert abs(best - 0.84) < 1e-9, best  # run 2 of docs/benchmarks/2026-09-05.md
133
134    good = "| arm | coverage (in-corpus) | coverage (general) | coverage (all) | cited | abstained | mean ms |\n|---|---|---|---|---|---|---|\n"
135    good += "| baseline | 0.52 | 0.92 | 0.73 | 0% | 0% | 479 |\n| rag-strict | 0.67 | 0.03 | 0.32 | 41% | 55% | 322 |\n| rag-general | 0.88 | 0.79 | 0.83 | 36% | 0% | 545 |\n"
136    ok, out = check(good, kept, 0.05)
137    assert ok, out
138    ok, out = check(good.replace("| 0.83 |", "| 0.70 |"), kept, 0.05)
139    assert not ok and any("below the kept" in line for line in out), out
140    assert any("docs/benchmarks/" in line for line in out), "the failure names no way forward"
141    ok, out = check(
142        good.replace("| 0.73 |", "| 0.90 |").replace("| 0.83 |", "| 0.80 |"), kept, 0.05
143    )
144    assert not ok and any("plain prompt" in line for line in out), out
145    ok, out = check("# nothing here\n", kept, 0.05)
146    assert not ok and "no summary table" in out[0]
147    with tempfile.TemporaryDirectory() as tmp:
148        d = Path(tmp)
149        (d / "benchmark-20260901-0000.md").write_text(good.replace("| 0.83 |", "| 0.10 |"))
150        (d / "benchmark-20260905-0950.md").write_text(good)
151        assert newest(d).name == "benchmark-20260905-0950.md"
152
153    questions = [{"id": "q01"}, {"id": "q02"}]
154    per_q = "\n## Per question\n\n| id | topic | in corpus | arm | coverage | cited | abstained | ms | answer (first 140 chars) |\n|---|---|---|---|---|---|---|---|---|\n"
155    full = (
156        good
157        + per_q
158        + "".join(
159            f"| {q['id']} | safety | yes | {arm} | 1.00 | | | 1 | a |\n"
160            for q in questions
161            for arm in ARMS
162        )
163    )
164    assert question_problems(full, questions) == [], question_problems(full, questions)
165    partial = full.replace(
166        "| q02 | safety | yes | rag-general |", "| q03 | safety | yes | rag-general |"
167    )
168    probs = question_problems(partial, questions)
169    assert any("q02" in p for p in probs) and any("q03" in p for p in probs), probs
170    assert question_problems(good, questions), "a report with no per-question table passed"
171    assert question_problems(full, []), "an empty question set passed"
172    print(f"benchmark_gate self-test: ok (kept best {METRIC_ARM} = {best:.2f})")
173    return 0
def main(argv: list[str]) -> int:
176def main(argv: list[str]) -> int:
177    ap = argparse.ArgumentParser(
178        description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter
179    )
180    ap.add_argument("--self-test", action="store_true")
181    sub = ap.add_subparsers(dest="cmd")
182    chk = sub.add_parser("check")
183    chk.add_argument(
184        "--new", required=True, type=Path, help="a benchmark report, or the directory data/"
185    )
186    chk.add_argument("--kept", default=Path("docs/benchmarks"), type=Path)
187    chk.add_argument("--tolerance", default=0.05, type=float)
188    chk.add_argument("--questions", type=Path, help="benchmark_questions.json (resources tarball)")
189    args = ap.parse_args(argv)
190    if args.self_test:
191        return self_test()
192    if args.cmd != "check":
193        ap.print_help()
194        return 2
195    new_path, kept_path = newest(args.new), newest(args.kept)
196    print(f"new: {new_path}\nkept: {kept_path}")
197    ok, lines = check(new_path.read_text(), kept_path.read_text(), args.tolerance)
198    if args.questions is not None:
199        problems = question_problems(new_path.read_text(), json.loads(args.questions.read_text()))
200        if problems:
201            ok = False
202            lines = lines[:-1] + [f"FAIL: {p} ({args.questions})" for p in problems]
203            lines.append("the model is not promoted")
204    print("\n".join(lines))
205    return 0 if ok else 1