Download harness/scripts/final_2x2.py from agentic-ptb/opus-max-record: direct link, hf CLI and curl.
- Browser
- Download file 5.52 kB
-
https://huggingface.co/agentic-ptb/opus-max-record/resolve/main/harness/scripts/final_2x2.py
- Command line
-
hf download hf://agentic-ptb/opus-max-record/harness/scripts/final_2x2.py
-
curl -L -o final_2x2.py https://huggingface.co/agentic-ptb/opus-max-record/resolve/main/harness/scripts/final_2x2.py
5.52 kB
| #!/usr/bin/env python3 | |
| """The final 2x2: {base, submitted} x {stock, pi-ws} on one shared task set. | |
| Prints the cell table (solve rate + Wilson 95% CI) and every comparison that matters, each as a | |
| *paired* McNemar test on the tasks the two runs actually share. Paired is the right test here: | |
| `verifiers` pins SEED=0, so `shuffle=true` draws the identical sample every run and the arms line | |
| up task for task. An unpaired test on the same data throws that away and needs a far larger effect | |
| to reach the same confidence. | |
| Usage: | |
| final_2x2.py base_stock=runs/base250_stock base_ws=runs/base250_ws \ | |
| sub_stock=runs/big_swe_stock sub_ws=runs/big_swe_ws | |
| Reports `ungraded` per cell — episodes with no reward record at all, i.e. lost to infrastructure | |
| rather than failed. They are excluded from the rate, and a cell where they differ a lot between | |
| arms is not a fair comparison however good the p-value looks. | |
| """ | |
| import json | |
| import math | |
| import os | |
| import sys | |
| def wilson(k, n, z=1.96): | |
| if n == 0: | |
| return 0.0, 0.0, 0.0 | |
| p = k / n | |
| d = 1 + z * z / n | |
| c = (p + z * z / (2 * n)) / d | |
| h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d | |
| return p, max(0.0, c - h), min(1.0, c + h) | |
| def mcnemar_exact(b, c): | |
| """Two-sided exact McNemar on discordant pairs: b wins for A, c wins for B.""" | |
| n = b + c | |
| if n == 0: | |
| return 1.0 | |
| tail = sum(math.comb(n, i) for i in range(0, min(b, c) + 1)) * 0.5**n | |
| return min(1.0, 2 * tail) | |
| def load(path): | |
| """-> {task_key: solved_bool}, ungraded_count. Task key prefers the dataset's own name.""" | |
| f = os.path.join(path, "traces.jsonl") | |
| per, ungraded = {}, 0 | |
| if not os.path.exists(f): | |
| return per, ungraded | |
| for line in open(f): | |
| rec = json.loads(line) | |
| for t in rec.get("traces", [rec]): | |
| rw = {k: v for k, v in (t.get("rewards") or {}).items() | |
| if isinstance(v, dict) and v.get("score") is not None} | |
| if not rw: | |
| ungraded += 1 | |
| continue | |
| data = (t.get("task") or {}).get("data") or {} | |
| key = data.get("name") or data.get("instance_id") or data.get("idx") | |
| if key is None: | |
| continue | |
| total = sum(v["score"] * v.get("weight", 1.0) for v in rw.values()) | |
| weight = sum(v.get("weight", 1.0) for v in rw.values()) or 1.0 | |
| per[key] = (total / weight) > 0 | |
| return per, ungraded | |
| def main(): | |
| cells = {} | |
| for arg in sys.argv[1:]: | |
| name, _, path = arg.partition("=") | |
| per, ungraded = load(path) | |
| cells[name] = (per, ungraded, path) | |
| if not cells: | |
| print(__doc__) | |
| return | |
| print(f"{'cell':<12} {'n':>4} {'solved':>7} {'score':>7} {'ci95':>16} {'ungraded':>8}") | |
| for name, (per, ungraded, path) in cells.items(): | |
| n, k = len(per), sum(per.values()) | |
| p, lo, hi = wilson(k, n) | |
| print(f"{name:<12} {n:>4} {k:>7} {p:>7.3f} [{lo:.3f}, {hi:.3f}] {ungraded:>8}") | |
| print() | |
| print("paired comparisons (only tasks present in both runs):") | |
| print(f"{'A -> B':<28} {'shared':>7} {'only A':>7} {'only B':>7} {'delta':>8} {'McNemar':>9}") | |
| names = list(cells) | |
| for i, a in enumerate(names): | |
| for b in names[i + 1:]: | |
| pa, pb = cells[a][0], cells[b][0] | |
| shared = set(pa) & set(pb) | |
| if not shared: | |
| continue | |
| only_a = sum(1 for t in shared if pa[t] and not pb[t]) | |
| only_b = sum(1 for t in shared if pb[t] and not pa[t]) | |
| ka = sum(1 for t in shared if pa[t]) | |
| kb = sum(1 for t in shared if pb[t]) | |
| delta = (kb - ka) / len(shared) | |
| p = mcnemar_exact(only_a, only_b) | |
| print(f"{a+' -> '+b:<28} {len(shared):>7} {only_a:>7} {only_b:>7} {delta:>+8.3f} {p:>9.4f}") | |
| # Matched subset. Cells finished at different n (the pool starved the base arms), and a | |
| # partially-completed run over-represents easy tasks — episodes that finish first are the | |
| # short ones. Restricting every cell to the tasks graded in ALL of them removes that | |
| # composition difference: the subset may be easier than the full suite, but it is the same | |
| # subset for all four, so cell-to-cell comparisons are clean. | |
| if len(cells) > 2: | |
| common = set.intersection(*[set(c[0]) for c in cells.values()]) | |
| if common: | |
| print() | |
| print(f"matched subset — the {len(common)} tasks graded in ALL cells:") | |
| for name, (per, _u, _p) in cells.items(): | |
| k = sum(per[t] for t in common) | |
| p_, lo, hi = wilson(k, len(common)) | |
| print(f" {name:<12} {k:>3}/{len(common)} = {p_:.3f} [{lo:.3f}, {hi:.3f}]") | |
| print() | |
| print(f" {'A -> B':<26} {'only A':>7} {'only B':>7} {'delta':>8} {'McNemar':>9}") | |
| for i, a in enumerate(names): | |
| for b in names[i + 1:]: | |
| pa, pb = cells[a][0], cells[b][0] | |
| only_a = sum(1 for t in common if pa[t] and not pb[t]) | |
| only_b = sum(1 for t in common if pb[t] and not pa[t]) | |
| ka = sum(1 for t in common if pa[t]) | |
| kb = sum(1 for t in common if pb[t]) | |
| delta = (kb - ka) / len(common) | |
| p = mcnemar_exact(only_a, only_b) | |
| print(f" {a+' -> '+b:<26} {only_a:>7} {only_b:>7} {delta:>+8.3f} {p:>9.4f}") | |
| if __name__ == "__main__": | |
| main() | |