"""Row-level paired analysis for public-benchmark runs. Usage: python scripts/paired_analysis.py OUT_DIR NAME=details.jsonl [NAME=details.jsonl ...] For every run and task: accuracy with a Wilson 95% CI, an exact binomial test against chance, NLL against the ln(k) uniform reference, ECE (15 equal-width bins, as in eval/harness) with a percentile-bootstrap 95% CI, and the distribution of predicted positions next to the gold distribution. For every pair of runs: the paired accuracy delta with a paired-bootstrap 95% CI, the exact McNemar test, and the 2x2 flip matrix. """ from __future__ import annotations import itertools import json import math import sys from collections import Counter, defaultdict from pathlib import Path import numpy as np from scipy.stats import binomtest BOOTSTRAP = 2000 BINS = 15 SEED = 0 def wilson(k: int, n: int, z: float = 1.959964) -> tuple[float, float]: p = k / n d = 1 + z * z / n c = (p + z * z / (2 * n)) / d h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d return c - h, c + h def ece(conf: np.ndarray, correct: np.ndarray) -> float: total = 0.0 for b in range(BINS): lo, hi = b / BINS, (b + 1) / BINS m = (conf >= lo) & ((conf < hi) if b < BINS - 1 else (conf <= hi)) if m.any(): total += m.mean() * abs(conf[m].mean() - correct[m].mean()) return float(total) def mcnemar_exact(b: int, c: int) -> float: n = b + c return 1.0 if n == 0 else float(binomtest(min(b, c), n, 0.5).pvalue) def load(path: Path) -> dict[str, dict[int, dict]]: by: dict[str, dict[int, dict]] = defaultdict(dict) with path.open(encoding="utf-8") as fp: for line in fp: r = json.loads(line) by[r["task"]][int(r["index"])] = r return by def run_stats(rows: list[dict], rng: np.random.Generator) -> dict: n = len(rows) k = int(rows[0]["option_count"]) correct = np.array([bool(r["correct"]) for r in rows], dtype=float) conf = np.array([float(r["confidence"]) for r in rows]) nll = np.array([-math.log(max(float(r["p_gold"]), 1e-12)) for r in rows]) hits = int(correct.sum()) lo, hi = wilson(hits, n) idx = rng.integers(0, n, size=(BOOTSTRAP, n)) eces = [ece(conf[i], correct[i]) for i in idx] gold = Counter(int(r["gold"]) for r in rows) pred = Counter(int(r["pred"]) for r in rows) return { "n": n, "k": k, "accuracy": hits / n, "accuracy_ci95": [lo, hi], "chance": 1 / k, "binom_p_vs_chance_two_sided": float(binomtest(hits, n, 1 / k).pvalue), "nll": float(nll.mean()), "ln_k": math.log(k), "nll_worse_than_uniform": bool(nll.mean() > math.log(k)), "ece_15bin": ece(conf, correct), "ece_15bin_ci95": [float(np.percentile(eces, 2.5)), float(np.percentile(eces, 97.5))], "gold_position_share": [gold[i] / n for i in range(k)], "pred_position_share": [pred[i] / n for i in range(k)], "max_pred_position_share": max(pred[i] / n for i in range(k)), } def pair_stats(a: list[dict], b: list[dict], rng: np.random.Generator) -> dict: ca = np.array([bool(r["correct"]) for r in a], dtype=float) cb = np.array([bool(r["correct"]) for r in b], dtype=float) n = len(ca) diff = cb - ca idx = rng.integers(0, n, size=(BOOTSTRAP, n)) boots = diff[idx].mean(axis=1) both = int(((ca == 1) & (cb == 1)).sum()) a_only = int(((ca == 1) & (cb == 0)).sum()) b_only = int(((ca == 0) & (cb == 1)).sum()) neither = int(((ca == 0) & (cb == 0)).sum()) return { "n": n, "delta_pp": 100 * float(diff.mean()), "delta_pp_ci95": [100 * float(np.percentile(boots, 2.5)), 100 * float(np.percentile(boots, 97.5))], "flip": {"both": both, "a_only": a_only, "b_only": b_only, "neither": neither}, "net_rows": b_only - a_only, "mcnemar_exact_p": mcnemar_exact(a_only, b_only), } def main() -> None: out_dir = Path(sys.argv[1]) runs = {} for arg in sys.argv[2:]: name, path = arg.split("=", 1) runs[name] = load(Path(path)) out_dir.mkdir(parents=True, exist_ok=True) rng = np.random.default_rng(SEED) tasks = sorted(set.intersection(*(set(r) for r in runs.values())), key=list(next(iter(runs.values()))).index) report = {"bootstrap": BOOTSTRAP, "ece_bins": BINS, "seed": SEED, "runs": {}, "pairs": {}} for name, by in runs.items(): report["runs"][name] = {t: run_stats([by[t][i] for i in sorted(by[t])], rng) for t in tasks} for a, b in itertools.combinations(runs, 2): key = f"{a}->{b}" report["pairs"][key] = {} for t in tasks: ids = sorted(set(runs[a][t]) & set(runs[b][t])) for i in ids: if runs[a][t][i]["gold"] != runs[b][t][i]["gold"]: raise SystemExit(f"gold mismatch {t}:{i} between {a} and {b}") report["pairs"][key][t] = pair_stats([runs[a][t][i] for i in ids], [runs[b][t][i] for i in ids], rng) (out_dir / "paired_analysis.json").write_text(json.dumps(report, indent=2), encoding="utf-8") lines = ["# Paired analysis", "", f"Bootstrap resamples: {BOOTSTRAP}; ECE: {BINS} equal-width bins; seed {SEED}.", ""] for name, stats in report["runs"].items(): lines += [f"## {name}", "", "| Task | n | k | Acc | 95% CI | p vs chance | NLL | ln k | ECE (95% CI) | max pred-position share |", "|---|---:|---:|---:|---|---:|---:|---:|---|---:|"] for t, s in stats.items(): lines.append( f"| {t} | {s['n']} | {s['k']} | {100*s['accuracy']:.2f}% | [{100*s['accuracy_ci95'][0]:.1f}, {100*s['accuracy_ci95'][1]:.1f}] " f"| {s['binom_p_vs_chance_two_sided']:.2g} | {s['nll']:.3f}{' (>ln k)' if s['nll_worse_than_uniform'] else ''} | {s['ln_k']:.3f} " f"| {s['ece_15bin']:.3f} [{s['ece_15bin_ci95'][0]:.3f}, {s['ece_15bin_ci95'][1]:.3f}] | {100*s['max_pred_position_share']:.0f}% |" ) lines.append("") for key, stats in report["pairs"].items(): a, b = key.split("->") lines += [f"## {a} -> {b}", "", f"| Task | Δ pp | 95% CI | {a} only | {b} only | net rows | McNemar p |", "|---|---:|---|---:|---:|---:|---:|"] for t, s in stats.items(): lines.append( f"| {t} | {s['delta_pp']:+.2f} | [{s['delta_pp_ci95'][0]:+.2f}, {s['delta_pp_ci95'][1]:+.2f}] " f"| {s['flip']['a_only']} | {s['flip']['b_only']} | {s['net_rows']:+d} | {s['mcnemar_exact_p']:.2g} |" ) lines.append("") (out_dir / "paired_analysis.md").write_text("\n".join(lines), encoding="utf-8") print("\n".join(lines)) if __name__ == "__main__": main()