openjev-e4b / eval /analysis /paired_analysis.py
bambamdevs's picture
Publish OpenJEV E4B 1.0
03223d7
Raw History Blame Contribute Delete
6.79 kB
"""Row-level paired analysis for public-benchmark runs.
Usage:
python scripts/paired_analysis.py OUT_DIR NAME=details.jsonl [NAME=details.jsonl ...]
For every run and task: accuracy with a Wilson 95% CI, an exact binomial test
against chance, NLL against the ln(k) uniform reference, ECE (15 equal-width
bins, as in eval/harness) with a percentile-bootstrap 95% CI, and the
distribution of predicted positions next to the gold distribution.
For every pair of runs: the paired accuracy delta with a paired-bootstrap 95%
CI, the exact McNemar test, and the 2x2 flip matrix.
"""
from __future__ import annotations
import itertools
import json
import math
import sys
from collections import Counter, defaultdict
from pathlib import Path
import numpy as np
from scipy.stats import binomtest
BOOTSTRAP = 2000
BINS = 15
SEED = 0
def wilson(k: int, n: int, z: float = 1.959964) -> tuple[float, float]:
p = k / n
d = 1 + z * z / n
c = (p + z * z / (2 * n)) / d
h = z * math.sqrt(p * (1 - p) / n + z * z / (4 * n * n)) / d
return c - h, c + h
def ece(conf: np.ndarray, correct: np.ndarray) -> float:
total = 0.0
for b in range(BINS):
lo, hi = b / BINS, (b + 1) / BINS
m = (conf >= lo) & ((conf < hi) if b < BINS - 1 else (conf <= hi))
if m.any():
total += m.mean() * abs(conf[m].mean() - correct[m].mean())
return float(total)
def mcnemar_exact(b: int, c: int) -> float:
n = b + c
return 1.0 if n == 0 else float(binomtest(min(b, c), n, 0.5).pvalue)
def load(path: Path) -> dict[str, dict[int, dict]]:
by: dict[str, dict[int, dict]] = defaultdict(dict)
with path.open(encoding="utf-8") as fp:
for line in fp:
r = json.loads(line)
by[r["task"]][int(r["index"])] = r
return by
def run_stats(rows: list[dict], rng: np.random.Generator) -> dict:
n = len(rows)
k = int(rows[0]["option_count"])
correct = np.array([bool(r["correct"]) for r in rows], dtype=float)
conf = np.array([float(r["confidence"]) for r in rows])
nll = np.array([-math.log(max(float(r["p_gold"]), 1e-12)) for r in rows])
hits = int(correct.sum())
lo, hi = wilson(hits, n)
idx = rng.integers(0, n, size=(BOOTSTRAP, n))
eces = [ece(conf[i], correct[i]) for i in idx]
gold = Counter(int(r["gold"]) for r in rows)
pred = Counter(int(r["pred"]) for r in rows)
return {
"n": n,
"k": k,
"accuracy": hits / n,
"accuracy_ci95": [lo, hi],
"chance": 1 / k,
"binom_p_vs_chance_two_sided": float(binomtest(hits, n, 1 / k).pvalue),
"nll": float(nll.mean()),
"ln_k": math.log(k),
"nll_worse_than_uniform": bool(nll.mean() > math.log(k)),
"ece_15bin": ece(conf, correct),
"ece_15bin_ci95": [float(np.percentile(eces, 2.5)), float(np.percentile(eces, 97.5))],
"gold_position_share": [gold[i] / n for i in range(k)],
"pred_position_share": [pred[i] / n for i in range(k)],
"max_pred_position_share": max(pred[i] / n for i in range(k)),
}
def pair_stats(a: list[dict], b: list[dict], rng: np.random.Generator) -> dict:
ca = np.array([bool(r["correct"]) for r in a], dtype=float)
cb = np.array([bool(r["correct"]) for r in b], dtype=float)
n = len(ca)
diff = cb - ca
idx = rng.integers(0, n, size=(BOOTSTRAP, n))
boots = diff[idx].mean(axis=1)
both = int(((ca == 1) & (cb == 1)).sum())
a_only = int(((ca == 1) & (cb == 0)).sum())
b_only = int(((ca == 0) & (cb == 1)).sum())
neither = int(((ca == 0) & (cb == 0)).sum())
return {
"n": n,
"delta_pp": 100 * float(diff.mean()),
"delta_pp_ci95": [100 * float(np.percentile(boots, 2.5)), 100 * float(np.percentile(boots, 97.5))],
"flip": {"both": both, "a_only": a_only, "b_only": b_only, "neither": neither},
"net_rows": b_only - a_only,
"mcnemar_exact_p": mcnemar_exact(a_only, b_only),
}
def main() -> None:
out_dir = Path(sys.argv[1])
runs = {}
for arg in sys.argv[2:]:
name, path = arg.split("=", 1)
runs[name] = load(Path(path))
out_dir.mkdir(parents=True, exist_ok=True)
rng = np.random.default_rng(SEED)
tasks = sorted(set.intersection(*(set(r) for r in runs.values())), key=list(next(iter(runs.values()))).index)
report = {"bootstrap": BOOTSTRAP, "ece_bins": BINS, "seed": SEED, "runs": {}, "pairs": {}}
for name, by in runs.items():
report["runs"][name] = {t: run_stats([by[t][i] for i in sorted(by[t])], rng) for t in tasks}
for a, b in itertools.combinations(runs, 2):
key = f"{a}->{b}"
report["pairs"][key] = {}
for t in tasks:
ids = sorted(set(runs[a][t]) & set(runs[b][t]))
for i in ids:
if runs[a][t][i]["gold"] != runs[b][t][i]["gold"]:
raise SystemExit(f"gold mismatch {t}:{i} between {a} and {b}")
report["pairs"][key][t] = pair_stats([runs[a][t][i] for i in ids], [runs[b][t][i] for i in ids], rng)
(out_dir / "paired_analysis.json").write_text(json.dumps(report, indent=2), encoding="utf-8")
lines = ["# Paired analysis", "", f"Bootstrap resamples: {BOOTSTRAP}; ECE: {BINS} equal-width bins; seed {SEED}.", ""]
for name, stats in report["runs"].items():
lines += [f"## {name}", "", "| Task | n | k | Acc | 95% CI | p vs chance | NLL | ln k | ECE (95% CI) | max pred-position share |",
"|---|---:|---:|---:|---|---:|---:|---:|---|---:|"]
for t, s in stats.items():
lines.append(
f"| {t} | {s['n']} | {s['k']} | {100*s['accuracy']:.2f}% | [{100*s['accuracy_ci95'][0]:.1f}, {100*s['accuracy_ci95'][1]:.1f}] "
f"| {s['binom_p_vs_chance_two_sided']:.2g} | {s['nll']:.3f}{' (>ln k)' if s['nll_worse_than_uniform'] else ''} | {s['ln_k']:.3f} "
f"| {s['ece_15bin']:.3f} [{s['ece_15bin_ci95'][0]:.3f}, {s['ece_15bin_ci95'][1]:.3f}] | {100*s['max_pred_position_share']:.0f}% |"
)
lines.append("")
for key, stats in report["pairs"].items():
a, b = key.split("->")
lines += [f"## {a} -> {b}", "", f"| Task | Δ pp | 95% CI | {a} only | {b} only | net rows | McNemar p |", "|---|---:|---|---:|---:|---:|---:|"]
for t, s in stats.items():
lines.append(
f"| {t} | {s['delta_pp']:+.2f} | [{s['delta_pp_ci95'][0]:+.2f}, {s['delta_pp_ci95'][1]:+.2f}] "
f"| {s['flip']['a_only']} | {s['flip']['b_only']} | {s['net_rows']:+d} | {s['mcnemar_exact_p']:.2g} |"
)
lines.append("")
(out_dir / "paired_analysis.md").write_text("\n".join(lines), encoding="utf-8")
print("\n".join(lines))
if __name__ == "__main__":
main()