Download harness/scripts/score.py from agentic-ptb/opus-max-record: direct link, hf CLI and curl.
- Browser
- Download file 2 kB
-
https://huggingface.co/agentic-ptb/opus-max-record/resolve/main/harness/scripts/score.py
- Command line
-
hf download hf://agentic-ptb/opus-max-record/harness/scripts/score.py
-
curl -L -o score.py https://huggingface.co/agentic-ptb/opus-max-record/resolve/main/harness/scripts/score.py
2 kB
| #!/usr/bin/env python3 | |
| """Summarise a verifiers eval output dir: mean reward + Wilson 95% CI.""" | |
| import json, sys, math, os, collections | |
| def wilson(k, n, z=1.96): | |
| if n == 0: return (0.0, 0.0, 0.0) | |
| p = k / n | |
| d = 1 + z*z/n | |
| c = (p + z*z/(2*n)) / d | |
| h = z*math.sqrt(p*(1-p)/n + z*z/(4*n*n)) / d | |
| return (p, max(0.0, c-h), min(1.0, c+h)) | |
| def summarize(path): | |
| f = os.path.join(path, "traces.jsonl") | |
| rows = [json.loads(l) for l in open(f)] if os.path.exists(f) else [] | |
| scores, stops, errs, ntok, turns, dur = [], collections.Counter(), 0, [], [], [] | |
| for r in rows: | |
| for t in r.get("traces", []): | |
| rw = {k: v for k, v in (t.get("rewards") or {}).items() | |
| if isinstance(v, dict) and v.get("score") is not None} | |
| if not rw: | |
| errs += 1; continue | |
| tot = sum(v["score"]*v.get("weight",1.0) for v in rw.values()) | |
| wsum = sum(v.get("weight",1.0) for v in rw.values()) or 1.0 | |
| scores.append(tot/wsum) | |
| stops[t.get("stop_condition")] += 1 | |
| calls = t.get("calls") or [] | |
| turns.append(len(calls)) | |
| ntok.append(sum((c.get("usage") or {}).get("completion_tokens",0) for c in calls)) | |
| ti = t.get("timing") or {} | |
| a = ti.get("agent") or {} | |
| if a.get("start") and a.get("end"): dur.append(a["end"]-a["start"]) | |
| if not r.get("ok", True): errs += 1 | |
| n = len(scores); k = sum(1 for s in scores if s >= 0.999) | |
| p, lo, hi = wilson(k, n) | |
| print(f"{path}") | |
| print(f" n={n} solved={k} score={p:.4f} ci95=[{lo:.4f},{hi:.4f}] errors={errs}") | |
| if n: | |
| print(f" mean_reward={sum(scores)/n:.4f} mean_turns={sum(turns)/len(turns):.1f} " | |
| f"mean_completion_tok={sum(ntok)/len(ntok):.0f} mean_agent_sec={(sum(dur)/len(dur) if dur else 0):.0f}") | |
| print(f" stops={dict(stops)}") | |
| return p, lo, hi, n, k | |
| if __name__ == "__main__": | |
| for p in sys.argv[1:]: summarize(p) | |