Spaces:
Sleeping
Sleeping
Download tools/classify.py from AdithyaSK/dataagent-phase0-evals: direct link, hf CLI and curl.
- Browser
- Download file 7.08 kB
-
https://huggingface.co/spaces/AdithyaSK/dataagent-phase0-evals/resolve/main/tools/classify.py
- Command line
-
hf download hf://spaces/AdithyaSK/dataagent-phase0-evals/tools/classify.py
-
curl -L -o classify.py https://huggingface.co/spaces/AdithyaSK/dataagent-phase0-evals/resolve/main/tools/classify.py
7.08 kB
| """Derive the stable / experimental tiers from a sweep, instead of asserting them. | |
| The tier is a claim about evidence, so it should be computed from the evidence and carry it. A badge | |
| that someone typed by hand goes stale the moment a harness improves or regresses, and a tier with no | |
| stated reason is just a colour. | |
| RULES, in the order they are applied: | |
| stable the harness produced graded rollouts for essentially every task AND has a verified | |
| training run. Both halves matter: capture working proves the tokens are right, and a | |
| completed training step proves the trainer can consume them — they are separate failure | |
| modes, and this stack has hit each independently. | |
| experimental anything else, with the specific gap named. Never a bare tier. | |
| WHAT IS NOT A REASON TO DOWNGRADE. A low pass rate. A harness scoring 0.0 on hard tasks is working | |
| correctly and reporting a real result; treating that as a defect would rank harnesses by how easy their | |
| tasks were. Only unmeasured rollouts, pauses, and known skew count against a harness here. | |
| Prompt re-render skew IS recorded as a caveat rather than a downgrade on its own: it is harmless for | |
| eval (nothing is trained) and disqualifying for training, so the caveat says which. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| HERE = Path(__file__).resolve().parents[1] | |
| # Measured over 670 turns against the engine's own prompt_token_ids. Eval-safe, training-unsafe. | |
| KNOWN_SKEW = { | |
| "claude-code": "+2 tokens per prompt re-render — harmless for eval, forks every turn when training", | |
| "gemini-cli": "+2 tokens per prompt re-render — harmless for eval, forks every turn when training", | |
| "kimi-cli": "-10 tokens per tool call — the largest measured skew; unsafe to train on", | |
| } | |
| # Reads os.environ inside run(), so concurrency depends on the context-local overlay. | |
| NEEDS_OVERLAY = {"goose", "claude-code", "gemini-cli"} | |
| NO_STEP_LIMIT_NOTE = ("no step-limit expression in its seam, so rollouts run to the timeout; " | |
| "the kill surfaces as exit 137 and the rollout is retried") | |
| def classify(sweep: dict, trained: set[str], measured_floor: float) -> dict: | |
| summary = sweep.get("summary", {}) | |
| per = summary.get("harnesses", {}) | |
| paused = summary.get("paused_harnesses", {}) | |
| k = summary.get("k", 4) | |
| out = {} | |
| for harness, m in sorted(per.items()): | |
| caveats, tier = [], "experimental" | |
| n_tasks = m.get("n_tasks") or 0 | |
| measured = m.get("n_measured") or 0 | |
| coverage = (measured / n_tasks) if n_tasks else 0.0 | |
| if harness in paused: | |
| caveats.append(f"PAUSED mid-sweep: {paused[harness]}") | |
| elif coverage < measured_floor: | |
| caveats.append( | |
| f"only {measured}/{n_tasks} tasks produced a graded rollout " | |
| f"({coverage:.0%} < {measured_floor:.0%} required)" | |
| ) | |
| elif harness not in trained: | |
| caveats.append( | |
| "eval measured but no verified training run — capture working does not prove the " | |
| "trainer can consume it, which is a separate failure mode" | |
| ) | |
| else: | |
| tier = "stable" | |
| if harness in KNOWN_SKEW: | |
| caveats.append(KNOWN_SKEW[harness]) | |
| if harness in NEEDS_OVERLAY: | |
| caveats.append("reads os.environ inside run(); concurrent only via the context-local overlay") | |
| entry = { | |
| "tier": tier, | |
| "evidence": ( | |
| f"pass@{k} {m.get(f'pass@{k}')}, pass@1 {m.get('pass@1')}, " | |
| f"{measured}/{n_tasks} tasks measured, mean {m.get('mean_turns')} turns" | |
| ), | |
| } | |
| if caveats: | |
| entry["caveats"] = caveats | |
| out[harness] = entry | |
| # A harness that never appeared in the sweep at all is not 'experimental', it is untested — saying | |
| # otherwise would imply it was tried. | |
| for harness in paused: | |
| out.setdefault(harness, {"tier": "experimental", "caveats": [f"PAUSED: {paused[harness]}"]}) | |
| return out | |
| def main() -> int: | |
| ap = argparse.ArgumentParser() | |
| # Several files, because a sweep can be split across jobs — and it was: one 15-harness job projected | |
| # past its time limit, so it became three. Merging here rather than requiring one file means the | |
| # split is an operational detail instead of something the classification has to know about. | |
| ap.add_argument("--sweep", required=True, nargs="+", help="one or more eval sweep JSONs") | |
| ap.add_argument("--project", default="data-agent") | |
| ap.add_argument("--trained", default="mini-swe-agent,opencode", | |
| help="harnesses with a verified training run") | |
| ap.add_argument("--measured-floor", type=float, default=0.9, | |
| help="fraction of tasks that must produce a graded rollout to be stable") | |
| ap.add_argument("--dry-run", action="store_true") | |
| args = ap.parse_args() | |
| merged = {"summary": {"harnesses": {}, "paused_harnesses": {}, "k": None}} | |
| for f in args.sweep: | |
| one = json.loads(Path(f).read_text()) | |
| sm = one.get("summary", {}) | |
| merged["summary"]["k"] = merged["summary"]["k"] or sm.get("k") | |
| merged["summary"]["paused_harnesses"].update(sm.get("paused_harnesses") or {}) | |
| for h, m in (sm.get("harnesses") or {}).items(): | |
| prev = merged["summary"]["harnesses"].get(h) | |
| # A harness can appear in more than one file — opencode ran in the cancelled job AND in the | |
| # relaunch. Keep whichever measured more tasks: that is the more complete evidence, and | |
| # averaging two partial runs of different sizes would invent a number neither produced. | |
| if prev is None or (m.get("n_measured") or 0) > (prev.get("n_measured") or 0): | |
| merged["summary"]["harnesses"][h] = m | |
| sweep = merged | |
| trained = {h.strip() for h in args.trained.split(",") if h.strip()} | |
| support = classify(sweep, trained, args.measured_floor) | |
| for h, e in sorted(support.items(), key=lambda kv: (kv[1]["tier"] != "stable", kv[0])): | |
| print(f" {h:18s} {e['tier']:13s} {e.get('evidence','')}") | |
| for c in e.get("caveats", []): | |
| print(f" · {c}") | |
| if args.dry_run: | |
| return 0 | |
| p = HERE / "data" / "projects" / args.project / "project.json" | |
| d = json.loads(p.read_text()) if p.exists() else {"project_id": args.project} | |
| d["support"] = support | |
| d["tier_rule"] = ( | |
| "stable = graded rollouts on >=90% of tasks AND a verified training run. experimental = " | |
| "anything else, with the gap named. A low pass rate is never a downgrade: a harness scoring 0.0 " | |
| "is reporting a real result, and penalising that would rank harnesses by task difficulty." | |
| ) | |
| d["support_source"] = [Path(f).name for f in args.sweep] | |
| p.write_text(json.dumps(d, indent=2)) | |
| print(f"\nwrote {p.relative_to(HERE)}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |