Spaces:
Sleeping
Sleeping
File size: 6,539 Bytes
4568249 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 | """Inspect AI eval coverage against the inspect_evals package modules."""
from __future__ import annotations
# All task/eval modules in inspect_evals (excluding utility modules:
# constants, utils, metadata, hf_dataset_script_helper).
ALL_MODULES: list[str] = [
"abstention_bench", "agent_bench", "agentdojo", "agentharm",
"agentic_misalignment", "agieval", "ahb", "aime2024", "aime2025",
"aime2026", "air_bench", "ape", "apps", "arc", "assistant_bench",
"b3", "bbeh", "bbh", "bbq", "bfcl", "bigcodebench", "bold", "boolq",
"browse_comp", "chembench", "class_eval", "coconot", "commonsense_qa",
"compute_eval", "core_bench", "cti_realm", "cve_bench", "cybench",
"cybergym", "cybermetric", "cyberseceval_2", "cyberseceval_3",
"docvqa", "drop", "ds1000", "fortress", "frontier_cs", "frontierscience",
"gaia", "gdm_in_house_ctf", "gdm_intercode_ctf", "gdm_self_proliferation",
"gdm_self_reasoning", "gdm_stealth", "gdpval", "gpqa", "gsm8k",
"healthbench", "hellaswag", "hle", "humaneval", "ifeval", "ifevalcode",
"infinite_bench", "instrumentaleval", "ipi_coding_agent", "kernelbench",
"lab_bench", "lingoly", "livebench", "livecodebench_pro", "make_me_pay",
"makemesay", "mask", "math", "mathvista", "mbpp", "medqa", "mgsm",
"mind2web", "mind2web_sc", "mle_bench", "mlrc_bench", "mmiu", "mmlu",
"mmlu_pro", "mmmu", "moru", "musr", "niah", "novelty_bench", "onet",
"osworld", "paperbench", "paws", "persistbench", "personality", "piqa",
"pre_flight", "pubmedqa", "race_h", "sad", "scbench", "scicode",
"sciknoweval", "sec_qa", "sevenllm", "simpleqa", "sosbench", "squad",
"stereoset", "strong_reject", "swe_bench", "swe_lancer", "sycophancy",
"tac", "tau2", "theagentcompany", "threecb", "truthfulqa", "uccb",
"usaco", "vimgolf_challenges", "vqa_rad", "vstar_bench", "winogrande",
"wmdp", "worldsense", "writingbench", "xstest", "zerobench",
]
# Intentionally skipped modules with reasons (fill in as needed).
SKIP_LIST: dict[str, str] = {}
# Overrides for task names that don't follow the standard module_ prefix.
# module_name -> task_name_prefix (or exact task name for single-task modules)
_PREFIX_OVERRIDES: dict[str, str] = {
"agieval": "agie_",
"ds1000": "DS-1000",
}
def module_for_task(task: str, module_set: set[str]) -> str | None:
"""Return the inspect_evals module that owns a given task name."""
for mod, prefix in _PREFIX_OVERRIDES.items():
if task == prefix or task.startswith(prefix):
return mod
best: str | None = None
for mod in module_set:
if task == mod or task.startswith(mod + "_") or task.startswith(mod + "-"):
if best is None or len(mod) > len(best):
best = mod
return best
def compute_coverage(evaluated_tasks: set[str]) -> dict[str, dict]:
"""
Returns dict: module -> {status, skip_reason, matched_tasks}
status: "evaluated" | "not_started" | "skipped"
"""
mod_set = set(ALL_MODULES)
hits: dict[str, list[str]] = {m: [] for m in ALL_MODULES}
for task in evaluated_tasks:
mod = module_for_task(task, mod_set)
if mod:
hits[mod].append(task)
return {
mod: {
"status": "skipped" if mod in SKIP_LIST else ("evaluated" if hits[mod] else "not_started"),
"skip_reason": SKIP_LIST.get(mod),
"matched_tasks": sorted(hits[mod]),
}
for mod in ALL_MODULES
}
def summary_stats(coverage: dict[str, dict]) -> dict[str, int]:
counts: dict[str, int] = {"evaluated": 0, "not_started": 0, "skipped": 0}
for info in coverage.values():
counts[info["status"]] += 1
return counts
# ββ CLI: generate inspect_summary.json βββββββββββββββββββββββββββββββββββββββ
if __name__ == "__main__":
import json
import os
from datetime import datetime, timezone
from pathlib import Path
from huggingface_hub import HfApi, hf_hub_download
from inspect_ai.log import read_eval_log
REPO = "MIMIR-AI-ROUTER/inspect_evals"
TOKEN = os.getenv("HF_TOKEN")
_PREFER = ("accuracy", "mean", "correct", "f_score", "jailbreak_rate")
def _primary(log):
if not log.results:
return None, None
for s in log.results.scores or []:
for p in _PREFER:
if p in (s.metrics or {}) and s.metrics[p].value is not None:
return f"{s.name}/{p}", s.metrics[p].value
for s in log.results.scores or []:
for mk, mv in (s.metrics or {}).items():
if mv.value is not None:
return f"{s.name}/{mk}", mv.value
return None, None
print("Fetching inspect eval file list...")
api = HfApi(token=TOKEN)
files = sorted(api.list_repo_files(REPO, repo_type="dataset"))
eval_files = [f for f in files if f.endswith(".eval")]
print(f" {len(eval_files)} .eval files")
entries = []
for i, fname in enumerate(eval_files, 1):
print(f" [{i}/{len(eval_files)}] {fname[:70]}")
local = hf_hub_download(REPO, fname, repo_type="dataset", token=TOKEN)
log = read_eval_log(local)
task = log.eval.task.split("/")[-1]
model = log.eval.model
n = log.eval.dataset.samples if log.eval.dataset else 0
all_metrics: dict[str, float] = {}
if log.results:
for s in log.results.scores or []:
for mk, mv in (s.metrics or {}).items():
if mv.value is not None:
all_metrics[f"{s.name}/{mk}"] = mv.value
pm, pv = _primary(log)
entries.append({
"task": task,
"model": model,
"filename": fname,
"n_samples": n,
"metrics": all_metrics,
"primary_metric": pm,
"primary_value": pv,
})
out = {
"generated_at": datetime.now(timezone.utc).isoformat(),
"entries": entries,
}
dest = Path(__file__).parent / "inspect_summary.json"
dest.write_text(json.dumps(out, indent=2))
print(f"\nWrote {dest} ({len(entries)} entries)")
ev_tasks = {e["task"] for e in entries}
cov = compute_coverage(ev_tasks)
stats = summary_stats(cov)
print(f" evaluated modules : {stats['evaluated']}/{len(ALL_MODULES)}")
print(f" not started : {stats['not_started']}")
print(f" skipped : {stats['skipped']}")
|