File size: 6,539 Bytes
4568249
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
"""Inspect AI eval coverage against the inspect_evals package modules."""

from __future__ import annotations

# All task/eval modules in inspect_evals (excluding utility modules:
# constants, utils, metadata, hf_dataset_script_helper).
ALL_MODULES: list[str] = [
    "abstention_bench", "agent_bench", "agentdojo", "agentharm",
    "agentic_misalignment", "agieval", "ahb", "aime2024", "aime2025",
    "aime2026", "air_bench", "ape", "apps", "arc", "assistant_bench",
    "b3", "bbeh", "bbh", "bbq", "bfcl", "bigcodebench", "bold", "boolq",
    "browse_comp", "chembench", "class_eval", "coconot", "commonsense_qa",
    "compute_eval", "core_bench", "cti_realm", "cve_bench", "cybench",
    "cybergym", "cybermetric", "cyberseceval_2", "cyberseceval_3",
    "docvqa", "drop", "ds1000", "fortress", "frontier_cs", "frontierscience",
    "gaia", "gdm_in_house_ctf", "gdm_intercode_ctf", "gdm_self_proliferation",
    "gdm_self_reasoning", "gdm_stealth", "gdpval", "gpqa", "gsm8k",
    "healthbench", "hellaswag", "hle", "humaneval", "ifeval", "ifevalcode",
    "infinite_bench", "instrumentaleval", "ipi_coding_agent", "kernelbench",
    "lab_bench", "lingoly", "livebench", "livecodebench_pro", "make_me_pay",
    "makemesay", "mask", "math", "mathvista", "mbpp", "medqa", "mgsm",
    "mind2web", "mind2web_sc", "mle_bench", "mlrc_bench", "mmiu", "mmlu",
    "mmlu_pro", "mmmu", "moru", "musr", "niah", "novelty_bench", "onet",
    "osworld", "paperbench", "paws", "persistbench", "personality", "piqa",
    "pre_flight", "pubmedqa", "race_h", "sad", "scbench", "scicode",
    "sciknoweval", "sec_qa", "sevenllm", "simpleqa", "sosbench", "squad",
    "stereoset", "strong_reject", "swe_bench", "swe_lancer", "sycophancy",
    "tac", "tau2", "theagentcompany", "threecb", "truthfulqa", "uccb",
    "usaco", "vimgolf_challenges", "vqa_rad", "vstar_bench", "winogrande",
    "wmdp", "worldsense", "writingbench", "xstest", "zerobench",
]

# Intentionally skipped modules with reasons (fill in as needed).
SKIP_LIST: dict[str, str] = {}

# Overrides for task names that don't follow the standard module_ prefix.
# module_name -> task_name_prefix (or exact task name for single-task modules)
_PREFIX_OVERRIDES: dict[str, str] = {
    "agieval": "agie_",
    "ds1000": "DS-1000",
}


def module_for_task(task: str, module_set: set[str]) -> str | None:
    """Return the inspect_evals module that owns a given task name."""
    for mod, prefix in _PREFIX_OVERRIDES.items():
        if task == prefix or task.startswith(prefix):
            return mod
    best: str | None = None
    for mod in module_set:
        if task == mod or task.startswith(mod + "_") or task.startswith(mod + "-"):
            if best is None or len(mod) > len(best):
                best = mod
    return best


def compute_coverage(evaluated_tasks: set[str]) -> dict[str, dict]:
    """
    Returns dict: module -> {status, skip_reason, matched_tasks}
    status: "evaluated" | "not_started" | "skipped"
    """
    mod_set = set(ALL_MODULES)
    hits: dict[str, list[str]] = {m: [] for m in ALL_MODULES}
    for task in evaluated_tasks:
        mod = module_for_task(task, mod_set)
        if mod:
            hits[mod].append(task)

    return {
        mod: {
            "status": "skipped" if mod in SKIP_LIST else ("evaluated" if hits[mod] else "not_started"),
            "skip_reason": SKIP_LIST.get(mod),
            "matched_tasks": sorted(hits[mod]),
        }
        for mod in ALL_MODULES
    }


def summary_stats(coverage: dict[str, dict]) -> dict[str, int]:
    counts: dict[str, int] = {"evaluated": 0, "not_started": 0, "skipped": 0}
    for info in coverage.values():
        counts[info["status"]] += 1
    return counts


# ── CLI: generate inspect_summary.json ───────────────────────────────────────

if __name__ == "__main__":
    import json
    import os
    from datetime import datetime, timezone
    from pathlib import Path

    from huggingface_hub import HfApi, hf_hub_download
    from inspect_ai.log import read_eval_log

    REPO = "MIMIR-AI-ROUTER/inspect_evals"
    TOKEN = os.getenv("HF_TOKEN")
    _PREFER = ("accuracy", "mean", "correct", "f_score", "jailbreak_rate")

    def _primary(log):
        if not log.results:
            return None, None
        for s in log.results.scores or []:
            for p in _PREFER:
                if p in (s.metrics or {}) and s.metrics[p].value is not None:
                    return f"{s.name}/{p}", s.metrics[p].value
        for s in log.results.scores or []:
            for mk, mv in (s.metrics or {}).items():
                if mv.value is not None:
                    return f"{s.name}/{mk}", mv.value
        return None, None

    print("Fetching inspect eval file list...")
    api = HfApi(token=TOKEN)
    files = sorted(api.list_repo_files(REPO, repo_type="dataset"))
    eval_files = [f for f in files if f.endswith(".eval")]
    print(f"  {len(eval_files)} .eval files")

    entries = []
    for i, fname in enumerate(eval_files, 1):
        print(f"  [{i}/{len(eval_files)}] {fname[:70]}")
        local = hf_hub_download(REPO, fname, repo_type="dataset", token=TOKEN)
        log = read_eval_log(local)

        task = log.eval.task.split("/")[-1]
        model = log.eval.model
        n = log.eval.dataset.samples if log.eval.dataset else 0

        all_metrics: dict[str, float] = {}
        if log.results:
            for s in log.results.scores or []:
                for mk, mv in (s.metrics or {}).items():
                    if mv.value is not None:
                        all_metrics[f"{s.name}/{mk}"] = mv.value

        pm, pv = _primary(log)
        entries.append({
            "task": task,
            "model": model,
            "filename": fname,
            "n_samples": n,
            "metrics": all_metrics,
            "primary_metric": pm,
            "primary_value": pv,
        })

    out = {
        "generated_at": datetime.now(timezone.utc).isoformat(),
        "entries": entries,
    }
    dest = Path(__file__).parent / "inspect_summary.json"
    dest.write_text(json.dumps(out, indent=2))
    print(f"\nWrote {dest}  ({len(entries)} entries)")

    ev_tasks = {e["task"] for e in entries}
    cov = compute_coverage(ev_tasks)
    stats = summary_stats(cov)
    print(f"  evaluated modules : {stats['evaluated']}/{len(ALL_MODULES)}")
    print(f"  not started       : {stats['not_started']}")
    print(f"  skipped           : {stats['skipped']}")