Cion-lab commited on
Commit
1c6dfb9
·
verified ·
1 Parent(s): c89f91b

run_benchmarks: task-list hard fail, per-task shot/denominator provenance, primary metric pre-registered, failed cells not zeros, push verified anonymously (E-045)

Browse files
Files changed (1) hide show
  1. eval/run_benchmarks.py +391 -276
eval/run_benchmarks.py CHANGED
@@ -1,276 +1,391 @@
1
- """Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
2
-
3
- Written before any score exists, and it refuses to deviate from the pin:
4
-
5
- * `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
6
- template -- a base model with a chat template would be an unearned capability (§3.9).
7
- * PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
8
- The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
9
- be tuned per task later either.
10
- * SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
11
- * One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
12
- moment it finishes so an interruption never loses a completed task (§5 Phase 6).
13
- * A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
14
- a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
15
- is not acceptable. Nothing is published if the smoke pass fails.
16
-
17
- No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
18
- the token never enters a log line or a published artifact.
19
- """
20
-
21
- import argparse
22
- import json
23
- import os
24
- import re
25
- import shutil
26
- import signal
27
- import subprocess
28
- import sys
29
- import time
30
-
31
- PINNED_LM_EVAL = "0.4.13"
32
- # Every row here is the harness's own task id, not a name invented for this script.
33
- TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
34
- "truthfulqa_mc2", "winogrande", "gsm8k"]
35
- # Chance level of the *metric being reported*, from the number of answer options in the task, not from a
36
- # remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
37
- CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
38
- "arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
39
- "hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
40
- "piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
41
- "truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
42
- "truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
43
- "gsm8k": ("0.00", "free-form exact match")}
44
-
45
-
46
- def sh(argv, label, timeout, env=None):
47
- """Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
48
- report nothing at all if the platform took the instance."""
49
- print("=== " + label, flush=True)
50
- t0 = time.time()
51
- e = dict(os.environ)
52
- e["PYTHONUNBUFFERED"] = "1"
53
- e.update(env or {})
54
- import threading
55
- # start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
56
- # notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
57
- p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
58
- env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
59
- start_new_session=True)
60
- killed = []
61
-
62
- def _kill():
63
- killed.append(True)
64
- try:
65
- os.killpg(os.getpgid(p.pid), signal.SIGTERM)
66
- except Exception:
67
- p.kill()
68
-
69
- timer = threading.Timer(timeout, _kill)
70
- timer.daemon = True
71
- timer.start()
72
- keep = []
73
- for line in p.stdout:
74
- line = line.rstrip("\n")
75
- keep.append(line)
76
- del keep[:-400]
77
- print(" |", line[:260], flush=True)
78
- timer.cancel()
79
- rc = p.wait()
80
- print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
81
- flush=True)
82
- return rc, "\n".join(keep)
83
-
84
-
85
- def tf_version():
86
- """Which transformers the harness will actually import -- see the --no-deps note below."""
87
- rc, out = sh([sys.executable, "-c", "import transformers; print(transformers.__version__)"],
88
- "probe_transformers", 180)
89
- for line in out.splitlines():
90
- if line.strip()[:1].isdigit():
91
- return line.strip()
92
- return "?"
93
-
94
-
95
- def install_harness():
96
- """Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
97
- transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
98
- resolve a different one."""
99
- rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
100
- have = ""
101
- for line in out.splitlines():
102
- if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
103
- have = line.strip().strip("|").strip()
104
- if have == PINNED_LM_EVAL:
105
- print("lm_eval", have, "already present", flush=True)
106
- return have
107
- print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
108
- flush=True)
109
- before = tf_version()
110
- # --no-deps deliberately. A dependency-resolving install can drop a different transformers into
111
- # ~/.local, where it shadows the image's copy on sys.path -- the exact thing this file's header says it
112
- # avoids, and it would move the harness pin silently (review finding C4).
113
- rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet", "--no-deps",
114
- "lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
115
- if rc != 0:
116
- raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
117
- after = tf_version()
118
- if before != after:
119
- raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
120
- "on the stack Gate 3 measured" % (before, after))
121
- rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
122
- "print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
123
- "verify_lm_eval", 300)
124
- got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
125
- if not got:
126
- raise SystemExit("lm_eval is not importable at the pinned version after install: "
127
- + out[-400:])
128
- return PINNED_LM_EVAL
129
-
130
-
131
- def run_task(model, task, shots, outdir, limit, gpu_h):
132
- """One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
133
- argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
134
- "--model_args", ",".join(["pretrained=" + model, "dtype=float16",
135
- "trust_remote_code=False"]),
136
- "--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
137
- "--log_samples"]
138
- if shots is not None:
139
- argv += ["--num_fewshot", str(shots)]
140
- if limit:
141
- argv += ["--limit", str(limit)]
142
- rc, out = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
143
- rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
144
- timeout=gpu_h * 3600)
145
- return rc, " ".join(argv)
146
-
147
-
148
- def harvest(outdir, task):
149
- """Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
150
- glob rather than assume a layout that changes between releases."""
151
- hits = []
152
- for root, _d, files in os.walk(outdir):
153
- for fn in files:
154
- if fn.startswith("results") and fn.endswith(".json"):
155
- try:
156
- j = json.load(open(os.path.join(root, fn)))
157
- except Exception:
158
- continue
159
- if task in (j.get("results") or {}):
160
- hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
161
- if not hits:
162
- return {}
163
- hits.sort()
164
- path, j = hits[-1][1], hits[-1][2]
165
- r = j["results"][task]
166
- keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
167
- return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
168
- "config_fewshot": (j.get("config") or {}).get("num_fewshot"),
169
- "model_repr": (j.get("config") or {}).get("model", [None])[0]
170
- if isinstance((j.get("config") or {}).get("model"), list) else None,
171
- "date": j.get("date"), "git_hash": j.get("git_hash")}
172
-
173
-
174
- def main():
175
- ap = argparse.ArgumentParser()
176
- ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
177
- ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
178
- ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
179
- ap.add_argument("--full-gpu-hours", type=float, default=4.0,
180
- help="per-task ceiling for the full table; MMLU is by far the longest")
181
- a = ap.parse_args()
182
- if not a.model:
183
- raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")
184
-
185
- import ounce100m_credentials
186
- print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
187
- ver = install_harness()
188
- os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars
189
- root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
190
- shutil.rmtree(root, ignore_errors=True)
191
- os.makedirs(root, exist_ok=True)
192
-
193
- # Confirm the task ids resolve before any GPU time goes into them: one typo among nine would otherwise
194
- # fail the all-or-nothing smoke gate and yield no table at all (C5; §2 says to confirm with --tasks list).
195
- rc, out = sh([sys.executable, "-m", "lm_eval", "--tasks", "list"], "list_tasks", 900)
196
- listed = set(re.findall(r"^[ \t]+([a-z][a-z0-9_]+)[ \t]*$", out, re.M))
197
- unknown = [t for t in TASKS if listed and t not in listed]
198
- if unknown:
199
- raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
200
- print("task ids checked against the harness:", len(TASKS), "requested,",
201
- (str(len(listed)) + " ids listed") if listed else "list output unparsed", flush=True)
202
-
203
- smoke = {}
204
- for t in TASKS:
205
- rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
206
- smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
207
- "cmd": cmd}
208
- print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
209
- bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]]
210
- print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
211
- if bad or a.smoke_only:
212
- json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
213
- raise SystemExit(3 if bad else 0)
214
-
215
- table, push_fail = {}, []
216
- for t in TASKS:
217
- rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
218
- for shots, key in ((None, "primary"), (5, "secondary_5shot")):
219
- d = os.path.join(root, key, t)
220
- rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
221
- h = harvest(d, t)
222
- rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
223
- "harness_fewshot": h.get("config_fewshot"), "results_file": h.get("file")}
224
- print("DONE %s %s rc %s rows %s" % (t, key, rc,
225
- json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
226
- table[t] = rec
227
- json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
228
- if not push(a, root): # after every task: an interruption keeps what finished
229
- push_fail.append("after " + t)
230
- if not push(a, root):
231
- push_fail.append("final")
232
- def ok(t, key):
233
- r = table.get(t, {}).get(key, {})
234
- return bool(r.get("rows")) and r.get("rc") == 0
235
-
236
- # Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
237
- # empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
238
- missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
239
- missing += push_fail
240
- # Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
241
- # are separate task ids in the harness (docs/05-eval-plan.md §2).
242
- print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
243
- 2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
244
- raise SystemExit(4 if missing else 0)
245
-
246
-
247
- def push(a, root):
248
- """Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
249
- repo = a.repo or a.model
250
- if not repo:
251
- print("(no --repo and no --model-derived target; results stay local)", flush=True)
252
- return False
253
- import ounce100m_credentials
254
- ounce100m_credentials.install()
255
- tok = os.environ.get("HF_TOKEN")
256
- if tok is None:
257
- print("FATAL: no token, refusing to claim results were published", flush=True)
258
- return False
259
- try:
260
- from huggingface_hub import HfApi
261
- rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
262
- path_in_repo="eval", commit_message="Phase 6 results",
263
- # The --limit 5 smoke rows must not sit in the public repo beside
264
- # the real table looking like results (C3).
265
- ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
266
- commit_description="lm-eval 0.4.13, task-default shots plus a "
267
- "uniform 5-shot column, per docs/05-eval-plan.md")
268
- print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
269
- return True
270
- except Exception as e:
271
- print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
272
- return False
273
-
274
-
275
- if __name__ == "__main__":
276
- main()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
2
+
3
+ Written before any score exists, and it refuses to deviate from the pin:
4
+
5
+ * `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
6
+ template -- a base model with a chat template would be an unearned capability (§3.9).
7
+ * PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
8
+ The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
9
+ be tuned per task later either.
10
+ * SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
11
+ * One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
12
+ moment it finishes so an interruption never loses a completed task (§5 Phase 6).
13
+ * A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
14
+ a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
15
+ is not acceptable. Nothing is published if the smoke pass fails.
16
+
17
+ No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
18
+ the token never enters a log line or a published artifact.
19
+ """
20
+
21
+ import argparse
22
+ import json
23
+ import os
24
+ import re
25
+ import shutil
26
+ import signal
27
+ import subprocess
28
+ import sys
29
+ import time
30
+
31
+ PINNED_LM_EVAL = "0.4.13"
32
+ # Every row here is the harness's own task id, not a name invented for this script.
33
+ TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
34
+ "truthfulqa_mc2", "winogrande", "gsm8k"]
35
+ # Chance level of the *metric being reported*, from the number of answer options in the task, not from a
36
+ # remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
37
+ CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
38
+ "arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
39
+ "hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
40
+ "piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
41
+ "truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
42
+ "truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
43
+ "gsm8k": ("0.00", "free-form exact match")}
44
+
45
+
46
+ # The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it
47
+ # here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores
48
+ # exist, which is the whole reason the plan was written down first (review E-045/10).
49
+ PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none",
50
+ "mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none",
51
+ "truthfulqa_mc2": "acc,none", "winogrande": "acc,none",
52
+ "gsm8k": "exact_match,flexible-extract"}
53
+ # The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes
54
+ # tokenizer and generate behaviour, and every number in this table would then describe a different
55
+ # environment than the one the card claims (E-045/12).
56
+ EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0")
57
+
58
+
59
+ def sh(argv, label, timeout, env=None):
60
+ """Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
61
+ report nothing at all if the platform took the instance."""
62
+ print("=== " + label, flush=True)
63
+ t0 = time.time()
64
+ e = dict(os.environ)
65
+ e["PYTHONUNBUFFERED"] = "1"
66
+ e.update(env or {})
67
+ import threading
68
+ # start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
69
+ # notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
70
+ p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
71
+ env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
72
+ start_new_session=True)
73
+ killed = []
74
+
75
+ def _kill():
76
+ killed.append(True)
77
+ try:
78
+ os.killpg(os.getpgid(p.pid), signal.SIGTERM)
79
+ except Exception:
80
+ p.kill()
81
+
82
+ timer = threading.Timer(timeout, _kill)
83
+ timer.daemon = True
84
+ timer.start()
85
+ keep, lines = [], []
86
+ for line in p.stdout:
87
+ line = line.rstrip("\n")
88
+ lines.append(line)
89
+ keep.append(line)
90
+ del keep[:-60] # only the printed tail is bounded; the returned text is the whole run
91
+ print(" |", line[:260], flush=True)
92
+ timer.cancel()
93
+ rc = p.wait()
94
+ print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
95
+ flush=True)
96
+ return rc, "\n".join(lines)
97
+
98
+
99
+ def tf_version():
100
+ """Which transformers the harness will actually import, and from where -- see the --no-deps note."""
101
+ rc, out = sh([sys.executable, "-c", "import transformers, os; "
102
+ "print(transformers.__version__, os.path.dirname(transformers.__file__))"],
103
+ "probe_transformers", 180)
104
+ for line in out.splitlines():
105
+ m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line)
106
+ if m:
107
+ return m.group(1), m.group(2)
108
+ return "?", "?"
109
+
110
+
111
+ def check_transformers(when):
112
+ """Assert the import the harness will get -- on every path, installing or not installing."""
113
+ v, where = tf_version()
114
+ print("transformers", v, "from", where, "at", when, flush=True)
115
+ if v != EXPECTED_TRANSFORMERS:
116
+ raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval "
117
+ "would describe an environment the model was not verified in"
118
+ % (v, where, EXPECTED_TRANSFORMERS))
119
+ if "/.local/" in where:
120
+ raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy"
121
+ % where)
122
+ return v
123
+
124
+
125
+ def install_harness():
126
+ """Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
127
+ transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
128
+ resolve a different one."""
129
+ rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
130
+ have = ""
131
+ for line in out.splitlines():
132
+ if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
133
+ have = line.strip().strip("|").strip()
134
+ if have == PINNED_LM_EVAL:
135
+ print("lm_eval", have, "already present", flush=True)
136
+ return have # main() still calls check_transformers(): an early return used to skip it
137
+ print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
138
+ flush=True)
139
+ before = check_transformers("before install")
140
+ # --no-deps deliberately. A dependency-resolving install can drop a different transformers into
141
+ # ~/.local, where it shadows the image's copy on sys.path -- the exact thing this file's header says it
142
+ # avoids, and it would move the harness pin silently (review finding C4).
143
+ rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet", "--no-deps",
144
+ "lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
145
+ if rc != 0:
146
+ raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
147
+ after = tf_version()[0]
148
+ if before != after:
149
+ raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
150
+ "on the stack Gate 3 measured" % (before, after))
151
+ rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
152
+ "print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
153
+ "verify_lm_eval", 300)
154
+ got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
155
+ if not got:
156
+ raise SystemExit("lm_eval is not importable at the pinned version after install: "
157
+ + out[-400:])
158
+ return PINNED_LM_EVAL
159
+
160
+
161
+ def run_task(model, task, shots, outdir, limit, gpu_h):
162
+ """One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
163
+ argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
164
+ "--model_args", ",".join(["pretrained=" + model, "dtype=float16",
165
+ "trust_remote_code=False"]),
166
+ "--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
167
+ "--log_samples"]
168
+ if shots is not None:
169
+ argv += ["--num_fewshot", str(shots)]
170
+ if limit:
171
+ argv += ["--limit", str(limit)]
172
+ grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
173
+ if grc != 0:
174
+ # The return code used to be discarded. With no card visible the harness falls back to CPU, and
175
+ # "still running" then reads like "still slow" for weeks (E-045, minor).
176
+ raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s"
177
+ % (grc, task, gout[-200:]))
178
+ rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
179
+ timeout=gpu_h * 3600)
180
+ return rc, " ".join(argv)
181
+
182
+
183
+ def harvest(outdir, task):
184
+ """Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
185
+ glob rather than assume a layout that changes between releases."""
186
+ hits = []
187
+ for root, _d, files in os.walk(outdir):
188
+ for fn in files:
189
+ if fn.startswith("results") and fn.endswith(".json"):
190
+ try:
191
+ j = json.load(open(os.path.join(root, fn)))
192
+ except Exception:
193
+ continue
194
+ if task in (j.get("results") or {}):
195
+ hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
196
+ if not hits:
197
+ return {}
198
+ hits.sort()
199
+ path, j = hits[-1][1], hits[-1][2]
200
+ r = j["results"][task]
201
+ keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
202
+ cfg = j.get("config") or {}
203
+ # `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY
204
+ # column -- the one the report quotes -- with no shot count at all. What was actually used is in
205
+ # `configs[task]`, and `n_input` is the denominator beside the score (E-045/8).
206
+ per = (j.get("configs") or {}).get(task) or {}
207
+ shots = per.get("num_fewshot")
208
+ if shots is None:
209
+ shots = cfg.get("num_fewshot")
210
+
211
+ def one(x):
212
+ return x[0] if isinstance(x, list) and x else x
213
+
214
+ # `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the
215
+ # denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry
216
+ # n_input, and a task that silently resolved to a subset otherwise looks exactly like a score.
217
+ n_logged = None
218
+ for fn in sorted(os.listdir(os.path.dirname(path))):
219
+ if fn.startswith("samples") and fn.endswith(".json"):
220
+ try:
221
+ sj = json.load(open(os.path.join(os.path.dirname(path), fn)))
222
+ except Exception:
223
+ continue
224
+ if isinstance(sj.get("samples"), (list, dict)):
225
+ n_logged = len(sj["samples"])
226
+ break
227
+
228
+ return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
229
+ "shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"),
230
+ "n_logged": n_logged,
231
+ "limit": cfg.get("limit"), "model": one(cfg.get("model")),
232
+ "model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None,
233
+ "dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"),
234
+ "split": per.get("split") or (per.get("test_args") or {}).get("split"),
235
+ "seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"),
236
+ "config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"),
237
+ "git_hash": j.get("git_hash"),
238
+ "lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None}
239
+
240
+
241
+ def main():
242
+ ap = argparse.ArgumentParser()
243
+ ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
244
+ ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
245
+ ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
246
+ ap.add_argument("--full-gpu-hours", type=float, default=4.0,
247
+ help="per-task ceiling for the full table; MMLU is by far the longest")
248
+ a = ap.parse_args()
249
+ if not a.model:
250
+ raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")
251
+
252
+ import ounce100m_credentials
253
+ print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
254
+ ver = install_harness()
255
+ tf = check_transformers("after harness resolved")
256
+ os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars
257
+ root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
258
+ shutil.rmtree(root, ignore_errors=True)
259
+ os.makedirs(root, exist_ok=True)
260
+
261
+ # Confirm the task ids resolve before any GPU time goes into them: one typo among nine would otherwise
262
+ # fail the all-or-nothing smoke gate and yield no table at all (C5; §2 says to confirm with --tasks list).
263
+ rc, out = sh([sys.executable, "-m", "lm_eval", "--tasks", "list"], "list_tasks", 900)
264
+ # 0.4.13 prints " - <id>" bullets; the old pattern required the id to end the line and matched
265
+ # nothing, then `listed and ...` voided the whole check, and the script went ahead and spent GPU hours
266
+ # on ids the harness may not know. An unparseable listing is now a hard failure, not a shrug (E-045/6).
267
+ listed = set(re.findall(r"^\s*(?:-\s*)?([a-z][a-z0-9_]{2,})\s*$", out, re.M))
268
+ if len(listed) < 50:
269
+ raise SystemExit("could not read the harness task list (%d ids parsed); refusing to run %s "
270
+ "unverified -- %r" % (len(listed), PINNED_LM_EVAL, out[-500:]))
271
+ unknown = [t for t in TASKS if t not in listed]
272
+ if unknown:
273
+ raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
274
+ print("task ids checked against the harness:", len(TASKS), "requested,", len(listed), "ids listed",
275
+ flush=True)
276
+
277
+ smoke = {}
278
+ for t in TASKS:
279
+ rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
280
+ smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
281
+ "cmd": cmd}
282
+ print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
283
+ bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]]
284
+ print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
285
+ if bad or a.smoke_only:
286
+ json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
287
+ raise SystemExit(3 if bad else 0)
288
+
289
+ table, push_fail = {}, []
290
+ for t in TASKS:
291
+ rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
292
+ for shots, key in ((None, "primary"), (5, "secondary_5shot")):
293
+ d = os.path.join(root, key, t)
294
+ rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
295
+ h = harvest(d, t)
296
+ rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
297
+ "shots": h.get("shots"), "harness_fewshot": h.get("config_fewshot"),
298
+ "n_input": h.get("n_input"), "n_effective": h.get("n_effective"),
299
+ "n_logged": h.get("n_logged"),
300
+ "limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"),
301
+ "seed": h.get("seed"), "git_hash": h.get("git_hash"),
302
+ "dataset_path": h.get("dataset_path"), "split": h.get("split"),
303
+ "results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t],
304
+ "primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])}
305
+ if rc != 0 or not h.get("rows"):
306
+ # A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON.
307
+ rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {}))
308
+ elif h.get("limit") not in (None, 0):
309
+ raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score"
310
+ % (t, key, h.get("limit")))
311
+ elif not (h.get("n_input") or h.get("n_logged")):
312
+ # No denominator beside a score is no denominator at all: say so in the record.
313
+ rec[key]["status"] = "NO ROW COUNT RECORDED"
314
+ elif key == "primary" and h.get("shots") is None:
315
+ raise SystemExit("%s: the harness recorded no shot count, so the table row would not be "
316
+ "reproducible" % t)
317
+ print("DONE %s %s rc %s rows %s" % (t, key, rc,
318
+ json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
319
+ table[t] = rec
320
+ json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
321
+ if not push(a, root): # after every task: an interruption keeps what finished
322
+ push_fail.append("after " + t)
323
+ if not push(a, root):
324
+ push_fail.append("final")
325
+ def ok(t, key):
326
+ r = table.get(t, {}).get(key, {})
327
+ return bool(r.get("rows")) and r.get("rc") == 0
328
+
329
+ # Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
330
+ # empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
331
+ missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
332
+ # The two columns run the same task on the same split with no limit, so a differing denominator means
333
+ # one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9).
334
+ for t in TASKS:
335
+ p1 = table.get(t, {}).get("primary", {})
336
+ p2 = table.get(t, {}).get("secondary_5shot", {})
337
+ n1 = p1.get("n_input") or p1.get("n_logged")
338
+ n2 = p2.get("n_input") or p2.get("n_logged")
339
+ if n1 and n2 and n1 != n2:
340
+ missing.append("%s rows %s != %s between columns" % (t, n1, n2))
341
+ missing += push_fail
342
+ # Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
343
+ # are separate task ids in the harness (docs/05-eval-plan.md §2).
344
+ print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
345
+ 2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
346
+ raise SystemExit(4 if missing else 0)
347
+
348
+
349
+ def push(a, root):
350
+ """Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
351
+ repo = a.repo or a.model
352
+ if not repo:
353
+ print("(no --repo and no --model-derived target; results stay local)", flush=True)
354
+ return False
355
+ import ounce100m_credentials
356
+ ounce100m_credentials.install()
357
+ tok = os.environ.get("HF_TOKEN")
358
+ if tok is None:
359
+ print("FATAL: no token, refusing to claim results were published", flush=True)
360
+ return False
361
+ try:
362
+ from huggingface_hub import HfApi
363
+ rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
364
+ path_in_repo="eval", commit_message="Phase 6 results",
365
+ # The --limit 5 smoke rows must not sit in the public repo beside
366
+ # the real table looking like results (C3).
367
+ ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
368
+ commit_description="lm-eval 0.4.13, task-default shots plus a "
369
+ "uniform 5-shot column, per docs/05-eval-plan.md")
370
+ print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
371
+ # E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the
372
+ # artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes.
373
+ try:
374
+ import hashlib as _h
375
+ import urllib.request as _u
376
+ want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest()
377
+ with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d"
378
+ % (repo, int(time.time())), timeout=300) as fh:
379
+ got = _h.sha256(fh.read()).hexdigest()
380
+ print("EVAL_READBACK sha_ok", got == want, flush=True)
381
+ return got == want
382
+ except Exception as e2:
383
+ print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True)
384
+ return False
385
+ except Exception as e:
386
+ print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
387
+ return False
388
+
389
+
390
+ if __name__ == "__main__":
391
+ main()