run_benchmarks: task-list hard fail, per-task shot/denominator provenance, primary metric pre-registered, failed cells not zeros, push verified anonymously (E-045)
Browse files- eval/run_benchmarks.py +391 -276
eval/run_benchmarks.py
CHANGED
|
@@ -1,276 +1,391 @@
|
|
| 1 |
-
"""Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
|
| 2 |
-
|
| 3 |
-
Written before any score exists, and it refuses to deviate from the pin:
|
| 4 |
-
|
| 5 |
-
* `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
|
| 6 |
-
template -- a base model with a chat template would be an unearned capability (§3.9).
|
| 7 |
-
* PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
|
| 8 |
-
The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
|
| 9 |
-
be tuned per task later either.
|
| 10 |
-
* SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
|
| 11 |
-
* One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
|
| 12 |
-
moment it finishes so an interruption never loses a completed task (§5 Phase 6).
|
| 13 |
-
* A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
|
| 14 |
-
a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
|
| 15 |
-
is not acceptable. Nothing is published if the smoke pass fails.
|
| 16 |
-
|
| 17 |
-
No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
|
| 18 |
-
the token never enters a log line or a published artifact.
|
| 19 |
-
"""
|
| 20 |
-
|
| 21 |
-
import argparse
|
| 22 |
-
import json
|
| 23 |
-
import os
|
| 24 |
-
import re
|
| 25 |
-
import shutil
|
| 26 |
-
import signal
|
| 27 |
-
import subprocess
|
| 28 |
-
import sys
|
| 29 |
-
import time
|
| 30 |
-
|
| 31 |
-
PINNED_LM_EVAL = "0.4.13"
|
| 32 |
-
# Every row here is the harness's own task id, not a name invented for this script.
|
| 33 |
-
TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
|
| 34 |
-
"truthfulqa_mc2", "winogrande", "gsm8k"]
|
| 35 |
-
# Chance level of the *metric being reported*, from the number of answer options in the task, not from a
|
| 36 |
-
# remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
|
| 37 |
-
CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
|
| 38 |
-
"arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
|
| 39 |
-
"hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
|
| 40 |
-
"piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
|
| 41 |
-
"truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
|
| 42 |
-
"truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
|
| 43 |
-
"gsm8k": ("0.00", "free-form exact match")}
|
| 44 |
-
|
| 45 |
-
|
| 46 |
-
|
| 47 |
-
|
| 48 |
-
|
| 49 |
-
|
| 50 |
-
|
| 51 |
-
|
| 52 |
-
|
| 53 |
-
|
| 54 |
-
|
| 55 |
-
|
| 56 |
-
|
| 57 |
-
|
| 58 |
-
|
| 59 |
-
|
| 60 |
-
|
| 61 |
-
|
| 62 |
-
|
| 63 |
-
|
| 64 |
-
|
| 65 |
-
|
| 66 |
-
|
| 67 |
-
|
| 68 |
-
|
| 69 |
-
|
| 70 |
-
|
| 71 |
-
|
| 72 |
-
|
| 73 |
-
|
| 74 |
-
|
| 75 |
-
|
| 76 |
-
|
| 77 |
-
|
| 78 |
-
|
| 79 |
-
|
| 80 |
-
|
| 81 |
-
|
| 82 |
-
|
| 83 |
-
|
| 84 |
-
|
| 85 |
-
|
| 86 |
-
|
| 87 |
-
|
| 88 |
-
|
| 89 |
-
|
| 90 |
-
|
| 91 |
-
|
| 92 |
-
|
| 93 |
-
|
| 94 |
-
|
| 95 |
-
|
| 96 |
-
|
| 97 |
-
|
| 98 |
-
|
| 99 |
-
|
| 100 |
-
|
| 101 |
-
|
| 102 |
-
|
| 103 |
-
|
| 104 |
-
|
| 105 |
-
|
| 106 |
-
|
| 107 |
-
|
| 108 |
-
|
| 109 |
-
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
|
| 113 |
-
|
| 114 |
-
|
| 115 |
-
if
|
| 116 |
-
raise SystemExit("
|
| 117 |
-
|
| 118 |
-
|
| 119 |
-
|
| 120 |
-
|
| 121 |
-
|
| 122 |
-
|
| 123 |
-
|
| 124 |
-
|
| 125 |
-
|
| 126 |
-
|
| 127 |
-
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
|
| 133 |
-
|
| 134 |
-
|
| 135 |
-
|
| 136 |
-
|
| 137 |
-
|
| 138 |
-
|
| 139 |
-
|
| 140 |
-
|
| 141 |
-
|
| 142 |
-
|
| 143 |
-
rc,
|
| 144 |
-
|
| 145 |
-
|
| 146 |
-
|
| 147 |
-
|
| 148 |
-
|
| 149 |
-
|
| 150 |
-
|
| 151 |
-
|
| 152 |
-
|
| 153 |
-
|
| 154 |
-
|
| 155 |
-
|
| 156 |
-
|
| 157 |
-
|
| 158 |
-
|
| 159 |
-
|
| 160 |
-
|
| 161 |
-
|
| 162 |
-
|
| 163 |
-
|
| 164 |
-
|
| 165 |
-
|
| 166 |
-
|
| 167 |
-
|
| 168 |
-
|
| 169 |
-
|
| 170 |
-
|
| 171 |
-
|
| 172 |
-
|
| 173 |
-
|
| 174 |
-
|
| 175 |
-
|
| 176 |
-
|
| 177 |
-
|
| 178 |
-
|
| 179 |
-
|
| 180 |
-
|
| 181 |
-
|
| 182 |
-
|
| 183 |
-
|
| 184 |
-
|
| 185 |
-
|
| 186 |
-
|
| 187 |
-
|
| 188 |
-
|
| 189 |
-
|
| 190 |
-
|
| 191 |
-
|
| 192 |
-
|
| 193 |
-
|
| 194 |
-
|
| 195 |
-
|
| 196 |
-
|
| 197 |
-
|
| 198 |
-
|
| 199 |
-
|
| 200 |
-
|
| 201 |
-
|
| 202 |
-
|
| 203 |
-
|
| 204 |
-
|
| 205 |
-
|
| 206 |
-
|
| 207 |
-
|
| 208 |
-
|
| 209 |
-
|
| 210 |
-
|
| 211 |
-
|
| 212 |
-
|
| 213 |
-
|
| 214 |
-
|
| 215 |
-
|
| 216 |
-
|
| 217 |
-
|
| 218 |
-
|
| 219 |
-
|
| 220 |
-
|
| 221 |
-
|
| 222 |
-
|
| 223 |
-
|
| 224 |
-
|
| 225 |
-
|
| 226 |
-
|
| 227 |
-
|
| 228 |
-
|
| 229 |
-
|
| 230 |
-
|
| 231 |
-
|
| 232 |
-
|
| 233 |
-
|
| 234 |
-
|
| 235 |
-
|
| 236 |
-
|
| 237 |
-
|
| 238 |
-
|
| 239 |
-
|
| 240 |
-
|
| 241 |
-
|
| 242 |
-
|
| 243 |
-
|
| 244 |
-
|
| 245 |
-
|
| 246 |
-
|
| 247 |
-
|
| 248 |
-
|
| 249 |
-
|
| 250 |
-
|
| 251 |
-
|
| 252 |
-
|
| 253 |
-
|
| 254 |
-
|
| 255 |
-
|
| 256 |
-
|
| 257 |
-
|
| 258 |
-
|
| 259 |
-
|
| 260 |
-
|
| 261 |
-
|
| 262 |
-
|
| 263 |
-
|
| 264 |
-
|
| 265 |
-
|
| 266 |
-
|
| 267 |
-
|
| 268 |
-
|
| 269 |
-
|
| 270 |
-
|
| 271 |
-
|
| 272 |
-
|
| 273 |
-
|
| 274 |
-
|
| 275 |
-
|
| 276 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Phase 6: the eight academic benchmarks, run exactly as docs/05-eval-plan.md froze them.
|
| 2 |
+
|
| 3 |
+
Written before any score exists, and it refuses to deviate from the pin:
|
| 4 |
+
|
| 5 |
+
* `lm-eval` **0.4.13**, backend `model=hf`, `dtype=float16`, one T4, no `--trust_remote_code`, no chat
|
| 6 |
+
template -- a base model with a chat template would be an unearned capability (§3.9).
|
| 7 |
+
* PRIMARY column = each task's own default `num_fewshot`, which means **no `--num_fewshot` flag at all**.
|
| 8 |
+
The point of using the harness's file rather than a number is that nobody chose it per task, so it cannot
|
| 9 |
+
be tuned per task later either.
|
| 10 |
+
* SECONDARY column = a uniform 5-shot run of every task, reported alongside, never instead.
|
| 11 |
+
* One invocation per task, all eight in a single job, and each task's result file is pushed to the Hub the
|
| 12 |
+
moment it finishes so an interruption never loses a completed task (§5 Phase 6).
|
| 13 |
+
* A `--limit 5` smoke pass over all eight runs first, because transformers 5.0.0 on the Kaggle image versus
|
| 14 |
+
a harness pinned in 2024 is an unverified combination (§5) and finding that out after 6 hours of GPU time
|
| 15 |
+
is not acceptable. Nothing is published if the smoke pass fails.
|
| 16 |
+
|
| 17 |
+
No credentials in this file: `ounce100m_credentials.install()` fetches the token at run time (D-006), and
|
| 18 |
+
the token never enters a log line or a published artifact.
|
| 19 |
+
"""
|
| 20 |
+
|
| 21 |
+
import argparse
|
| 22 |
+
import json
|
| 23 |
+
import os
|
| 24 |
+
import re
|
| 25 |
+
import shutil
|
| 26 |
+
import signal
|
| 27 |
+
import subprocess
|
| 28 |
+
import sys
|
| 29 |
+
import time
|
| 30 |
+
|
| 31 |
+
PINNED_LM_EVAL = "0.4.13"
|
| 32 |
+
# Every row here is the harness's own task id, not a name invented for this script.
|
| 33 |
+
TASKS = ["arc_challenge", "arc_easy", "hellaswag", "mmlu", "piqa", "truthfulqa_mc1",
|
| 34 |
+
"truthfulqa_mc2", "winogrande", "gsm8k"]
|
| 35 |
+
# Chance level of the *metric being reported*, from the number of answer options in the task, not from a
|
| 36 |
+
# remembered leaderboard. Where a metric is not chance-normalised that is said rather than guessed at.
|
| 37 |
+
CHANCE = {"arc_challenge": ("0.25-0.33", "items mix 3 and 4 options"),
|
| 38 |
+
"arc_easy": ("0.25-0.33", "items mix 3 and 4 options"),
|
| 39 |
+
"hellaswag": ("0.25", "4 continuations"), "mmlu": ("0.25", "4 options, macro over 57 subjects"),
|
| 40 |
+
"piqa": ("0.50", "2 options"), "winogrande": ("0.50", "2 options"),
|
| 41 |
+
"truthfulqa_mc1": ("~0.20", "mean over questions of 1/#choices"),
|
| 42 |
+
"truthfulqa_mc2": ("n/a", "mc2 is not chance-normalised"),
|
| 43 |
+
"gsm8k": ("0.00", "free-form exact match")}
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
# The metric column for each row, copied from the table docs/05-eval-plan.md section 2 froze. Naming it
|
| 47 |
+
# here is what stops "which of GSM8K's two numbers did we publish?" from being decided after the scores
|
| 48 |
+
# exist, which is the whole reason the plan was written down first (review E-045/10).
|
| 49 |
+
PRIMARY_METRIC = {"arc_challenge": "acc,none", "arc_easy": "acc,none", "hellaswag": "acc,none",
|
| 50 |
+
"mmlu": "acc,none", "piqa": "acc,none", "truthfulqa_mc1": "acc,none",
|
| 51 |
+
"truthfulqa_mc2": "acc,none", "winogrande": "acc,none",
|
| 52 |
+
"gsm8k": "exact_match,flexible-extract"}
|
| 53 |
+
# The stack Gate 3 measured and Gate 5 loaded against. A silently shadowed transformers in ~/.local changes
|
| 54 |
+
# tokenizer and generate behaviour, and every number in this table would then describe a different
|
| 55 |
+
# environment than the one the card claims (E-045/12).
|
| 56 |
+
EXPECTED_TRANSFORMERS = os.environ.get("EXPECTED_TRANSFORMERS", "5.0.0")
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def sh(argv, label, timeout, env=None):
|
| 60 |
+
"""Streamed like every other long-running stage in this project -- a buffered 6-hour eval job would
|
| 61 |
+
report nothing at all if the platform took the instance."""
|
| 62 |
+
print("=== " + label, flush=True)
|
| 63 |
+
t0 = time.time()
|
| 64 |
+
e = dict(os.environ)
|
| 65 |
+
e["PYTHONUNBUFFERED"] = "1"
|
| 66 |
+
e.update(env or {})
|
| 67 |
+
import threading
|
| 68 |
+
# start_new_session, because _kill() signals getpgid(pid). Without its own group that group is the
|
| 69 |
+
# notebook's, so a timeout on the longest task would SIGTERM the very script watching for it (C1).
|
| 70 |
+
p = subprocess.Popen(argv, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True,
|
| 71 |
+
env=e, bufsize=1, cwd=os.path.dirname(os.path.abspath(__file__)),
|
| 72 |
+
start_new_session=True)
|
| 73 |
+
killed = []
|
| 74 |
+
|
| 75 |
+
def _kill():
|
| 76 |
+
killed.append(True)
|
| 77 |
+
try:
|
| 78 |
+
os.killpg(os.getpgid(p.pid), signal.SIGTERM)
|
| 79 |
+
except Exception:
|
| 80 |
+
p.kill()
|
| 81 |
+
|
| 82 |
+
timer = threading.Timer(timeout, _kill)
|
| 83 |
+
timer.daemon = True
|
| 84 |
+
timer.start()
|
| 85 |
+
keep, lines = [], []
|
| 86 |
+
for line in p.stdout:
|
| 87 |
+
line = line.rstrip("\n")
|
| 88 |
+
lines.append(line)
|
| 89 |
+
keep.append(line)
|
| 90 |
+
del keep[:-60] # only the printed tail is bounded; the returned text is the whole run
|
| 91 |
+
print(" |", line[:260], flush=True)
|
| 92 |
+
timer.cancel()
|
| 93 |
+
rc = p.wait()
|
| 94 |
+
print("%s_RC %s%s seconds %.1f" % (label, rc, " TIMEOUT" if killed else "", time.time() - t0),
|
| 95 |
+
flush=True)
|
| 96 |
+
return rc, "\n".join(lines)
|
| 97 |
+
|
| 98 |
+
|
| 99 |
+
def tf_version():
|
| 100 |
+
"""Which transformers the harness will actually import, and from where -- see the --no-deps note."""
|
| 101 |
+
rc, out = sh([sys.executable, "-c", "import transformers, os; "
|
| 102 |
+
"print(transformers.__version__, os.path.dirname(transformers.__file__))"],
|
| 103 |
+
"probe_transformers", 180)
|
| 104 |
+
for line in out.splitlines():
|
| 105 |
+
m = re.match(r"^\s*\|?\s*(\d[\w.]*)\s+(/\S+)", line)
|
| 106 |
+
if m:
|
| 107 |
+
return m.group(1), m.group(2)
|
| 108 |
+
return "?", "?"
|
| 109 |
+
|
| 110 |
+
|
| 111 |
+
def check_transformers(when):
|
| 112 |
+
"""Assert the import the harness will get -- on every path, installing or not installing."""
|
| 113 |
+
v, where = tf_version()
|
| 114 |
+
print("transformers", v, "from", where, "at", when, flush=True)
|
| 115 |
+
if v != EXPECTED_TRANSFORMERS:
|
| 116 |
+
raise SystemExit("transformers is %s (%s), not the %s Gate 3 measured and Gate 5 loaded: the eval "
|
| 117 |
+
"would describe an environment the model was not verified in"
|
| 118 |
+
% (v, where, EXPECTED_TRANSFORMERS))
|
| 119 |
+
if "/.local/" in where:
|
| 120 |
+
raise SystemExit("transformers is being imported from a user dir (%s), which shadows the image copy"
|
| 121 |
+
% where)
|
| 122 |
+
return v
|
| 123 |
+
|
| 124 |
+
|
| 125 |
+
def install_harness():
|
| 126 |
+
"""Pin the harness in a uv inline-script environment rather than into the Kaggle image's site-packages:
|
| 127 |
+
transformers 5.0.0 must stay exactly as Gate 3 measured it, and `pip install` into the image would
|
| 128 |
+
resolve a different one."""
|
| 129 |
+
rc, out = sh([sys.executable, "-c", "import lm_eval; print(lm_eval.__version__)"], "probe_lm_eval", 120)
|
| 130 |
+
have = ""
|
| 131 |
+
for line in out.splitlines():
|
| 132 |
+
if re.match(r"^\s*\|?\s*[0-9]+\.[0-9]+", line):
|
| 133 |
+
have = line.strip().strip("|").strip()
|
| 134 |
+
if have == PINNED_LM_EVAL:
|
| 135 |
+
print("lm_eval", have, "already present", flush=True)
|
| 136 |
+
return have # main() still calls check_transformers(): an early return used to skip it
|
| 137 |
+
print("lm_eval %r is not the pin %s -- installing it as a user package" % (have, PINNED_LM_EVAL),
|
| 138 |
+
flush=True)
|
| 139 |
+
before = check_transformers("before install")
|
| 140 |
+
# --no-deps deliberately. A dependency-resolving install can drop a different transformers into
|
| 141 |
+
# ~/.local, where it shadows the image's copy on sys.path -- the exact thing this file's header says it
|
| 142 |
+
# avoids, and it would move the harness pin silently (review finding C4).
|
| 143 |
+
rc, _ = sh([sys.executable, "-m", "pip", "install", "--user", "--quiet", "--no-deps",
|
| 144 |
+
"lm-eval==%s" % PINNED_LM_EVAL], "pip_lm_eval", 1800)
|
| 145 |
+
if rc != 0:
|
| 146 |
+
raise SystemExit("could not install the pinned harness; refusing to eval on a different one")
|
| 147 |
+
after = tf_version()[0]
|
| 148 |
+
if before != after:
|
| 149 |
+
raise SystemExit("installing the harness moved transformers %s -> %s: the eval would no longer run "
|
| 150 |
+
"on the stack Gate 3 measured" % (before, after))
|
| 151 |
+
rc, out = sh([sys.executable, "-c", "import lm_eval, os; "
|
| 152 |
+
"print(lm_eval.__version__, os.path.dirname(lm_eval.__file__))"],
|
| 153 |
+
"verify_lm_eval", 300)
|
| 154 |
+
got = [l.split()[0] for l in out.splitlines() if l.startswith(PINNED_LM_EVAL)]
|
| 155 |
+
if not got:
|
| 156 |
+
raise SystemExit("lm_eval is not importable at the pinned version after install: "
|
| 157 |
+
+ out[-400:])
|
| 158 |
+
return PINNED_LM_EVAL
|
| 159 |
+
|
| 160 |
+
|
| 161 |
+
def run_task(model, task, shots, outdir, limit, gpu_h):
|
| 162 |
+
"""One harness invocation. `shots=None` means do not pass the flag at all, which is the PRIMARY column."""
|
| 163 |
+
argv = [sys.executable, "-m", "lm_eval", "--model", "hf",
|
| 164 |
+
"--model_args", ",".join(["pretrained=" + model, "dtype=float16",
|
| 165 |
+
"trust_remote_code=False"]),
|
| 166 |
+
"--tasks", task, "--batch_size", "8", "--seed", "42", "--output_path", outdir,
|
| 167 |
+
"--log_samples"]
|
| 168 |
+
if shots is not None:
|
| 169 |
+
argv += ["--num_fewshot", str(shots)]
|
| 170 |
+
if limit:
|
| 171 |
+
argv += ["--limit", str(limit)]
|
| 172 |
+
grc, gout = sh(["nvidia-smi", "--query-gpu=memory.used", "--format=csv,noheader"], "gpu_before", 60)
|
| 173 |
+
if grc != 0:
|
| 174 |
+
# The return code used to be discarded. With no card visible the harness falls back to CPU, and
|
| 175 |
+
# "still running" then reads like "still slow" for weeks (E-045, minor).
|
| 176 |
+
raise SystemExit("nvidia-smi failed (rc %s) before %s: refusing to evaluate on CPU -- %s"
|
| 177 |
+
% (grc, task, gout[-200:]))
|
| 178 |
+
rc, out = sh(argv, "EVAL_%s_%s" % (task, "default" if shots is None else "%dshot" % shots),
|
| 179 |
+
timeout=gpu_h * 3600)
|
| 180 |
+
return rc, " ".join(argv)
|
| 181 |
+
|
| 182 |
+
|
| 183 |
+
def harvest(outdir, task):
|
| 184 |
+
"""Find the rows the harness just wrote. It nests results under <timestamp>/<task>/results.json, so
|
| 185 |
+
glob rather than assume a layout that changes between releases."""
|
| 186 |
+
hits = []
|
| 187 |
+
for root, _d, files in os.walk(outdir):
|
| 188 |
+
for fn in files:
|
| 189 |
+
if fn.startswith("results") and fn.endswith(".json"):
|
| 190 |
+
try:
|
| 191 |
+
j = json.load(open(os.path.join(root, fn)))
|
| 192 |
+
except Exception:
|
| 193 |
+
continue
|
| 194 |
+
if task in (j.get("results") or {}):
|
| 195 |
+
hits.append((os.path.getmtime(os.path.join(root, fn)), os.path.join(root, fn), j))
|
| 196 |
+
if not hits:
|
| 197 |
+
return {}
|
| 198 |
+
hits.sort()
|
| 199 |
+
path, j = hits[-1][1], hits[-1][2]
|
| 200 |
+
r = j["results"][task]
|
| 201 |
+
keep = {k: v for k, v in r.items() if not isinstance(v, (dict, list)) and k != "alias"}
|
| 202 |
+
cfg = j.get("config") or {}
|
| 203 |
+
# `config["num_fewshot"]` is null unless the flag was passed, so reading only it left the PRIMARY
|
| 204 |
+
# column -- the one the report quotes -- with no shot count at all. What was actually used is in
|
| 205 |
+
# `configs[task]`, and `n_input` is the denominator beside the score (E-045/8).
|
| 206 |
+
per = (j.get("configs") or {}).get(task) or {}
|
| 207 |
+
shots = per.get("num_fewshot")
|
| 208 |
+
if shots is None:
|
| 209 |
+
shots = cfg.get("num_fewshot")
|
| 210 |
+
|
| 211 |
+
def one(x):
|
| 212 |
+
return x[0] if isinstance(x, list) and x else x
|
| 213 |
+
|
| 214 |
+
# `--log_samples` writes samples_<task>_<hash>.json beside results.json, and its own row count is the
|
| 215 |
+
# denominator that can be checked after the fact -- 0.4.13's results rows do not reliably carry
|
| 216 |
+
# n_input, and a task that silently resolved to a subset otherwise looks exactly like a score.
|
| 217 |
+
n_logged = None
|
| 218 |
+
for fn in sorted(os.listdir(os.path.dirname(path))):
|
| 219 |
+
if fn.startswith("samples") and fn.endswith(".json"):
|
| 220 |
+
try:
|
| 221 |
+
sj = json.load(open(os.path.join(os.path.dirname(path), fn)))
|
| 222 |
+
except Exception:
|
| 223 |
+
continue
|
| 224 |
+
if isinstance(sj.get("samples"), (list, dict)):
|
| 225 |
+
n_logged = len(sj["samples"])
|
| 226 |
+
break
|
| 227 |
+
|
| 228 |
+
return {"file": path, "rows": keep, "n_task_versions": len(j.get("task_versions") or {}),
|
| 229 |
+
"shots": shots, "n_input": r.get("n_input"), "n_effective": r.get("n_effective"),
|
| 230 |
+
"n_logged": n_logged,
|
| 231 |
+
"limit": cfg.get("limit"), "model": one(cfg.get("model")),
|
| 232 |
+
"model_args": cfg.get("model_args") if isinstance(cfg.get("model_args"), (dict, str)) else None,
|
| 233 |
+
"dataset_path": per.get("dataset_path"), "dataset_name": per.get("dataset_name"),
|
| 234 |
+
"split": per.get("split") or (per.get("test_args") or {}).get("split"),
|
| 235 |
+
"seed": cfg.get("seed"), "dtype": cfg.get("dtype"), "repeats": cfg.get("repeats"),
|
| 236 |
+
"config_fewshot": cfg.get("num_fewshot"), "date": j.get("date"),
|
| 237 |
+
"git_hash": j.get("git_hash"),
|
| 238 |
+
"lm_eval_version": (j.get("lm_eval") or {}).get("version") if isinstance(j.get("lm_eval"), dict) else None}
|
| 239 |
+
|
| 240 |
+
|
| 241 |
+
def main():
|
| 242 |
+
ap = argparse.ArgumentParser()
|
| 243 |
+
ap.add_argument("--model", default="", help="Hub model id, e.g. Cion-lab/ounce106m-v1")
|
| 244 |
+
ap.add_argument("--repo", default="", help="where to push results (defaults to --model)")
|
| 245 |
+
ap.add_argument("--smoke-only", action="store_true", help="the --limit 5 pass and nothing else")
|
| 246 |
+
ap.add_argument("--full-gpu-hours", type=float, default=4.0,
|
| 247 |
+
help="per-task ceiling for the full table; MMLU is by far the longest")
|
| 248 |
+
a = ap.parse_args()
|
| 249 |
+
if not a.model:
|
| 250 |
+
raise SystemExit("--model is required (the public Hub repo, loaded clean-room)")
|
| 251 |
+
|
| 252 |
+
import ounce100m_credentials
|
| 253 |
+
print("creds:", json.dumps(ounce100m_credentials.install(verify=True)), flush=True)
|
| 254 |
+
ver = install_harness()
|
| 255 |
+
tf = check_transformers("after harness resolved")
|
| 256 |
+
os.environ["CUDA_VISIBLE_DEVICES"] = "0" # one card, no gather path that could reorder exemplars
|
| 257 |
+
root = os.path.join(os.path.dirname(os.path.abspath(__file__)), "results")
|
| 258 |
+
shutil.rmtree(root, ignore_errors=True)
|
| 259 |
+
os.makedirs(root, exist_ok=True)
|
| 260 |
+
|
| 261 |
+
# Confirm the task ids resolve before any GPU time goes into them: one typo among nine would otherwise
|
| 262 |
+
# fail the all-or-nothing smoke gate and yield no table at all (C5; §2 says to confirm with --tasks list).
|
| 263 |
+
rc, out = sh([sys.executable, "-m", "lm_eval", "--tasks", "list"], "list_tasks", 900)
|
| 264 |
+
# 0.4.13 prints " - <id>" bullets; the old pattern required the id to end the line and matched
|
| 265 |
+
# nothing, then `listed and ...` voided the whole check, and the script went ahead and spent GPU hours
|
| 266 |
+
# on ids the harness may not know. An unparseable listing is now a hard failure, not a shrug (E-045/6).
|
| 267 |
+
listed = set(re.findall(r"^\s*(?:-\s*)?([a-z][a-z0-9_]{2,})\s*$", out, re.M))
|
| 268 |
+
if len(listed) < 50:
|
| 269 |
+
raise SystemExit("could not read the harness task list (%d ids parsed); refusing to run %s "
|
| 270 |
+
"unverified -- %r" % (len(listed), PINNED_LM_EVAL, out[-500:]))
|
| 271 |
+
unknown = [t for t in TASKS if t not in listed]
|
| 272 |
+
if unknown:
|
| 273 |
+
raise SystemExit("these task ids do not resolve in lm-eval %s: %s" % (ver, unknown))
|
| 274 |
+
print("task ids checked against the harness:", len(TASKS), "requested,", len(listed), "ids listed",
|
| 275 |
+
flush=True)
|
| 276 |
+
|
| 277 |
+
smoke = {}
|
| 278 |
+
for t in TASKS:
|
| 279 |
+
rc, cmd = run_task(a.model, t, None, os.path.join(root, "smoke"), 5, 0.5)
|
| 280 |
+
smoke[t] = {"rc": rc, "rows": harvest(os.path.join(root, "smoke"), t).get("rows", {}),
|
| 281 |
+
"cmd": cmd}
|
| 282 |
+
print("SMOKE %s rc %s rows %d" % (t, rc, len(smoke[t]["rows"])), flush=True)
|
| 283 |
+
bad = [t for t in TASKS if smoke[t]["rc"] != 0 or not smoke[t]["rows"]]
|
| 284 |
+
print("SMOKE_PASS" if not bad else "SMOKE_FAIL " + str(bad), flush=True)
|
| 285 |
+
if bad or a.smoke_only:
|
| 286 |
+
json.dump(smoke, open(os.path.join(root, "smoke.json"), "w"), indent=1, default=str)
|
| 287 |
+
raise SystemExit(3 if bad else 0)
|
| 288 |
+
|
| 289 |
+
table, push_fail = {}, []
|
| 290 |
+
for t in TASKS:
|
| 291 |
+
rec = {"chance": CHANCE.get(t, ("?", "?")), "lm_eval": ver}
|
| 292 |
+
for shots, key in ((None, "primary"), (5, "secondary_5shot")):
|
| 293 |
+
d = os.path.join(root, key, t)
|
| 294 |
+
rc, cmd = run_task(a.model, t, shots, d, 0, a.full_gpu_hours)
|
| 295 |
+
h = harvest(d, t)
|
| 296 |
+
rec[key] = {"rc": rc, "rows": h.get("rows", {}), "command": cmd,
|
| 297 |
+
"shots": h.get("shots"), "harness_fewshot": h.get("config_fewshot"),
|
| 298 |
+
"n_input": h.get("n_input"), "n_effective": h.get("n_effective"),
|
| 299 |
+
"n_logged": h.get("n_logged"),
|
| 300 |
+
"limit": h.get("limit"), "model": h.get("model"), "dtype": h.get("dtype"),
|
| 301 |
+
"seed": h.get("seed"), "git_hash": h.get("git_hash"),
|
| 302 |
+
"dataset_path": h.get("dataset_path"), "split": h.get("split"),
|
| 303 |
+
"results_file": h.get("file"), "primary_metric": PRIMARY_METRIC[t],
|
| 304 |
+
"primary": (h.get("rows") or {}).get(PRIMARY_METRIC[t])}
|
| 305 |
+
if rc != 0 or not h.get("rows"):
|
| 306 |
+
# A cell pushed as `rows: {}` reads as "scored zero" to anyone holding only the JSON.
|
| 307 |
+
rec[key]["status"] = "FAILED rc=%s rows=%s" % (rc, len(h.get("rows") or {}))
|
| 308 |
+
elif h.get("limit") not in (None, 0):
|
| 309 |
+
raise SystemExit("%s/%s ran with --limit %s: a subset must not publish as a score"
|
| 310 |
+
% (t, key, h.get("limit")))
|
| 311 |
+
elif not (h.get("n_input") or h.get("n_logged")):
|
| 312 |
+
# No denominator beside a score is no denominator at all: say so in the record.
|
| 313 |
+
rec[key]["status"] = "NO ROW COUNT RECORDED"
|
| 314 |
+
elif key == "primary" and h.get("shots") is None:
|
| 315 |
+
raise SystemExit("%s: the harness recorded no shot count, so the table row would not be "
|
| 316 |
+
"reproducible" % t)
|
| 317 |
+
print("DONE %s %s rc %s rows %s" % (t, key, rc,
|
| 318 |
+
json.dumps(rec[key]["rows"], default=str)[:200]), flush=True)
|
| 319 |
+
table[t] = rec
|
| 320 |
+
json.dump(table, open(os.path.join(root, "results.json"), "w"), indent=1, default=str)
|
| 321 |
+
if not push(a, root): # after every task: an interruption keeps what finished
|
| 322 |
+
push_fail.append("after " + t)
|
| 323 |
+
if not push(a, root):
|
| 324 |
+
push_fail.append("final")
|
| 325 |
+
def ok(t, key):
|
| 326 |
+
r = table.get(t, {}).get(key, {})
|
| 327 |
+
return bool(r.get("rows")) and r.get("rc") == 0
|
| 328 |
+
|
| 329 |
+
# Both columns and both exit codes. `primary` rows alone would report a full house with an entirely
|
| 330 |
+
# empty 5-shot column, or with a leg that exited non-zero after writing something partial (C2).
|
| 331 |
+
missing = [f"{t}/{k}" for t in TASKS for k in ("primary", "secondary_5shot") if not ok(t, k)]
|
| 332 |
+
# The two columns run the same task on the same split with no limit, so a differing denominator means
|
| 333 |
+
# one of them resolved to a smaller subset -- a real-looking number for less data (E-045/9).
|
| 334 |
+
for t in TASKS:
|
| 335 |
+
p1 = table.get(t, {}).get("primary", {})
|
| 336 |
+
p2 = table.get(t, {}).get("secondary_5shot", {})
|
| 337 |
+
n1 = p1.get("n_input") or p1.get("n_logged")
|
| 338 |
+
n2 = p2.get("n_input") or p2.get("n_logged")
|
| 339 |
+
if n1 and n2 and n1 != n2:
|
| 340 |
+
missing.append("%s rows %s != %s between columns" % (t, n1, n2))
|
| 341 |
+
missing += push_fail
|
| 342 |
+
# Nine invocations, eight benchmarks: TruthfulQA is one row of the published table and its mc1 and mc2
|
| 343 |
+
# are separate task ids in the harness (docs/05-eval-plan.md §2).
|
| 344 |
+
print("VERDICT PHASE6 cells %d/%d (8 benchmarks x 2 columns) lm_eval %s missing %s" % (
|
| 345 |
+
2 * len(TASKS) - len(missing), 2 * len(TASKS), ver, missing or "none"), flush=True)
|
| 346 |
+
raise SystemExit(4 if missing else 0)
|
| 347 |
+
|
| 348 |
+
|
| 349 |
+
def push(a, root):
|
| 350 |
+
"""Publish into the model repo, which is where a reader expects the evaluation to live (§8)."""
|
| 351 |
+
repo = a.repo or a.model
|
| 352 |
+
if not repo:
|
| 353 |
+
print("(no --repo and no --model-derived target; results stay local)", flush=True)
|
| 354 |
+
return False
|
| 355 |
+
import ounce100m_credentials
|
| 356 |
+
ounce100m_credentials.install()
|
| 357 |
+
tok = os.environ.get("HF_TOKEN")
|
| 358 |
+
if tok is None:
|
| 359 |
+
print("FATAL: no token, refusing to claim results were published", flush=True)
|
| 360 |
+
return False
|
| 361 |
+
try:
|
| 362 |
+
from huggingface_hub import HfApi
|
| 363 |
+
rc = HfApi(token=tok).upload_folder(repo_id=repo, repo_type="model", folder_path=root,
|
| 364 |
+
path_in_repo="eval", commit_message="Phase 6 results",
|
| 365 |
+
# The --limit 5 smoke rows must not sit in the public repo beside
|
| 366 |
+
# the real table looking like results (C3).
|
| 367 |
+
ignore_patterns=["smoke*", "smoke/*", "smoke.json"],
|
| 368 |
+
commit_description="lm-eval 0.4.13, task-default shots plus a "
|
| 369 |
+
"uniform 5-shot column, per docs/05-eval-plan.md")
|
| 370 |
+
print("PUSHED eval/ ->", getattr(rc, "commit_url", str(rc)[:120]), flush=True)
|
| 371 |
+
# E-031's rule applied here: `upload_folder` returning is not evidence. `results.json` is the
|
| 372 |
+
# artifact the report cites, so re-fetch it over `resolve/` with no token and compare bytes.
|
| 373 |
+
try:
|
| 374 |
+
import hashlib as _h
|
| 375 |
+
import urllib.request as _u
|
| 376 |
+
want = _h.sha256(open(os.path.join(root, "results.json"), "rb").read()).hexdigest()
|
| 377 |
+
with _u.urlopen("https://huggingface.co/%s/resolve/main/eval/results.json?cb=%d"
|
| 378 |
+
% (repo, int(time.time())), timeout=300) as fh:
|
| 379 |
+
got = _h.sha256(fh.read()).hexdigest()
|
| 380 |
+
print("EVAL_READBACK sha_ok", got == want, flush=True)
|
| 381 |
+
return got == want
|
| 382 |
+
except Exception as e2:
|
| 383 |
+
print("EVAL_READBACK_FAILED", type(e2).__name__, str(e2)[:160], flush=True)
|
| 384 |
+
return False
|
| 385 |
+
except Exception as e:
|
| 386 |
+
print("PUSH_FAILED", type(e).__name__, str(e)[:200], flush=True)
|
| 387 |
+
return False
|
| 388 |
+
|
| 389 |
+
|
| 390 |
+
if __name__ == "__main__":
|
| 391 |
+
main()
|