File size: 6,639 Bytes
58258b8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 | #!/usr/bin/env python3
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
"""Reference writer + self-test fixture for the local results pipeline.
Two jobs:
1. ``write_seed_result`` / ``write_compile_result`` are the CANONICAL writers.
The local eval runner should call these (or emit byte-identical files) so
compile_results_local.py can read its output. Nothing else defines the
contract.
2. Running this file as a script materializes a small fake eval tree plus a
matching models markdown, so the whole pipeline can be smoke-tested with no
GPU and no real eval:
python make_fixture_eval_tree.py /tmp/fixture
python compile_results_local.py --root /tmp/fixture/evalout \\
--models /tmp/fixture/evals.md --output /tmp/fixture/results.md \\
--timestamp "2026-09-01 12:00" --no-cache
cat /tmp/fixture/results.md
"""
from __future__ import annotations
import json
import os
import sys
# Display column name -> evalchemy task name. Must match compile_results_local.py.
BENCH_TO_TASK = {
"GPQA_DIAMOND": "GPQADiamond",
"JEEBENCH": "JEEBench",
"HUMANITYS_LAST_EXAM": "HLE",
"OLYMPIADBENCH_PHYSICS": "OlympiadBench_Physics",
"LIVECODEBENCH": "LiveCodeBench",
"LIVECODEBENCH_V5_OFFICIAL": "LiveCodeBenchv5_official",
"LIVECODEBENCH_V6_OFFICIAL": "LiveCodeBenchv6_official",
}
def model_dir_name(model: str, step: int | None) -> str:
"""Eval output dir for a model/checkpoint.
Baseline (HF id, step None): 'Qwen/Qwen3-8B' -> 'Qwen_Qwen3-8B'.
Experiment: '<experiment_name>-step<N>'.
"""
if step is None:
return model.replace("/", "_")
return f"{model}-step{step}"
def write_seed_result(
root: str,
model: str,
step: int | None,
task: str,
seed: int,
accuracy: float,
timestamp: str = "2026-09-01T00-00-00.000000",
extra: dict | None = None,
) -> str:
"""Write one (checkpoint, task, seed) result file. ``accuracy`` is a fraction.
Layout: <root>/<model_dir>/<task>_seed<seed>/<model_sanitized>/results_<ISO>.json
The reader takes ``results[task]["accuracy_avg"]``, falling back to
``["accuracy"]``, so single-pass benchmarks may emit either key.
"""
d = os.path.join(root, model_dir_name(model, step), f"{task}_seed{seed}", model.replace("/", "__"))
os.makedirs(d, exist_ok=True)
payload = {"results": {task: {"accuracy_avg": accuracy, **(extra or {})}}}
path = os.path.join(d, f"results_{timestamp}.json")
with open(path, "w", encoding="utf-8") as fh:
json.dump(payload, fh)
return path
def write_compile_result(
root: str,
model: str,
step: int | None,
task: str,
seeds: list[int],
mean: float,
std: float,
) -> str:
"""Write the averaged-across-seeds file for one (checkpoint, task).
Layout: <root>/<model_dir>/compile_<task>_avg<N>seeds/compiled_results/averaged_results.json
OPTIONAL. Omit it and compile_results_local.py averages the per-seed files
itself (mean + sample n-1 std), which is the recommended local mode:
evalchemy's own compile step re-grades with naive string equality and is
wrong for HMMT / JEEBench / OlympiadBench_Physics / LiveCodeBench*.
"""
n = len(seeds)
d = os.path.join(root, model_dir_name(model, step), f"compile_{task}_avg{n}seeds", "compiled_results")
os.makedirs(d, exist_ok=True)
payload = [
{
"base_model_name": model,
"dataset_name": task,
"num_seeds": n,
"seeds": seeds,
"correct_mean": mean,
"correct_std": std,
}
]
path = os.path.join(d, "averaged_results.json")
with open(path, "w", encoding="utf-8") as fh:
json.dump(payload, fh)
return path
# ---------------------------------------------------------------------------
# Fixture
# ---------------------------------------------------------------------------
EVALS_MD = """# Local Evals
## Models to Evaluate
### Baselines
- Qwen/Qwen3-8B
### SFT 8B Experiments
- exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01
"""
BASELINE = "Qwen/Qwen3-8B"
EXPERIMENT = "exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01"
MATH_SEEDS = list(range(42, 52)) # 10 seeds
def build(out_dir: str) -> None:
evalout = os.path.join(out_dir, "evalout")
os.makedirs(evalout, exist_ok=True)
with open(os.path.join(out_dir, "evals.md"), "w", encoding="utf-8") as fh:
fh.write(EVALS_MD)
# Baseline: MATH500/OlympiadBench single-seed, AIME* via compile files,
# HMMT via per-seed files (compile is never trusted for HMMT).
write_seed_result(evalout, BASELINE, None, "MATH500", 42, 0.884)
write_seed_result(evalout, BASELINE, None, "OlympiadBench", 42, 0.617)
write_compile_result(evalout, BASELINE, None, "AIME24", MATH_SEEDS, 0.6566666666666666, 0.06858355272004873)
write_compile_result(evalout, BASELINE, None, "AIME25", MATH_SEEDS, 0.5866666666666667, 0.045460606091052)
write_compile_result(evalout, BASELINE, None, "AIME26", MATH_SEEDS, 0.5933333333333334, 0.05426322800056239)
for seed, acc in zip(MATH_SEEDS, [0.38, 0.40, 0.42, 0.36, 0.44, 0.38, 0.40, 0.42, 0.34, 0.39], strict=True):
write_seed_result(evalout, BASELINE, None, "HMMT", seed, acc)
steps = {
100: dict(
math500=0.872,
olympiad=0.596,
aime24=0.650,
aime25=0.567,
aime26=0.523,
hmmt=[0.36, 0.38, 0.40, 0.34, 0.42, 0.36, 0.38, 0.40, 0.32, 0.37],
),
200: dict(
math500=0.906,
olympiad=0.642,
aime24=0.737,
aime25=0.620,
aime26=0.637,
hmmt=[0.41, 0.43, 0.45, 0.39, 0.47, 0.41, 0.43, 0.45, 0.37, 0.42],
),
}
for step, d in steps.items():
write_seed_result(evalout, EXPERIMENT, step, "MATH500", 42, d["math500"])
write_seed_result(evalout, EXPERIMENT, step, "OlympiadBench", 42, d["olympiad"])
write_compile_result(evalout, EXPERIMENT, step, "AIME24", MATH_SEEDS, d["aime24"], 0.0623)
write_compile_result(evalout, EXPERIMENT, step, "AIME25", MATH_SEEDS, d["aime25"], 0.0531)
write_compile_result(evalout, EXPERIMENT, step, "AIME26", MATH_SEEDS, d["aime26"], 0.0688)
for seed, acc in zip(MATH_SEEDS, d["hmmt"], strict=True):
write_seed_result(evalout, EXPERIMENT, step, "HMMT", seed, acc)
print(f"fixture: {evalout}\nmodels: {os.path.join(out_dir, 'evals.md')}")
if __name__ == "__main__":
build(sys.argv[1] if len(sys.argv) > 1 else "fixture")
|