svd-code / gpu-sft /claude /make_fixture_eval_tree.py
fzzhang's picture
Upload folder using huggingface_hub
58258b8 verified
Raw History Blame Contribute Delete
6.64 kB
#!/usr/bin/env python3
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
"""Reference writer + self-test fixture for the local results pipeline.
Two jobs:
1. ``write_seed_result`` / ``write_compile_result`` are the CANONICAL writers.
The local eval runner should call these (or emit byte-identical files) so
compile_results_local.py can read its output. Nothing else defines the
contract.
2. Running this file as a script materializes a small fake eval tree plus a
matching models markdown, so the whole pipeline can be smoke-tested with no
GPU and no real eval:
python make_fixture_eval_tree.py /tmp/fixture
python compile_results_local.py --root /tmp/fixture/evalout \\
--models /tmp/fixture/evals.md --output /tmp/fixture/results.md \\
--timestamp "2026-09-01 12:00" --no-cache
cat /tmp/fixture/results.md
"""
from __future__ import annotations
import json
import os
import sys
# Display column name -> evalchemy task name. Must match compile_results_local.py.
BENCH_TO_TASK = {
"GPQA_DIAMOND": "GPQADiamond",
"JEEBENCH": "JEEBench",
"HUMANITYS_LAST_EXAM": "HLE",
"OLYMPIADBENCH_PHYSICS": "OlympiadBench_Physics",
"LIVECODEBENCH": "LiveCodeBench",
"LIVECODEBENCH_V5_OFFICIAL": "LiveCodeBenchv5_official",
"LIVECODEBENCH_V6_OFFICIAL": "LiveCodeBenchv6_official",
}
def model_dir_name(model: str, step: int | None) -> str:
"""Eval output dir for a model/checkpoint.
Baseline (HF id, step None): 'Qwen/Qwen3-8B' -> 'Qwen_Qwen3-8B'.
Experiment: '<experiment_name>-step<N>'.
"""
if step is None:
return model.replace("/", "_")
return f"{model}-step{step}"
def write_seed_result(
root: str,
model: str,
step: int | None,
task: str,
seed: int,
accuracy: float,
timestamp: str = "2026-09-01T00-00-00.000000",
extra: dict | None = None,
) -> str:
"""Write one (checkpoint, task, seed) result file. ``accuracy`` is a fraction.
Layout: <root>/<model_dir>/<task>_seed<seed>/<model_sanitized>/results_<ISO>.json
The reader takes ``results[task]["accuracy_avg"]``, falling back to
``["accuracy"]``, so single-pass benchmarks may emit either key.
"""
d = os.path.join(root, model_dir_name(model, step), f"{task}_seed{seed}", model.replace("/", "__"))
os.makedirs(d, exist_ok=True)
payload = {"results": {task: {"accuracy_avg": accuracy, **(extra or {})}}}
path = os.path.join(d, f"results_{timestamp}.json")
with open(path, "w", encoding="utf-8") as fh:
json.dump(payload, fh)
return path
def write_compile_result(
root: str,
model: str,
step: int | None,
task: str,
seeds: list[int],
mean: float,
std: float,
) -> str:
"""Write the averaged-across-seeds file for one (checkpoint, task).
Layout: <root>/<model_dir>/compile_<task>_avg<N>seeds/compiled_results/averaged_results.json
OPTIONAL. Omit it and compile_results_local.py averages the per-seed files
itself (mean + sample n-1 std), which is the recommended local mode:
evalchemy's own compile step re-grades with naive string equality and is
wrong for HMMT / JEEBench / OlympiadBench_Physics / LiveCodeBench*.
"""
n = len(seeds)
d = os.path.join(root, model_dir_name(model, step), f"compile_{task}_avg{n}seeds", "compiled_results")
os.makedirs(d, exist_ok=True)
payload = [
{
"base_model_name": model,
"dataset_name": task,
"num_seeds": n,
"seeds": seeds,
"correct_mean": mean,
"correct_std": std,
}
]
path = os.path.join(d, "averaged_results.json")
with open(path, "w", encoding="utf-8") as fh:
json.dump(payload, fh)
return path
# ---------------------------------------------------------------------------
# Fixture
# ---------------------------------------------------------------------------
EVALS_MD = """# Local Evals
## Models to Evaluate
### Baselines
- Qwen/Qwen3-8B
### SFT 8B Experiments
- exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01
"""
BASELINE = "Qwen/Qwen3-8B"
EXPERIMENT = "exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01"
MATH_SEEDS = list(range(42, 52)) # 10 seeds
def build(out_dir: str) -> None:
evalout = os.path.join(out_dir, "evalout")
os.makedirs(evalout, exist_ok=True)
with open(os.path.join(out_dir, "evals.md"), "w", encoding="utf-8") as fh:
fh.write(EVALS_MD)
# Baseline: MATH500/OlympiadBench single-seed, AIME* via compile files,
# HMMT via per-seed files (compile is never trusted for HMMT).
write_seed_result(evalout, BASELINE, None, "MATH500", 42, 0.884)
write_seed_result(evalout, BASELINE, None, "OlympiadBench", 42, 0.617)
write_compile_result(evalout, BASELINE, None, "AIME24", MATH_SEEDS, 0.6566666666666666, 0.06858355272004873)
write_compile_result(evalout, BASELINE, None, "AIME25", MATH_SEEDS, 0.5866666666666667, 0.045460606091052)
write_compile_result(evalout, BASELINE, None, "AIME26", MATH_SEEDS, 0.5933333333333334, 0.05426322800056239)
for seed, acc in zip(MATH_SEEDS, [0.38, 0.40, 0.42, 0.36, 0.44, 0.38, 0.40, 0.42, 0.34, 0.39], strict=True):
write_seed_result(evalout, BASELINE, None, "HMMT", seed, acc)
steps = {
100: dict(
math500=0.872,
olympiad=0.596,
aime24=0.650,
aime25=0.567,
aime26=0.523,
hmmt=[0.36, 0.38, 0.40, 0.34, 0.42, 0.36, 0.38, 0.40, 0.32, 0.37],
),
200: dict(
math500=0.906,
olympiad=0.642,
aime24=0.737,
aime25=0.620,
aime26=0.637,
hmmt=[0.41, 0.43, 0.45, 0.39, 0.47, 0.41, 0.43, 0.45, 0.37, 0.42],
),
}
for step, d in steps.items():
write_seed_result(evalout, EXPERIMENT, step, "MATH500", 42, d["math500"])
write_seed_result(evalout, EXPERIMENT, step, "OlympiadBench", 42, d["olympiad"])
write_compile_result(evalout, EXPERIMENT, step, "AIME24", MATH_SEEDS, d["aime24"], 0.0623)
write_compile_result(evalout, EXPERIMENT, step, "AIME25", MATH_SEEDS, d["aime25"], 0.0531)
write_compile_result(evalout, EXPERIMENT, step, "AIME26", MATH_SEEDS, d["aime26"], 0.0688)
for seed, acc in zip(MATH_SEEDS, d["hmmt"], strict=True):
write_seed_result(evalout, EXPERIMENT, step, "HMMT", seed, acc)
print(f"fixture: {evalout}\nmodels: {os.path.join(out_dir, 'evals.md')}")
if __name__ == "__main__":
build(sys.argv[1] if len(sys.argv) > 1 else "fixture")