#!/usr/bin/env python3 # Copyright The Marin Authors # SPDX-License-Identifier: Apache-2.0 """Reference writer + self-test fixture for the local results pipeline. Two jobs: 1. ``write_seed_result`` / ``write_compile_result`` are the CANONICAL writers. The local eval runner should call these (or emit byte-identical files) so compile_results_local.py can read its output. Nothing else defines the contract. 2. Running this file as a script materializes a small fake eval tree plus a matching models markdown, so the whole pipeline can be smoke-tested with no GPU and no real eval: python make_fixture_eval_tree.py /tmp/fixture python compile_results_local.py --root /tmp/fixture/evalout \\ --models /tmp/fixture/evals.md --output /tmp/fixture/results.md \\ --timestamp "2026-09-01 12:00" --no-cache cat /tmp/fixture/results.md """ from __future__ import annotations import json import os import sys # Display column name -> evalchemy task name. Must match compile_results_local.py. BENCH_TO_TASK = { "GPQA_DIAMOND": "GPQADiamond", "JEEBENCH": "JEEBench", "HUMANITYS_LAST_EXAM": "HLE", "OLYMPIADBENCH_PHYSICS": "OlympiadBench_Physics", "LIVECODEBENCH": "LiveCodeBench", "LIVECODEBENCH_V5_OFFICIAL": "LiveCodeBenchv5_official", "LIVECODEBENCH_V6_OFFICIAL": "LiveCodeBenchv6_official", } def model_dir_name(model: str, step: int | None) -> str: """Eval output dir for a model/checkpoint. Baseline (HF id, step None): 'Qwen/Qwen3-8B' -> 'Qwen_Qwen3-8B'. Experiment: '-step'. """ if step is None: return model.replace("/", "_") return f"{model}-step{step}" def write_seed_result( root: str, model: str, step: int | None, task: str, seed: int, accuracy: float, timestamp: str = "2026-09-01T00-00-00.000000", extra: dict | None = None, ) -> str: """Write one (checkpoint, task, seed) result file. ``accuracy`` is a fraction. Layout: //_seed//results_.json The reader takes ``results[task]["accuracy_avg"]``, falling back to ``["accuracy"]``, so single-pass benchmarks may emit either key. """ d = os.path.join(root, model_dir_name(model, step), f"{task}_seed{seed}", model.replace("/", "__")) os.makedirs(d, exist_ok=True) payload = {"results": {task: {"accuracy_avg": accuracy, **(extra or {})}}} path = os.path.join(d, f"results_{timestamp}.json") with open(path, "w", encoding="utf-8") as fh: json.dump(payload, fh) return path def write_compile_result( root: str, model: str, step: int | None, task: str, seeds: list[int], mean: float, std: float, ) -> str: """Write the averaged-across-seeds file for one (checkpoint, task). Layout: //compile__avgseeds/compiled_results/averaged_results.json OPTIONAL. Omit it and compile_results_local.py averages the per-seed files itself (mean + sample n-1 std), which is the recommended local mode: evalchemy's own compile step re-grades with naive string equality and is wrong for HMMT / JEEBench / OlympiadBench_Physics / LiveCodeBench*. """ n = len(seeds) d = os.path.join(root, model_dir_name(model, step), f"compile_{task}_avg{n}seeds", "compiled_results") os.makedirs(d, exist_ok=True) payload = [ { "base_model_name": model, "dataset_name": task, "num_seeds": n, "seeds": seeds, "correct_mean": mean, "correct_std": std, } ] path = os.path.join(d, "averaged_results.json") with open(path, "w", encoding="utf-8") as fh: json.dump(payload, fh) return path # --------------------------------------------------------------------------- # Fixture # --------------------------------------------------------------------------- EVALS_MD = """# Local Evals ## Models to Evaluate ### Baselines - Qwen/Qwen3-8B ### SFT 8B Experiments - exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01 """ BASELINE = "Qwen/Qwen3-8B" EXPERIMENT = "exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01" MATH_SEEDS = list(range(42, 52)) # 10 seeds def build(out_dir: str) -> None: evalout = os.path.join(out_dir, "evalout") os.makedirs(evalout, exist_ok=True) with open(os.path.join(out_dir, "evals.md"), "w", encoding="utf-8") as fh: fh.write(EVALS_MD) # Baseline: MATH500/OlympiadBench single-seed, AIME* via compile files, # HMMT via per-seed files (compile is never trusted for HMMT). write_seed_result(evalout, BASELINE, None, "MATH500", 42, 0.884) write_seed_result(evalout, BASELINE, None, "OlympiadBench", 42, 0.617) write_compile_result(evalout, BASELINE, None, "AIME24", MATH_SEEDS, 0.6566666666666666, 0.06858355272004873) write_compile_result(evalout, BASELINE, None, "AIME25", MATH_SEEDS, 0.5866666666666667, 0.045460606091052) write_compile_result(evalout, BASELINE, None, "AIME26", MATH_SEEDS, 0.5933333333333334, 0.05426322800056239) for seed, acc in zip(MATH_SEEDS, [0.38, 0.40, 0.42, 0.36, 0.44, 0.38, 0.40, 0.42, 0.34, 0.39], strict=True): write_seed_result(evalout, BASELINE, None, "HMMT", seed, acc) steps = { 100: dict( math500=0.872, olympiad=0.596, aime24=0.650, aime25=0.567, aime26=0.523, hmmt=[0.36, 0.38, 0.40, 0.34, 0.42, 0.36, 0.38, 0.40, 0.32, 0.37], ), 200: dict( math500=0.906, olympiad=0.642, aime24=0.737, aime25=0.620, aime26=0.637, hmmt=[0.41, 0.43, 0.45, 0.39, 0.47, 0.41, 0.43, 0.45, 0.37, 0.42], ), } for step, d in steps.items(): write_seed_result(evalout, EXPERIMENT, step, "MATH500", 42, d["math500"]) write_seed_result(evalout, EXPERIMENT, step, "OlympiadBench", 42, d["olympiad"]) write_compile_result(evalout, EXPERIMENT, step, "AIME24", MATH_SEEDS, d["aime24"], 0.0623) write_compile_result(evalout, EXPERIMENT, step, "AIME25", MATH_SEEDS, d["aime25"], 0.0531) write_compile_result(evalout, EXPERIMENT, step, "AIME26", MATH_SEEDS, d["aime26"], 0.0688) for seed, acc in zip(MATH_SEEDS, d["hmmt"], strict=True): write_seed_result(evalout, EXPERIMENT, step, "HMMT", seed, acc) print(f"fixture: {evalout}\nmodels: {os.path.join(out_dir, 'evals.md')}") if __name__ == "__main__": build(sys.argv[1] if len(sys.argv) > 1 else "fixture")