Download gpu-sft/claude/make_fixture_eval_tree.py from fzzhang/svd-code: direct link, hf CLI and curl.
- Browser
- Download file 6.64 kB
-
https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/claude/make_fixture_eval_tree.py
- Command line
-
hf download hf://fzzhang/svd-code/gpu-sft/claude/make_fixture_eval_tree.py
-
curl -L -o make_fixture_eval_tree.py https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/claude/make_fixture_eval_tree.py
6.64 kB
| #!/usr/bin/env python3 | |
| # Copyright The Marin Authors | |
| # SPDX-License-Identifier: Apache-2.0 | |
| """Reference writer + self-test fixture for the local results pipeline. | |
| Two jobs: | |
| 1. ``write_seed_result`` / ``write_compile_result`` are the CANONICAL writers. | |
| The local eval runner should call these (or emit byte-identical files) so | |
| compile_results_local.py can read its output. Nothing else defines the | |
| contract. | |
| 2. Running this file as a script materializes a small fake eval tree plus a | |
| matching models markdown, so the whole pipeline can be smoke-tested with no | |
| GPU and no real eval: | |
| python make_fixture_eval_tree.py /tmp/fixture | |
| python compile_results_local.py --root /tmp/fixture/evalout \\ | |
| --models /tmp/fixture/evals.md --output /tmp/fixture/results.md \\ | |
| --timestamp "2026-09-01 12:00" --no-cache | |
| cat /tmp/fixture/results.md | |
| """ | |
| from __future__ import annotations | |
| import json | |
| import os | |
| import sys | |
| # Display column name -> evalchemy task name. Must match compile_results_local.py. | |
| BENCH_TO_TASK = { | |
| "GPQA_DIAMOND": "GPQADiamond", | |
| "JEEBENCH": "JEEBench", | |
| "HUMANITYS_LAST_EXAM": "HLE", | |
| "OLYMPIADBENCH_PHYSICS": "OlympiadBench_Physics", | |
| "LIVECODEBENCH": "LiveCodeBench", | |
| "LIVECODEBENCH_V5_OFFICIAL": "LiveCodeBenchv5_official", | |
| "LIVECODEBENCH_V6_OFFICIAL": "LiveCodeBenchv6_official", | |
| } | |
| def model_dir_name(model: str, step: int | None) -> str: | |
| """Eval output dir for a model/checkpoint. | |
| Baseline (HF id, step None): 'Qwen/Qwen3-8B' -> 'Qwen_Qwen3-8B'. | |
| Experiment: '<experiment_name>-step<N>'. | |
| """ | |
| if step is None: | |
| return model.replace("/", "_") | |
| return f"{model}-step{step}" | |
| def write_seed_result( | |
| root: str, | |
| model: str, | |
| step: int | None, | |
| task: str, | |
| seed: int, | |
| accuracy: float, | |
| timestamp: str = "2026-09-01T00-00-00.000000", | |
| extra: dict | None = None, | |
| ) -> str: | |
| """Write one (checkpoint, task, seed) result file. ``accuracy`` is a fraction. | |
| Layout: <root>/<model_dir>/<task>_seed<seed>/<model_sanitized>/results_<ISO>.json | |
| The reader takes ``results[task]["accuracy_avg"]``, falling back to | |
| ``["accuracy"]``, so single-pass benchmarks may emit either key. | |
| """ | |
| d = os.path.join(root, model_dir_name(model, step), f"{task}_seed{seed}", model.replace("/", "__")) | |
| os.makedirs(d, exist_ok=True) | |
| payload = {"results": {task: {"accuracy_avg": accuracy, **(extra or {})}}} | |
| path = os.path.join(d, f"results_{timestamp}.json") | |
| with open(path, "w", encoding="utf-8") as fh: | |
| json.dump(payload, fh) | |
| return path | |
| def write_compile_result( | |
| root: str, | |
| model: str, | |
| step: int | None, | |
| task: str, | |
| seeds: list[int], | |
| mean: float, | |
| std: float, | |
| ) -> str: | |
| """Write the averaged-across-seeds file for one (checkpoint, task). | |
| Layout: <root>/<model_dir>/compile_<task>_avg<N>seeds/compiled_results/averaged_results.json | |
| OPTIONAL. Omit it and compile_results_local.py averages the per-seed files | |
| itself (mean + sample n-1 std), which is the recommended local mode: | |
| evalchemy's own compile step re-grades with naive string equality and is | |
| wrong for HMMT / JEEBench / OlympiadBench_Physics / LiveCodeBench*. | |
| """ | |
| n = len(seeds) | |
| d = os.path.join(root, model_dir_name(model, step), f"compile_{task}_avg{n}seeds", "compiled_results") | |
| os.makedirs(d, exist_ok=True) | |
| payload = [ | |
| { | |
| "base_model_name": model, | |
| "dataset_name": task, | |
| "num_seeds": n, | |
| "seeds": seeds, | |
| "correct_mean": mean, | |
| "correct_std": std, | |
| } | |
| ] | |
| path = os.path.join(d, "averaged_results.json") | |
| with open(path, "w", encoding="utf-8") as fh: | |
| json.dump(payload, fh) | |
| return path | |
| # --------------------------------------------------------------------------- | |
| # Fixture | |
| # --------------------------------------------------------------------------- | |
| EVALS_MD = """# Local Evals | |
| ## Models to Evaluate | |
| ### Baselines | |
| - Qwen/Qwen3-8B | |
| ### SFT 8B Experiments | |
| - exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01 | |
| """ | |
| BASELINE = "Qwen/Qwen3-8B" | |
| EXPERIMENT = "exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01" | |
| MATH_SEEDS = list(range(42, 52)) # 10 seeds | |
| def build(out_dir: str) -> None: | |
| evalout = os.path.join(out_dir, "evalout") | |
| os.makedirs(evalout, exist_ok=True) | |
| with open(os.path.join(out_dir, "evals.md"), "w", encoding="utf-8") as fh: | |
| fh.write(EVALS_MD) | |
| # Baseline: MATH500/OlympiadBench single-seed, AIME* via compile files, | |
| # HMMT via per-seed files (compile is never trusted for HMMT). | |
| write_seed_result(evalout, BASELINE, None, "MATH500", 42, 0.884) | |
| write_seed_result(evalout, BASELINE, None, "OlympiadBench", 42, 0.617) | |
| write_compile_result(evalout, BASELINE, None, "AIME24", MATH_SEEDS, 0.6566666666666666, 0.06858355272004873) | |
| write_compile_result(evalout, BASELINE, None, "AIME25", MATH_SEEDS, 0.5866666666666667, 0.045460606091052) | |
| write_compile_result(evalout, BASELINE, None, "AIME26", MATH_SEEDS, 0.5933333333333334, 0.05426322800056239) | |
| for seed, acc in zip(MATH_SEEDS, [0.38, 0.40, 0.42, 0.36, 0.44, 0.38, 0.40, 0.42, 0.34, 0.39], strict=True): | |
| write_seed_result(evalout, BASELINE, None, "HMMT", seed, acc) | |
| steps = { | |
| 100: dict( | |
| math500=0.872, | |
| olympiad=0.596, | |
| aime24=0.650, | |
| aime25=0.567, | |
| aime26=0.523, | |
| hmmt=[0.36, 0.38, 0.40, 0.34, 0.42, 0.36, 0.38, 0.40, 0.32, 0.37], | |
| ), | |
| 200: dict( | |
| math500=0.906, | |
| olympiad=0.642, | |
| aime24=0.737, | |
| aime25=0.620, | |
| aime26=0.637, | |
| hmmt=[0.41, 0.43, 0.45, 0.39, 0.47, 0.41, 0.43, 0.45, 0.37, 0.42], | |
| ), | |
| } | |
| for step, d in steps.items(): | |
| write_seed_result(evalout, EXPERIMENT, step, "MATH500", 42, d["math500"]) | |
| write_seed_result(evalout, EXPERIMENT, step, "OlympiadBench", 42, d["olympiad"]) | |
| write_compile_result(evalout, EXPERIMENT, step, "AIME24", MATH_SEEDS, d["aime24"], 0.0623) | |
| write_compile_result(evalout, EXPERIMENT, step, "AIME25", MATH_SEEDS, d["aime25"], 0.0531) | |
| write_compile_result(evalout, EXPERIMENT, step, "AIME26", MATH_SEEDS, d["aime26"], 0.0688) | |
| for seed, acc in zip(MATH_SEEDS, d["hmmt"], strict=True): | |
| write_seed_result(evalout, EXPERIMENT, step, "HMMT", seed, acc) | |
| print(f"fixture: {evalout}\nmodels: {os.path.join(out_dir, 'evals.md')}") | |
| if __name__ == "__main__": | |
| build(sys.argv[1] if len(sys.argv) > 1 else "fixture") | |