File size: 6,639 Bytes
58258b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
#!/usr/bin/env python3
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0

"""Reference writer + self-test fixture for the local results pipeline.

Two jobs:

1. ``write_seed_result`` / ``write_compile_result`` are the CANONICAL writers.
   The local eval runner should call these (or emit byte-identical files) so
   compile_results_local.py can read its output. Nothing else defines the
   contract.

2. Running this file as a script materializes a small fake eval tree plus a
   matching models markdown, so the whole pipeline can be smoke-tested with no
   GPU and no real eval:

       python make_fixture_eval_tree.py /tmp/fixture
       python compile_results_local.py --root /tmp/fixture/evalout \\
           --models /tmp/fixture/evals.md --output /tmp/fixture/results.md \\
           --timestamp "2026-09-01 12:00" --no-cache
       cat /tmp/fixture/results.md
"""

from __future__ import annotations

import json
import os
import sys

# Display column name -> evalchemy task name. Must match compile_results_local.py.
BENCH_TO_TASK = {
    "GPQA_DIAMOND": "GPQADiamond",
    "JEEBENCH": "JEEBench",
    "HUMANITYS_LAST_EXAM": "HLE",
    "OLYMPIADBENCH_PHYSICS": "OlympiadBench_Physics",
    "LIVECODEBENCH": "LiveCodeBench",
    "LIVECODEBENCH_V5_OFFICIAL": "LiveCodeBenchv5_official",
    "LIVECODEBENCH_V6_OFFICIAL": "LiveCodeBenchv6_official",
}


def model_dir_name(model: str, step: int | None) -> str:
    """Eval output dir for a model/checkpoint.

    Baseline (HF id, step None): 'Qwen/Qwen3-8B' -> 'Qwen_Qwen3-8B'.
    Experiment: '<experiment_name>-step<N>'.
    """
    if step is None:
        return model.replace("/", "_")
    return f"{model}-step{step}"


def write_seed_result(
    root: str,
    model: str,
    step: int | None,
    task: str,
    seed: int,
    accuracy: float,
    timestamp: str = "2026-09-01T00-00-00.000000",
    extra: dict | None = None,
) -> str:
    """Write one (checkpoint, task, seed) result file. ``accuracy`` is a fraction.

    Layout: <root>/<model_dir>/<task>_seed<seed>/<model_sanitized>/results_<ISO>.json
    The reader takes ``results[task]["accuracy_avg"]``, falling back to
    ``["accuracy"]``, so single-pass benchmarks may emit either key.
    """
    d = os.path.join(root, model_dir_name(model, step), f"{task}_seed{seed}", model.replace("/", "__"))
    os.makedirs(d, exist_ok=True)
    payload = {"results": {task: {"accuracy_avg": accuracy, **(extra or {})}}}
    path = os.path.join(d, f"results_{timestamp}.json")
    with open(path, "w", encoding="utf-8") as fh:
        json.dump(payload, fh)
    return path


def write_compile_result(
    root: str,
    model: str,
    step: int | None,
    task: str,
    seeds: list[int],
    mean: float,
    std: float,
) -> str:
    """Write the averaged-across-seeds file for one (checkpoint, task).

    Layout: <root>/<model_dir>/compile_<task>_avg<N>seeds/compiled_results/averaged_results.json

    OPTIONAL. Omit it and compile_results_local.py averages the per-seed files
    itself (mean + sample n-1 std), which is the recommended local mode:
    evalchemy's own compile step re-grades with naive string equality and is
    wrong for HMMT / JEEBench / OlympiadBench_Physics / LiveCodeBench*.
    """
    n = len(seeds)
    d = os.path.join(root, model_dir_name(model, step), f"compile_{task}_avg{n}seeds", "compiled_results")
    os.makedirs(d, exist_ok=True)
    payload = [
        {
            "base_model_name": model,
            "dataset_name": task,
            "num_seeds": n,
            "seeds": seeds,
            "correct_mean": mean,
            "correct_std": std,
        }
    ]
    path = os.path.join(d, "averaged_results.json")
    with open(path, "w", encoding="utf-8") as fh:
        json.dump(payload, fh)
    return path


# ---------------------------------------------------------------------------
# Fixture
# ---------------------------------------------------------------------------

EVALS_MD = """# Local Evals

## Models to Evaluate

### Baselines
- Qwen/Qwen3-8B

### SFT 8B Experiments
- exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01
"""

BASELINE = "Qwen/Qwen3-8B"
EXPERIMENT = "exp_sft_qwen3_8b_selfinstill_mynewdata_n8_vr5_2k_lr5e6_wd01"
MATH_SEEDS = list(range(42, 52))  # 10 seeds


def build(out_dir: str) -> None:
    evalout = os.path.join(out_dir, "evalout")
    os.makedirs(evalout, exist_ok=True)
    with open(os.path.join(out_dir, "evals.md"), "w", encoding="utf-8") as fh:
        fh.write(EVALS_MD)

    # Baseline: MATH500/OlympiadBench single-seed, AIME* via compile files,
    # HMMT via per-seed files (compile is never trusted for HMMT).
    write_seed_result(evalout, BASELINE, None, "MATH500", 42, 0.884)
    write_seed_result(evalout, BASELINE, None, "OlympiadBench", 42, 0.617)
    write_compile_result(evalout, BASELINE, None, "AIME24", MATH_SEEDS, 0.6566666666666666, 0.06858355272004873)
    write_compile_result(evalout, BASELINE, None, "AIME25", MATH_SEEDS, 0.5866666666666667, 0.045460606091052)
    write_compile_result(evalout, BASELINE, None, "AIME26", MATH_SEEDS, 0.5933333333333334, 0.05426322800056239)
    for seed, acc in zip(MATH_SEEDS, [0.38, 0.40, 0.42, 0.36, 0.44, 0.38, 0.40, 0.42, 0.34, 0.39], strict=True):
        write_seed_result(evalout, BASELINE, None, "HMMT", seed, acc)

    steps = {
        100: dict(
            math500=0.872,
            olympiad=0.596,
            aime24=0.650,
            aime25=0.567,
            aime26=0.523,
            hmmt=[0.36, 0.38, 0.40, 0.34, 0.42, 0.36, 0.38, 0.40, 0.32, 0.37],
        ),
        200: dict(
            math500=0.906,
            olympiad=0.642,
            aime24=0.737,
            aime25=0.620,
            aime26=0.637,
            hmmt=[0.41, 0.43, 0.45, 0.39, 0.47, 0.41, 0.43, 0.45, 0.37, 0.42],
        ),
    }
    for step, d in steps.items():
        write_seed_result(evalout, EXPERIMENT, step, "MATH500", 42, d["math500"])
        write_seed_result(evalout, EXPERIMENT, step, "OlympiadBench", 42, d["olympiad"])
        write_compile_result(evalout, EXPERIMENT, step, "AIME24", MATH_SEEDS, d["aime24"], 0.0623)
        write_compile_result(evalout, EXPERIMENT, step, "AIME25", MATH_SEEDS, d["aime25"], 0.0531)
        write_compile_result(evalout, EXPERIMENT, step, "AIME26", MATH_SEEDS, d["aime26"], 0.0688)
        for seed, acc in zip(MATH_SEEDS, d["hmmt"], strict=True):
            write_seed_result(evalout, EXPERIMENT, step, "HMMT", seed, acc)

    print(f"fixture: {evalout}\nmodels:  {os.path.join(out_dir, 'evals.md')}")


if __name__ == "__main__":
    build(sys.argv[1] if len(sys.argv) > 1 else "fixture")