File size: 8,186 Bytes
58258b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
#!/usr/bin/env bash
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
#
# Bootstrap a self-contained CUDA evalchemy + vLLM environment for evaluating
# LOCAL HF-format checkpoints on a single GPU box. No GCS, no Iris, no TPU.
#
#   ./bootstrap_gpu_eval.sh /opt/marin-gpu-eval
#
# Idempotent: re-running skips completed phases. Delete the phase markers under
# $ROOT/.markers to force a phase to re-run.
#
# ---------------------------------------------------------------------------
# WHY THESE PINS
# ---------------------------------------------------------------------------
# evalchemy    teetone/evalchemy @ 7f24168  (2026-03-31)
#     This is the pin the TPU pipeline used to produce the current results.md.
#     Use it if you need numbers comparable to the existing tables.
#     460b022 (2026-05-07) is the newer pin carried on the marin-agents
#     `evalfixhfauth` branch; it skips questions whose prompt >= max_model_len
#     instead of crashing, and adds the TTC task variants. Set
#     EVALCHEMY_COMMIT=460b022 if you hit prompt-length crashes.
#
# lm-eval      stanford-crfm/lm-evaluation-harness @ d5e3391f22cde186c827674d5c3ec7c5f4fe0cab
#     evalchemy is NOT pip-installable; it is a source checkout that imports
#     lm_eval. This exact fork+commit is what Marin pins. Upstream lm-eval 0.5.x
#     moved eval_logger and broke evalchemy's eval_tracker import.
#
# datasets     >=3.6,<4.0.0   HARD upper bound.
#     datasets v4 removed dataset-script support. LiveCodeBench loads
#     `livecodebench/code_generation_lite` with trust_remote_code=True, which is
#     a dataset script. On datasets>=4 every LCB task fails at load time.
#
# transformers >=4.51,<5      Qwen3 architecture support landed in 4.51.0.
#
# vllm         >=0.10         CUDA wheel. Do NOT install tpu-inference / the
#     `vllm` dependency group from lib/marin/pyproject.toml: that group pins
#     jax==0.10.0 + libtpu==0.0.40 + tpu-inference==0.22.1 and is TPU-only.
#
# ---------------------------------------------------------------------------
# HUGGING FACE AUTH -- what the `evalfixhfauth` branch actually does
# ---------------------------------------------------------------------------
# The marin-agents branch name is stale/aspirational: no commit on it touches HF
# authentication. The real HF-related fixes on that branch are three, and a
# local user must reproduce all three:
#
#   1. HF_DATASETS_TRUST_REMOTE_CODE=1 in the eval process env (LiveCodeBench
#      needs it). This script exports it in the generated activate snippet, and
#      gpu_eval_driver.py sets it on every subprocess.
#   2. datasets<4.0.0 pin (above).
#   3. HF_TOKEN supplied as a process env var at submit time -- it is never in
#      source. On TPU it was threaded through `iris job run -e HF_TOKEN ...`.
#      LOCALLY YOU MUST DO THIS INSTEAD:
#          huggingface-cli login          # writes ~/.cache/huggingface/token
#      or  export HF_TOKEN=hf_xxx         # per shell / systemd unit
#      Required for gated datasets: cais/hle (HLE) and Idavidrein/gpqa
#      (GPQADiamond) -- you must ALSO click "Agree" on those dataset pages with
#      the same account. The math suite (MATH500, OlympiadBench, AIME24/25/26,
#      HMMT) and the code suite (LiveCodeBench*) are public and need no token,
#      though an anonymous box will hit HF rate limits during a 10-seed sweep.
#
set -euo pipefail

ROOT="${1:-/opt/marin-gpu-eval}"
EVALCHEMY_REPO="${EVALCHEMY_REPO:-https://github.com/teetone/evalchemy.git}"
EVALCHEMY_COMMIT="${EVALCHEMY_COMMIT:-7f24168}"
PYTHON_VERSION="${PYTHON_VERSION:-3.11}"
LM_EVAL_REQ="lm-eval[math]@git+https://github.com/stanford-crfm/lm-evaluation-harness@d5e3391f22cde186c827674d5c3ec7c5f4fe0cab"

VENV="$ROOT/venv"
EVALCHEMY_DIR="$ROOT/evalchemy"
CACHE="$ROOT/cache"
MARKERS="$ROOT/.markers"
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"

mkdir -p "$ROOT" "$CACHE" "$MARKERS"

log() { printf '[bootstrap] %s\n' "$*" >&2; }

if ! command -v uv >/dev/null 2>&1; then
  log "uv not found on PATH; install it with: curl -LsSf https://astral.sh/uv/install.sh | sh"
  exit 1
fi

# --------------------------------------------------------------------------
# 1. venv
# --------------------------------------------------------------------------
if [[ ! -x "$VENV/bin/python" ]]; then
  log "creating venv at $VENV (python $PYTHON_VERSION)"
  uv venv --python "$PYTHON_VERSION" --seed "$VENV"
fi
PY="$VENV/bin/python"

# --------------------------------------------------------------------------
# 2. dependencies
# --------------------------------------------------------------------------
if [[ ! -f "$MARKERS/deps" ]]; then
  log "installing CUDA vLLM + lm-eval + evalchemy runtime deps"
  # Driver-aware: on driver 535 (no CUDA 12.6+) invoke with
  #   TORCH_PRE="torch==2.6.0 torchvision==0.21.0 --index-url https://download.pytorch.org/whl/cu124"
  #   VLLM_SPEC="vllm==0.8.5"
  # vLLM >=0.9 ships cu126/cu130 wheels that will NOT run on driver 535.
  if [[ -n "${TORCH_PRE:-}" ]]; then
    log "pre-installing pinned torch: $TORCH_PRE"
    UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" $TORCH_PRE
  fi
  UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" \
    "${VLLM_SPEC:-vllm>=0.10}" \
    "transformers>=4.51,<5" \
    "datasets>=3.6,<4.0.0" \
    "fsspec[http]>=2023.1.0" \
    "accelerate>=1.8" \
    "safetensors" \
    "sentencepiece" \
    "$LM_EVAL_REQ" \
    "sympy>=1.12.1,<1.14" \
    "antlr4-python3-runtime==4.11" \
    "bespokelabs-curator==0.1.16" \
    "sqlalchemy" \
    "fire" \
    "pandas" \
    "numpy"
  touch "$MARKERS/deps"
else
  log "deps marker present; skipping install"
fi

# Guard: datasets v4 silently breaks every LiveCodeBench task.
"$PY" - <<'PYCHECK'
import sys
from importlib.metadata import version
dv = version("datasets")
major = int(dv.split(".")[0])
if major >= 4:
    sys.exit(
        f"datasets {dv} installed but LiveCodeBench requires <4.0.0 "
        "(v4 dropped dataset-script support). Reinstall with 'datasets>=3.6,<4.0.0'."
    )
print(f"[bootstrap] datasets {dv} OK")
PYCHECK

# --------------------------------------------------------------------------
# 3. evalchemy checkout, pinned + patched
# --------------------------------------------------------------------------
if [[ ! -d "$EVALCHEMY_DIR/.git" ]]; then
  log "cloning evalchemy -> $EVALCHEMY_DIR"
  git clone "$EVALCHEMY_REPO" "$EVALCHEMY_DIR"
fi

log "pinning evalchemy to $EVALCHEMY_COMMIT and reverting previous patches"
git -C "$EVALCHEMY_DIR" fetch --all --quiet || true
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT"
# Hard-reset the files the patcher rewrites so patching is deterministic.
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" -- eval/

log "applying GPU patches"
"$PY" "$HERE/patch_evalchemy_gpu.py" "$EVALCHEMY_DIR"

# --------------------------------------------------------------------------
# 4. activate snippet
# --------------------------------------------------------------------------
cat > "$ROOT/activate.sh" <<ACTEOF
# source this before running gpu_eval_driver.py / gpu_eval_monitor.py by hand
export MARIN_GPU_EVAL_ROOT="$ROOT"
export EVALCHEMY_DIR="$EVALCHEMY_DIR"
export VIRTUAL_ENV="$VENV"
export PATH="$VENV/bin:\$PATH"

# HF / dataset behaviour required by LiveCodeBench and the math suite
export HF_DATASETS_TRUST_REMOTE_CODE=1
export HF_HOME="$CACHE/huggingface"
export HF_HUB_CACHE="$CACHE/huggingface/hub"
export HF_DATASETS_CACHE="$CACHE/datasets"
export HF_ALLOW_CODE_EVAL=1

# vLLM / torch caches kept off \$HOME so concurrent runs do not fight
export VLLM_CACHE_ROOT="$CACHE/vllm"
export VLLM_CONFIG_ROOT="$CACHE/vllm"
export TORCHINDUCTOR_CACHE_DIR="$CACHE/torchinductor"
export TRITON_CACHE_DIR="$CACHE/triton"
export TOKENIZERS_PARALLELISM=false

# Supply your own:
#   export HF_TOKEN=hf_xxx     # needed for cais/hle and Idavidrein/gpqa
ACTEOF

log "done"
log ""
log "  source $ROOT/activate.sh"
log "  export HF_TOKEN=...            # only needed for the science suite"
log "  python $HERE/gpu_eval_driver.py --help"
log ""
log "evalchemy: $EVALCHEMY_DIR @ $(git -C "$EVALCHEMY_DIR" rev-parse --short HEAD)"
log "python:    $PY"