File size: 8,186 Bytes
58258b8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 | #!/usr/bin/env bash
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
#
# Bootstrap a self-contained CUDA evalchemy + vLLM environment for evaluating
# LOCAL HF-format checkpoints on a single GPU box. No GCS, no Iris, no TPU.
#
# ./bootstrap_gpu_eval.sh /opt/marin-gpu-eval
#
# Idempotent: re-running skips completed phases. Delete the phase markers under
# $ROOT/.markers to force a phase to re-run.
#
# ---------------------------------------------------------------------------
# WHY THESE PINS
# ---------------------------------------------------------------------------
# evalchemy teetone/evalchemy @ 7f24168 (2026-03-31)
# This is the pin the TPU pipeline used to produce the current results.md.
# Use it if you need numbers comparable to the existing tables.
# 460b022 (2026-05-07) is the newer pin carried on the marin-agents
# `evalfixhfauth` branch; it skips questions whose prompt >= max_model_len
# instead of crashing, and adds the TTC task variants. Set
# EVALCHEMY_COMMIT=460b022 if you hit prompt-length crashes.
#
# lm-eval stanford-crfm/lm-evaluation-harness @ d5e3391f22cde186c827674d5c3ec7c5f4fe0cab
# evalchemy is NOT pip-installable; it is a source checkout that imports
# lm_eval. This exact fork+commit is what Marin pins. Upstream lm-eval 0.5.x
# moved eval_logger and broke evalchemy's eval_tracker import.
#
# datasets >=3.6,<4.0.0 HARD upper bound.
# datasets v4 removed dataset-script support. LiveCodeBench loads
# `livecodebench/code_generation_lite` with trust_remote_code=True, which is
# a dataset script. On datasets>=4 every LCB task fails at load time.
#
# transformers >=4.51,<5 Qwen3 architecture support landed in 4.51.0.
#
# vllm >=0.10 CUDA wheel. Do NOT install tpu-inference / the
# `vllm` dependency group from lib/marin/pyproject.toml: that group pins
# jax==0.10.0 + libtpu==0.0.40 + tpu-inference==0.22.1 and is TPU-only.
#
# ---------------------------------------------------------------------------
# HUGGING FACE AUTH -- what the `evalfixhfauth` branch actually does
# ---------------------------------------------------------------------------
# The marin-agents branch name is stale/aspirational: no commit on it touches HF
# authentication. The real HF-related fixes on that branch are three, and a
# local user must reproduce all three:
#
# 1. HF_DATASETS_TRUST_REMOTE_CODE=1 in the eval process env (LiveCodeBench
# needs it). This script exports it in the generated activate snippet, and
# gpu_eval_driver.py sets it on every subprocess.
# 2. datasets<4.0.0 pin (above).
# 3. HF_TOKEN supplied as a process env var at submit time -- it is never in
# source. On TPU it was threaded through `iris job run -e HF_TOKEN ...`.
# LOCALLY YOU MUST DO THIS INSTEAD:
# huggingface-cli login # writes ~/.cache/huggingface/token
# or export HF_TOKEN=hf_xxx # per shell / systemd unit
# Required for gated datasets: cais/hle (HLE) and Idavidrein/gpqa
# (GPQADiamond) -- you must ALSO click "Agree" on those dataset pages with
# the same account. The math suite (MATH500, OlympiadBench, AIME24/25/26,
# HMMT) and the code suite (LiveCodeBench*) are public and need no token,
# though an anonymous box will hit HF rate limits during a 10-seed sweep.
#
set -euo pipefail
ROOT="${1:-/opt/marin-gpu-eval}"
EVALCHEMY_REPO="${EVALCHEMY_REPO:-https://github.com/teetone/evalchemy.git}"
EVALCHEMY_COMMIT="${EVALCHEMY_COMMIT:-7f24168}"
PYTHON_VERSION="${PYTHON_VERSION:-3.11}"
LM_EVAL_REQ="lm-eval[math]@git+https://github.com/stanford-crfm/lm-evaluation-harness@d5e3391f22cde186c827674d5c3ec7c5f4fe0cab"
VENV="$ROOT/venv"
EVALCHEMY_DIR="$ROOT/evalchemy"
CACHE="$ROOT/cache"
MARKERS="$ROOT/.markers"
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
mkdir -p "$ROOT" "$CACHE" "$MARKERS"
log() { printf '[bootstrap] %s\n' "$*" >&2; }
if ! command -v uv >/dev/null 2>&1; then
log "uv not found on PATH; install it with: curl -LsSf https://astral.sh/uv/install.sh | sh"
exit 1
fi
# --------------------------------------------------------------------------
# 1. venv
# --------------------------------------------------------------------------
if [[ ! -x "$VENV/bin/python" ]]; then
log "creating venv at $VENV (python $PYTHON_VERSION)"
uv venv --python "$PYTHON_VERSION" --seed "$VENV"
fi
PY="$VENV/bin/python"
# --------------------------------------------------------------------------
# 2. dependencies
# --------------------------------------------------------------------------
if [[ ! -f "$MARKERS/deps" ]]; then
log "installing CUDA vLLM + lm-eval + evalchemy runtime deps"
# Driver-aware: on driver 535 (no CUDA 12.6+) invoke with
# TORCH_PRE="torch==2.6.0 torchvision==0.21.0 --index-url https://download.pytorch.org/whl/cu124"
# VLLM_SPEC="vllm==0.8.5"
# vLLM >=0.9 ships cu126/cu130 wheels that will NOT run on driver 535.
if [[ -n "${TORCH_PRE:-}" ]]; then
log "pre-installing pinned torch: $TORCH_PRE"
UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" $TORCH_PRE
fi
UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" \
"${VLLM_SPEC:-vllm>=0.10}" \
"transformers>=4.51,<5" \
"datasets>=3.6,<4.0.0" \
"fsspec[http]>=2023.1.0" \
"accelerate>=1.8" \
"safetensors" \
"sentencepiece" \
"$LM_EVAL_REQ" \
"sympy>=1.12.1,<1.14" \
"antlr4-python3-runtime==4.11" \
"bespokelabs-curator==0.1.16" \
"sqlalchemy" \
"fire" \
"pandas" \
"numpy"
touch "$MARKERS/deps"
else
log "deps marker present; skipping install"
fi
# Guard: datasets v4 silently breaks every LiveCodeBench task.
"$PY" - <<'PYCHECK'
import sys
from importlib.metadata import version
dv = version("datasets")
major = int(dv.split(".")[0])
if major >= 4:
sys.exit(
f"datasets {dv} installed but LiveCodeBench requires <4.0.0 "
"(v4 dropped dataset-script support). Reinstall with 'datasets>=3.6,<4.0.0'."
)
print(f"[bootstrap] datasets {dv} OK")
PYCHECK
# --------------------------------------------------------------------------
# 3. evalchemy checkout, pinned + patched
# --------------------------------------------------------------------------
if [[ ! -d "$EVALCHEMY_DIR/.git" ]]; then
log "cloning evalchemy -> $EVALCHEMY_DIR"
git clone "$EVALCHEMY_REPO" "$EVALCHEMY_DIR"
fi
log "pinning evalchemy to $EVALCHEMY_COMMIT and reverting previous patches"
git -C "$EVALCHEMY_DIR" fetch --all --quiet || true
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT"
# Hard-reset the files the patcher rewrites so patching is deterministic.
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" -- eval/
log "applying GPU patches"
"$PY" "$HERE/patch_evalchemy_gpu.py" "$EVALCHEMY_DIR"
# --------------------------------------------------------------------------
# 4. activate snippet
# --------------------------------------------------------------------------
cat > "$ROOT/activate.sh" <<ACTEOF
# source this before running gpu_eval_driver.py / gpu_eval_monitor.py by hand
export MARIN_GPU_EVAL_ROOT="$ROOT"
export EVALCHEMY_DIR="$EVALCHEMY_DIR"
export VIRTUAL_ENV="$VENV"
export PATH="$VENV/bin:\$PATH"
# HF / dataset behaviour required by LiveCodeBench and the math suite
export HF_DATASETS_TRUST_REMOTE_CODE=1
export HF_HOME="$CACHE/huggingface"
export HF_HUB_CACHE="$CACHE/huggingface/hub"
export HF_DATASETS_CACHE="$CACHE/datasets"
export HF_ALLOW_CODE_EVAL=1
# vLLM / torch caches kept off \$HOME so concurrent runs do not fight
export VLLM_CACHE_ROOT="$CACHE/vllm"
export VLLM_CONFIG_ROOT="$CACHE/vllm"
export TORCHINDUCTOR_CACHE_DIR="$CACHE/torchinductor"
export TRITON_CACHE_DIR="$CACHE/triton"
export TOKENIZERS_PARALLELISM=false
# Supply your own:
# export HF_TOKEN=hf_xxx # needed for cais/hle and Idavidrein/gpqa
ACTEOF
log "done"
log ""
log " source $ROOT/activate.sh"
log " export HF_TOKEN=... # only needed for the science suite"
log " python $HERE/gpu_eval_driver.py --help"
log ""
log "evalchemy: $EVALCHEMY_DIR @ $(git -C "$EVALCHEMY_DIR" rev-parse --short HEAD)"
log "python: $PY"
|