svd-code / gpu-sft /scripts /gpu_eval /bootstrap_gpu_eval.sh
fzzhang's picture
Upload folder using huggingface_hub
58258b8 verified
Raw History Blame Contribute Delete
8.19 kB
#!/usr/bin/env bash
# Copyright The Marin Authors
# SPDX-License-Identifier: Apache-2.0
#
# Bootstrap a self-contained CUDA evalchemy + vLLM environment for evaluating
# LOCAL HF-format checkpoints on a single GPU box. No GCS, no Iris, no TPU.
#
# ./bootstrap_gpu_eval.sh /opt/marin-gpu-eval
#
# Idempotent: re-running skips completed phases. Delete the phase markers under
# $ROOT/.markers to force a phase to re-run.
#
# ---------------------------------------------------------------------------
# WHY THESE PINS
# ---------------------------------------------------------------------------
# evalchemy teetone/evalchemy @ 7f24168 (2026-03-31)
# This is the pin the TPU pipeline used to produce the current results.md.
# Use it if you need numbers comparable to the existing tables.
# 460b022 (2026-05-07) is the newer pin carried on the marin-agents
# `evalfixhfauth` branch; it skips questions whose prompt >= max_model_len
# instead of crashing, and adds the TTC task variants. Set
# EVALCHEMY_COMMIT=460b022 if you hit prompt-length crashes.
#
# lm-eval stanford-crfm/lm-evaluation-harness @ d5e3391f22cde186c827674d5c3ec7c5f4fe0cab
# evalchemy is NOT pip-installable; it is a source checkout that imports
# lm_eval. This exact fork+commit is what Marin pins. Upstream lm-eval 0.5.x
# moved eval_logger and broke evalchemy's eval_tracker import.
#
# datasets >=3.6,<4.0.0 HARD upper bound.
# datasets v4 removed dataset-script support. LiveCodeBench loads
# `livecodebench/code_generation_lite` with trust_remote_code=True, which is
# a dataset script. On datasets>=4 every LCB task fails at load time.
#
# transformers >=4.51,<5 Qwen3 architecture support landed in 4.51.0.
#
# vllm >=0.10 CUDA wheel. Do NOT install tpu-inference / the
# `vllm` dependency group from lib/marin/pyproject.toml: that group pins
# jax==0.10.0 + libtpu==0.0.40 + tpu-inference==0.22.1 and is TPU-only.
#
# ---------------------------------------------------------------------------
# HUGGING FACE AUTH -- what the `evalfixhfauth` branch actually does
# ---------------------------------------------------------------------------
# The marin-agents branch name is stale/aspirational: no commit on it touches HF
# authentication. The real HF-related fixes on that branch are three, and a
# local user must reproduce all three:
#
# 1. HF_DATASETS_TRUST_REMOTE_CODE=1 in the eval process env (LiveCodeBench
# needs it). This script exports it in the generated activate snippet, and
# gpu_eval_driver.py sets it on every subprocess.
# 2. datasets<4.0.0 pin (above).
# 3. HF_TOKEN supplied as a process env var at submit time -- it is never in
# source. On TPU it was threaded through `iris job run -e HF_TOKEN ...`.
# LOCALLY YOU MUST DO THIS INSTEAD:
# huggingface-cli login # writes ~/.cache/huggingface/token
# or export HF_TOKEN=hf_xxx # per shell / systemd unit
# Required for gated datasets: cais/hle (HLE) and Idavidrein/gpqa
# (GPQADiamond) -- you must ALSO click "Agree" on those dataset pages with
# the same account. The math suite (MATH500, OlympiadBench, AIME24/25/26,
# HMMT) and the code suite (LiveCodeBench*) are public and need no token,
# though an anonymous box will hit HF rate limits during a 10-seed sweep.
#
set -euo pipefail
ROOT="${1:-/opt/marin-gpu-eval}"
EVALCHEMY_REPO="${EVALCHEMY_REPO:-https://github.com/teetone/evalchemy.git}"
EVALCHEMY_COMMIT="${EVALCHEMY_COMMIT:-7f24168}"
PYTHON_VERSION="${PYTHON_VERSION:-3.11}"
LM_EVAL_REQ="lm-eval[math]@git+https://github.com/stanford-crfm/lm-evaluation-harness@d5e3391f22cde186c827674d5c3ec7c5f4fe0cab"
VENV="$ROOT/venv"
EVALCHEMY_DIR="$ROOT/evalchemy"
CACHE="$ROOT/cache"
MARKERS="$ROOT/.markers"
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
mkdir -p "$ROOT" "$CACHE" "$MARKERS"
log() { printf '[bootstrap] %s\n' "$*" >&2; }
if ! command -v uv >/dev/null 2>&1; then
log "uv not found on PATH; install it with: curl -LsSf https://astral.sh/uv/install.sh | sh"
exit 1
fi
# --------------------------------------------------------------------------
# 1. venv
# --------------------------------------------------------------------------
if [[ ! -x "$VENV/bin/python" ]]; then
log "creating venv at $VENV (python $PYTHON_VERSION)"
uv venv --python "$PYTHON_VERSION" --seed "$VENV"
fi
PY="$VENV/bin/python"
# --------------------------------------------------------------------------
# 2. dependencies
# --------------------------------------------------------------------------
if [[ ! -f "$MARKERS/deps" ]]; then
log "installing CUDA vLLM + lm-eval + evalchemy runtime deps"
# Driver-aware: on driver 535 (no CUDA 12.6+) invoke with
# TORCH_PRE="torch==2.6.0 torchvision==0.21.0 --index-url https://download.pytorch.org/whl/cu124"
# VLLM_SPEC="vllm==0.8.5"
# vLLM >=0.9 ships cu126/cu130 wheels that will NOT run on driver 535.
if [[ -n "${TORCH_PRE:-}" ]]; then
log "pre-installing pinned torch: $TORCH_PRE"
UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" $TORCH_PRE
fi
UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" \
"${VLLM_SPEC:-vllm>=0.10}" \
"transformers>=4.51,<5" \
"datasets>=3.6,<4.0.0" \
"fsspec[http]>=2023.1.0" \
"accelerate>=1.8" \
"safetensors" \
"sentencepiece" \
"$LM_EVAL_REQ" \
"sympy>=1.12.1,<1.14" \
"antlr4-python3-runtime==4.11" \
"bespokelabs-curator==0.1.16" \
"sqlalchemy" \
"fire" \
"pandas" \
"numpy"
touch "$MARKERS/deps"
else
log "deps marker present; skipping install"
fi
# Guard: datasets v4 silently breaks every LiveCodeBench task.
"$PY" - <<'PYCHECK'
import sys
from importlib.metadata import version
dv = version("datasets")
major = int(dv.split(".")[0])
if major >= 4:
sys.exit(
f"datasets {dv} installed but LiveCodeBench requires <4.0.0 "
"(v4 dropped dataset-script support). Reinstall with 'datasets>=3.6,<4.0.0'."
)
print(f"[bootstrap] datasets {dv} OK")
PYCHECK
# --------------------------------------------------------------------------
# 3. evalchemy checkout, pinned + patched
# --------------------------------------------------------------------------
if [[ ! -d "$EVALCHEMY_DIR/.git" ]]; then
log "cloning evalchemy -> $EVALCHEMY_DIR"
git clone "$EVALCHEMY_REPO" "$EVALCHEMY_DIR"
fi
log "pinning evalchemy to $EVALCHEMY_COMMIT and reverting previous patches"
git -C "$EVALCHEMY_DIR" fetch --all --quiet || true
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT"
# Hard-reset the files the patcher rewrites so patching is deterministic.
git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" -- eval/
log "applying GPU patches"
"$PY" "$HERE/patch_evalchemy_gpu.py" "$EVALCHEMY_DIR"
# --------------------------------------------------------------------------
# 4. activate snippet
# --------------------------------------------------------------------------
cat > "$ROOT/activate.sh" <<ACTEOF
# source this before running gpu_eval_driver.py / gpu_eval_monitor.py by hand
export MARIN_GPU_EVAL_ROOT="$ROOT"
export EVALCHEMY_DIR="$EVALCHEMY_DIR"
export VIRTUAL_ENV="$VENV"
export PATH="$VENV/bin:\$PATH"
# HF / dataset behaviour required by LiveCodeBench and the math suite
export HF_DATASETS_TRUST_REMOTE_CODE=1
export HF_HOME="$CACHE/huggingface"
export HF_HUB_CACHE="$CACHE/huggingface/hub"
export HF_DATASETS_CACHE="$CACHE/datasets"
export HF_ALLOW_CODE_EVAL=1
# vLLM / torch caches kept off \$HOME so concurrent runs do not fight
export VLLM_CACHE_ROOT="$CACHE/vllm"
export VLLM_CONFIG_ROOT="$CACHE/vllm"
export TORCHINDUCTOR_CACHE_DIR="$CACHE/torchinductor"
export TRITON_CACHE_DIR="$CACHE/triton"
export TOKENIZERS_PARALLELISM=false
# Supply your own:
# export HF_TOKEN=hf_xxx # needed for cais/hle and Idavidrein/gpqa
ACTEOF
log "done"
log ""
log " source $ROOT/activate.sh"
log " export HF_TOKEN=... # only needed for the science suite"
log " python $HERE/gpu_eval_driver.py --help"
log ""
log "evalchemy: $EVALCHEMY_DIR @ $(git -C "$EVALCHEMY_DIR" rev-parse --short HEAD)"
log "python: $PY"