#!/usr/bin/env bash # Copyright The Marin Authors # SPDX-License-Identifier: Apache-2.0 # # Bootstrap a self-contained CUDA evalchemy + vLLM environment for evaluating # LOCAL HF-format checkpoints on a single GPU box. No GCS, no Iris, no TPU. # # ./bootstrap_gpu_eval.sh /opt/marin-gpu-eval # # Idempotent: re-running skips completed phases. Delete the phase markers under # $ROOT/.markers to force a phase to re-run. # # --------------------------------------------------------------------------- # WHY THESE PINS # --------------------------------------------------------------------------- # evalchemy teetone/evalchemy @ 7f24168 (2026-03-31) # This is the pin the TPU pipeline used to produce the current results.md. # Use it if you need numbers comparable to the existing tables. # 460b022 (2026-05-07) is the newer pin carried on the marin-agents # `evalfixhfauth` branch; it skips questions whose prompt >= max_model_len # instead of crashing, and adds the TTC task variants. Set # EVALCHEMY_COMMIT=460b022 if you hit prompt-length crashes. # # lm-eval stanford-crfm/lm-evaluation-harness @ d5e3391f22cde186c827674d5c3ec7c5f4fe0cab # evalchemy is NOT pip-installable; it is a source checkout that imports # lm_eval. This exact fork+commit is what Marin pins. Upstream lm-eval 0.5.x # moved eval_logger and broke evalchemy's eval_tracker import. # # datasets >=3.6,<4.0.0 HARD upper bound. # datasets v4 removed dataset-script support. LiveCodeBench loads # `livecodebench/code_generation_lite` with trust_remote_code=True, which is # a dataset script. On datasets>=4 every LCB task fails at load time. # # transformers >=4.51,<5 Qwen3 architecture support landed in 4.51.0. # # vllm >=0.10 CUDA wheel. Do NOT install tpu-inference / the # `vllm` dependency group from lib/marin/pyproject.toml: that group pins # jax==0.10.0 + libtpu==0.0.40 + tpu-inference==0.22.1 and is TPU-only. # # --------------------------------------------------------------------------- # HUGGING FACE AUTH -- what the `evalfixhfauth` branch actually does # --------------------------------------------------------------------------- # The marin-agents branch name is stale/aspirational: no commit on it touches HF # authentication. The real HF-related fixes on that branch are three, and a # local user must reproduce all three: # # 1. HF_DATASETS_TRUST_REMOTE_CODE=1 in the eval process env (LiveCodeBench # needs it). This script exports it in the generated activate snippet, and # gpu_eval_driver.py sets it on every subprocess. # 2. datasets<4.0.0 pin (above). # 3. HF_TOKEN supplied as a process env var at submit time -- it is never in # source. On TPU it was threaded through `iris job run -e HF_TOKEN ...`. # LOCALLY YOU MUST DO THIS INSTEAD: # huggingface-cli login # writes ~/.cache/huggingface/token # or export HF_TOKEN=hf_xxx # per shell / systemd unit # Required for gated datasets: cais/hle (HLE) and Idavidrein/gpqa # (GPQADiamond) -- you must ALSO click "Agree" on those dataset pages with # the same account. The math suite (MATH500, OlympiadBench, AIME24/25/26, # HMMT) and the code suite (LiveCodeBench*) are public and need no token, # though an anonymous box will hit HF rate limits during a 10-seed sweep. # set -euo pipefail ROOT="${1:-/opt/marin-gpu-eval}" EVALCHEMY_REPO="${EVALCHEMY_REPO:-https://github.com/teetone/evalchemy.git}" EVALCHEMY_COMMIT="${EVALCHEMY_COMMIT:-7f24168}" PYTHON_VERSION="${PYTHON_VERSION:-3.11}" LM_EVAL_REQ="lm-eval[math]@git+https://github.com/stanford-crfm/lm-evaluation-harness@d5e3391f22cde186c827674d5c3ec7c5f4fe0cab" VENV="$ROOT/venv" EVALCHEMY_DIR="$ROOT/evalchemy" CACHE="$ROOT/cache" MARKERS="$ROOT/.markers" HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" mkdir -p "$ROOT" "$CACHE" "$MARKERS" log() { printf '[bootstrap] %s\n' "$*" >&2; } if ! command -v uv >/dev/null 2>&1; then log "uv not found on PATH; install it with: curl -LsSf https://astral.sh/uv/install.sh | sh" exit 1 fi # -------------------------------------------------------------------------- # 1. venv # -------------------------------------------------------------------------- if [[ ! -x "$VENV/bin/python" ]]; then log "creating venv at $VENV (python $PYTHON_VERSION)" uv venv --python "$PYTHON_VERSION" --seed "$VENV" fi PY="$VENV/bin/python" # -------------------------------------------------------------------------- # 2. dependencies # -------------------------------------------------------------------------- if [[ ! -f "$MARKERS/deps" ]]; then log "installing CUDA vLLM + lm-eval + evalchemy runtime deps" # Driver-aware: on driver 535 (no CUDA 12.6+) invoke with # TORCH_PRE="torch==2.6.0 torchvision==0.21.0 --index-url https://download.pytorch.org/whl/cu124" # VLLM_SPEC="vllm==0.8.5" # vLLM >=0.9 ships cu126/cu130 wheels that will NOT run on driver 535. if [[ -n "${TORCH_PRE:-}" ]]; then log "pre-installing pinned torch: $TORCH_PRE" UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" $TORCH_PRE fi UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" \ "${VLLM_SPEC:-vllm>=0.10}" \ "transformers>=4.51,<5" \ "datasets>=3.6,<4.0.0" \ "fsspec[http]>=2023.1.0" \ "accelerate>=1.8" \ "safetensors" \ "sentencepiece" \ "$LM_EVAL_REQ" \ "sympy>=1.12.1,<1.14" \ "antlr4-python3-runtime==4.11" \ "bespokelabs-curator==0.1.16" \ "sqlalchemy" \ "fire" \ "pandas" \ "numpy" touch "$MARKERS/deps" else log "deps marker present; skipping install" fi # Guard: datasets v4 silently breaks every LiveCodeBench task. "$PY" - <<'PYCHECK' import sys from importlib.metadata import version dv = version("datasets") major = int(dv.split(".")[0]) if major >= 4: sys.exit( f"datasets {dv} installed but LiveCodeBench requires <4.0.0 " "(v4 dropped dataset-script support). Reinstall with 'datasets>=3.6,<4.0.0'." ) print(f"[bootstrap] datasets {dv} OK") PYCHECK # -------------------------------------------------------------------------- # 3. evalchemy checkout, pinned + patched # -------------------------------------------------------------------------- if [[ ! -d "$EVALCHEMY_DIR/.git" ]]; then log "cloning evalchemy -> $EVALCHEMY_DIR" git clone "$EVALCHEMY_REPO" "$EVALCHEMY_DIR" fi log "pinning evalchemy to $EVALCHEMY_COMMIT and reverting previous patches" git -C "$EVALCHEMY_DIR" fetch --all --quiet || true git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" # Hard-reset the files the patcher rewrites so patching is deterministic. git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" -- eval/ log "applying GPU patches" "$PY" "$HERE/patch_evalchemy_gpu.py" "$EVALCHEMY_DIR" # -------------------------------------------------------------------------- # 4. activate snippet # -------------------------------------------------------------------------- cat > "$ROOT/activate.sh" <