Download gpu-sft/scripts/gpu_eval/bootstrap_gpu_eval.sh from fzzhang/svd-code: direct link, hf CLI and curl.
- Browser
- Download file 8.19 kB
-
https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/scripts/gpu_eval/bootstrap_gpu_eval.sh
- Command line
-
hf download hf://fzzhang/svd-code/gpu-sft/scripts/gpu_eval/bootstrap_gpu_eval.sh
-
curl -L -o bootstrap_gpu_eval.sh https://huggingface.co/fzzhang/svd-code/resolve/main/gpu-sft/scripts/gpu_eval/bootstrap_gpu_eval.sh
8.19 kB
| # Copyright The Marin Authors | |
| # SPDX-License-Identifier: Apache-2.0 | |
| # | |
| # Bootstrap a self-contained CUDA evalchemy + vLLM environment for evaluating | |
| # LOCAL HF-format checkpoints on a single GPU box. No GCS, no Iris, no TPU. | |
| # | |
| # ./bootstrap_gpu_eval.sh /opt/marin-gpu-eval | |
| # | |
| # Idempotent: re-running skips completed phases. Delete the phase markers under | |
| # $ROOT/.markers to force a phase to re-run. | |
| # | |
| # --------------------------------------------------------------------------- | |
| # WHY THESE PINS | |
| # --------------------------------------------------------------------------- | |
| # evalchemy teetone/evalchemy @ 7f24168 (2026-03-31) | |
| # This is the pin the TPU pipeline used to produce the current results.md. | |
| # Use it if you need numbers comparable to the existing tables. | |
| # 460b022 (2026-05-07) is the newer pin carried on the marin-agents | |
| # `evalfixhfauth` branch; it skips questions whose prompt >= max_model_len | |
| # instead of crashing, and adds the TTC task variants. Set | |
| # EVALCHEMY_COMMIT=460b022 if you hit prompt-length crashes. | |
| # | |
| # lm-eval stanford-crfm/lm-evaluation-harness @ d5e3391f22cde186c827674d5c3ec7c5f4fe0cab | |
| # evalchemy is NOT pip-installable; it is a source checkout that imports | |
| # lm_eval. This exact fork+commit is what Marin pins. Upstream lm-eval 0.5.x | |
| # moved eval_logger and broke evalchemy's eval_tracker import. | |
| # | |
| # datasets >=3.6,<4.0.0 HARD upper bound. | |
| # datasets v4 removed dataset-script support. LiveCodeBench loads | |
| # `livecodebench/code_generation_lite` with trust_remote_code=True, which is | |
| # a dataset script. On datasets>=4 every LCB task fails at load time. | |
| # | |
| # transformers >=4.51,<5 Qwen3 architecture support landed in 4.51.0. | |
| # | |
| # vllm >=0.10 CUDA wheel. Do NOT install tpu-inference / the | |
| # `vllm` dependency group from lib/marin/pyproject.toml: that group pins | |
| # jax==0.10.0 + libtpu==0.0.40 + tpu-inference==0.22.1 and is TPU-only. | |
| # | |
| # --------------------------------------------------------------------------- | |
| # HUGGING FACE AUTH -- what the `evalfixhfauth` branch actually does | |
| # --------------------------------------------------------------------------- | |
| # The marin-agents branch name is stale/aspirational: no commit on it touches HF | |
| # authentication. The real HF-related fixes on that branch are three, and a | |
| # local user must reproduce all three: | |
| # | |
| # 1. HF_DATASETS_TRUST_REMOTE_CODE=1 in the eval process env (LiveCodeBench | |
| # needs it). This script exports it in the generated activate snippet, and | |
| # gpu_eval_driver.py sets it on every subprocess. | |
| # 2. datasets<4.0.0 pin (above). | |
| # 3. HF_TOKEN supplied as a process env var at submit time -- it is never in | |
| # source. On TPU it was threaded through `iris job run -e HF_TOKEN ...`. | |
| # LOCALLY YOU MUST DO THIS INSTEAD: | |
| # huggingface-cli login # writes ~/.cache/huggingface/token | |
| # or export HF_TOKEN=hf_xxx # per shell / systemd unit | |
| # Required for gated datasets: cais/hle (HLE) and Idavidrein/gpqa | |
| # (GPQADiamond) -- you must ALSO click "Agree" on those dataset pages with | |
| # the same account. The math suite (MATH500, OlympiadBench, AIME24/25/26, | |
| # HMMT) and the code suite (LiveCodeBench*) are public and need no token, | |
| # though an anonymous box will hit HF rate limits during a 10-seed sweep. | |
| # | |
| set -euo pipefail | |
| ROOT="${1:-/opt/marin-gpu-eval}" | |
| EVALCHEMY_REPO="${EVALCHEMY_REPO:-https://github.com/teetone/evalchemy.git}" | |
| EVALCHEMY_COMMIT="${EVALCHEMY_COMMIT:-7f24168}" | |
| PYTHON_VERSION="${PYTHON_VERSION:-3.11}" | |
| LM_EVAL_REQ="lm-eval[math]@git+https://github.com/stanford-crfm/lm-evaluation-harness@d5e3391f22cde186c827674d5c3ec7c5f4fe0cab" | |
| VENV="$ROOT/venv" | |
| EVALCHEMY_DIR="$ROOT/evalchemy" | |
| CACHE="$ROOT/cache" | |
| MARKERS="$ROOT/.markers" | |
| HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" | |
| mkdir -p "$ROOT" "$CACHE" "$MARKERS" | |
| log() { printf '[bootstrap] %s\n' "$*" >&2; } | |
| if ! command -v uv >/dev/null 2>&1; then | |
| log "uv not found on PATH; install it with: curl -LsSf https://astral.sh/uv/install.sh | sh" | |
| exit 1 | |
| fi | |
| # -------------------------------------------------------------------------- | |
| # 1. venv | |
| # -------------------------------------------------------------------------- | |
| if [[ ! -x "$VENV/bin/python" ]]; then | |
| log "creating venv at $VENV (python $PYTHON_VERSION)" | |
| uv venv --python "$PYTHON_VERSION" --seed "$VENV" | |
| fi | |
| PY="$VENV/bin/python" | |
| # -------------------------------------------------------------------------- | |
| # 2. dependencies | |
| # -------------------------------------------------------------------------- | |
| if [[ ! -f "$MARKERS/deps" ]]; then | |
| log "installing CUDA vLLM + lm-eval + evalchemy runtime deps" | |
| # Driver-aware: on driver 535 (no CUDA 12.6+) invoke with | |
| # TORCH_PRE="torch==2.6.0 torchvision==0.21.0 --index-url https://download.pytorch.org/whl/cu124" | |
| # VLLM_SPEC="vllm==0.8.5" | |
| # vLLM >=0.9 ships cu126/cu130 wheels that will NOT run on driver 535. | |
| if [[ -n "${TORCH_PRE:-}" ]]; then | |
| log "pre-installing pinned torch: $TORCH_PRE" | |
| UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" $TORCH_PRE | |
| fi | |
| UV_CACHE_DIR="$CACHE/uv" uv pip install --python "$PY" \ | |
| "${VLLM_SPEC:-vllm>=0.10}" \ | |
| "transformers>=4.51,<5" \ | |
| "datasets>=3.6,<4.0.0" \ | |
| "fsspec[http]>=2023.1.0" \ | |
| "accelerate>=1.8" \ | |
| "safetensors" \ | |
| "sentencepiece" \ | |
| "$LM_EVAL_REQ" \ | |
| "sympy>=1.12.1,<1.14" \ | |
| "antlr4-python3-runtime==4.11" \ | |
| "bespokelabs-curator==0.1.16" \ | |
| "sqlalchemy" \ | |
| "fire" \ | |
| "pandas" \ | |
| "numpy" | |
| touch "$MARKERS/deps" | |
| else | |
| log "deps marker present; skipping install" | |
| fi | |
| # Guard: datasets v4 silently breaks every LiveCodeBench task. | |
| "$PY" - <<'PYCHECK' | |
| import sys | |
| from importlib.metadata import version | |
| dv = version("datasets") | |
| major = int(dv.split(".")[0]) | |
| if major >= 4: | |
| sys.exit( | |
| f"datasets {dv} installed but LiveCodeBench requires <4.0.0 " | |
| "(v4 dropped dataset-script support). Reinstall with 'datasets>=3.6,<4.0.0'." | |
| ) | |
| print(f"[bootstrap] datasets {dv} OK") | |
| PYCHECK | |
| # -------------------------------------------------------------------------- | |
| # 3. evalchemy checkout, pinned + patched | |
| # -------------------------------------------------------------------------- | |
| if [[ ! -d "$EVALCHEMY_DIR/.git" ]]; then | |
| log "cloning evalchemy -> $EVALCHEMY_DIR" | |
| git clone "$EVALCHEMY_REPO" "$EVALCHEMY_DIR" | |
| fi | |
| log "pinning evalchemy to $EVALCHEMY_COMMIT and reverting previous patches" | |
| git -C "$EVALCHEMY_DIR" fetch --all --quiet || true | |
| git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" | |
| # Hard-reset the files the patcher rewrites so patching is deterministic. | |
| git -C "$EVALCHEMY_DIR" checkout --quiet "$EVALCHEMY_COMMIT" -- eval/ | |
| log "applying GPU patches" | |
| "$PY" "$HERE/patch_evalchemy_gpu.py" "$EVALCHEMY_DIR" | |
| # -------------------------------------------------------------------------- | |
| # 4. activate snippet | |
| # -------------------------------------------------------------------------- | |
| cat > "$ROOT/activate.sh" <<ACTEOF | |
| # source this before running gpu_eval_driver.py / gpu_eval_monitor.py by hand | |
| export MARIN_GPU_EVAL_ROOT="$ROOT" | |
| export EVALCHEMY_DIR="$EVALCHEMY_DIR" | |
| export VIRTUAL_ENV="$VENV" | |
| export PATH="$VENV/bin:\$PATH" | |
| # HF / dataset behaviour required by LiveCodeBench and the math suite | |
| export HF_DATASETS_TRUST_REMOTE_CODE=1 | |
| export HF_HOME="$CACHE/huggingface" | |
| export HF_HUB_CACHE="$CACHE/huggingface/hub" | |
| export HF_DATASETS_CACHE="$CACHE/datasets" | |
| export HF_ALLOW_CODE_EVAL=1 | |
| # vLLM / torch caches kept off \$HOME so concurrent runs do not fight | |
| export VLLM_CACHE_ROOT="$CACHE/vllm" | |
| export VLLM_CONFIG_ROOT="$CACHE/vllm" | |
| export TORCHINDUCTOR_CACHE_DIR="$CACHE/torchinductor" | |
| export TRITON_CACHE_DIR="$CACHE/triton" | |
| export TOKENIZERS_PARALLELISM=false | |
| # Supply your own: | |
| # export HF_TOKEN=hf_xxx # needed for cais/hle and Idavidrein/gpqa | |
| ACTEOF | |
| log "done" | |
| log "" | |
| log " source $ROOT/activate.sh" | |
| log " export HF_TOKEN=... # only needed for the science suite" | |
| log " python $HERE/gpu_eval_driver.py --help" | |
| log "" | |
| log "evalchemy: $EVALCHEMY_DIR @ $(git -C "$EVALCHEMY_DIR" rev-parse --short HEAD)" | |
| log "python: $PY" | |