#!/usr/bin/env bash # Run the Qwen3-8B baseline + the completed SFT checkpoints through evalchemy, sequentially, # writing results to HDFS so a machine kill doesn't lose completed evals. # # source /opt/tiger/lmc_muon/gpu-eval/activate.sh # (this script also sources it) # nohup bash gpu-sft/scripts/gpu_eval/run_all_evals.sh \ # > /mnt/hdfs/fangzhao_writable/marin-eval/eval_all.log 2>&1 & # # Each eval uses the full node (tp defaults to #GPUs). Jobs run one at a time. Results go to # $RESULTS on HDFS; a completed task's JSON is durable immediately, so a kill only costs the # in-flight task (re-run the script — the driver skips finished tasks unless --force). source /opt/tiger/lmc_muon/gpu-eval/activate.sh 2>/dev/null || true set -uo pipefail DRIVER=/opt/tiger/lmc_muon/self-verified-distillation/gpu-sft/scripts/gpu_eval/gpu_eval_driver.py RESULTS="${RESULTS:-/mnt/hdfs/fangzhao_writable/marin-eval/results}" HDFS_SFT=/mnt/hdfs/fangzhao_writable/marin-sft MODELS=/opt/tiger/lmc_muon/models PROFILE="${PROFILE:-h100-80g}" mkdir -p "$RESULTS" "$MODELS" run() { # run echo "=========== eval $2 / $3 $(date -u) ===========" python "$DRIVER" --checkpoint "$1" --experiment "$2" --suite "$3" \ --gpu-profile "$PROFILE" --results-root "$RESULTS" \ || echo "!! $2/$3 FAILED (continuing to next)" } # Stage the completed checkpoints from HDFS -> local, keeping a 'step-1999' leaf so the driver # parses the step number (16 GB each, ~30 s at 574 MB/s). for e in coding_depth4v3_nofilter hs_competition_nofilter; do DST="$MODELS/$e/step-1999" if [ ! -f "$DST/config.json" ]; then echo "staging $e/hf/step-1999 from HDFS -> $DST" mkdir -p "$MODELS/$e" && cp -r "$HDFS_SFT/$e/hf/step-1999" "$DST" fi done if [ "${PARALLEL:-0}" = "1" ]; then LOGDIR="$RESULTS/_logs"; mkdir -p "$LOGDIR" echo "PARALLEL mode: 4 jobs x tp=2 across GPU pairs (per-job logs in $LOGDIR)" CUDA_VISIBLE_DEVICES=0,1 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_math.log" 2>&1 & CUDA_VISIBLE_DEVICES=2,3 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_code.log" 2>&1 & CUDA_VISIBLE_DEVICES=4,5 python "$DRIVER" --checkpoint "$MODELS/hs_competition_nofilter/step-1999" --experiment hs_competition_nofilter --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/hs_math.log" 2>&1 & CUDA_VISIBLE_DEVICES=6,7 python "$DRIVER" --checkpoint "$MODELS/coding_depth4v3_nofilter/step-1999" --experiment coding_depth4v3_nofilter --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/coding_code.log" 2>&1 & wait echo "=========== ALL PARALLEL EVALS DONE $(date -u) ===========" echo "results -> $RESULTS ; per-job logs -> $LOGDIR" exit 0 fi # 1) baseline (the denominator for every (+delta)) on both suites run "$MODELS/Qwen3-8B" qwen3_8b_base math run "$MODELS/Qwen3-8B" qwen3_8b_base code # 2) the two completed checkpoints on their matching suites run "$MODELS/hs_competition_nofilter/step-1999" hs_competition_nofilter math run "$MODELS/coding_depth4v3_nofilter/step-1999" coding_depth4v3_nofilter code echo "=========== ALL EVALS DONE $(date -u) ===========" echo "results -> $RESULTS"