svd-code / gpu-sft /scripts /gpu_eval /run_all_evals.sh
fzzhang's picture
Upload folder using huggingface_hub
58258b8 verified
Raw History Blame Contribute Delete
3.59 kB
#!/usr/bin/env bash
# Run the Qwen3-8B baseline + the completed SFT checkpoints through evalchemy, sequentially,
# writing results to HDFS so a machine kill doesn't lose completed evals.
#
# source /opt/tiger/lmc_muon/gpu-eval/activate.sh # (this script also sources it)
# nohup bash gpu-sft/scripts/gpu_eval/run_all_evals.sh \
# > /mnt/hdfs/fangzhao_writable/marin-eval/eval_all.log 2>&1 &
#
# Each eval uses the full node (tp defaults to #GPUs). Jobs run one at a time. Results go to
# $RESULTS on HDFS; a completed task's JSON is durable immediately, so a kill only costs the
# in-flight task (re-run the script — the driver skips finished tasks unless --force).
source /opt/tiger/lmc_muon/gpu-eval/activate.sh 2>/dev/null || true
set -uo pipefail
DRIVER=/opt/tiger/lmc_muon/self-verified-distillation/gpu-sft/scripts/gpu_eval/gpu_eval_driver.py
RESULTS="${RESULTS:-/mnt/hdfs/fangzhao_writable/marin-eval/results}"
HDFS_SFT=/mnt/hdfs/fangzhao_writable/marin-sft
MODELS=/opt/tiger/lmc_muon/models
PROFILE="${PROFILE:-h100-80g}"
mkdir -p "$RESULTS" "$MODELS"
run() { # run <checkpoint_dir> <experiment> <suite>
echo "=========== eval $2 / $3 $(date -u) ==========="
python "$DRIVER" --checkpoint "$1" --experiment "$2" --suite "$3" \
--gpu-profile "$PROFILE" --results-root "$RESULTS" \
|| echo "!! $2/$3 FAILED (continuing to next)"
}
# Stage the completed checkpoints from HDFS -> local, keeping a 'step-1999' leaf so the driver
# parses the step number (16 GB each, ~30 s at 574 MB/s).
for e in coding_depth4v3_nofilter hs_competition_nofilter; do
DST="$MODELS/$e/step-1999"
if [ ! -f "$DST/config.json" ]; then
echo "staging $e/hf/step-1999 from HDFS -> $DST"
mkdir -p "$MODELS/$e" && cp -r "$HDFS_SFT/$e/hf/step-1999" "$DST"
fi
done
if [ "${PARALLEL:-0}" = "1" ]; then
LOGDIR="$RESULTS/_logs"; mkdir -p "$LOGDIR"
echo "PARALLEL mode: 4 jobs x tp=2 across GPU pairs (per-job logs in $LOGDIR)"
CUDA_VISIBLE_DEVICES=0,1 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_math.log" 2>&1 &
CUDA_VISIBLE_DEVICES=2,3 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_code.log" 2>&1 &
CUDA_VISIBLE_DEVICES=4,5 python "$DRIVER" --checkpoint "$MODELS/hs_competition_nofilter/step-1999" --experiment hs_competition_nofilter --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/hs_math.log" 2>&1 &
CUDA_VISIBLE_DEVICES=6,7 python "$DRIVER" --checkpoint "$MODELS/coding_depth4v3_nofilter/step-1999" --experiment coding_depth4v3_nofilter --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/coding_code.log" 2>&1 &
wait
echo "=========== ALL PARALLEL EVALS DONE $(date -u) ==========="
echo "results -> $RESULTS ; per-job logs -> $LOGDIR"
exit 0
fi
# 1) baseline (the denominator for every (+delta)) on both suites
run "$MODELS/Qwen3-8B" qwen3_8b_base math
run "$MODELS/Qwen3-8B" qwen3_8b_base code
# 2) the two completed checkpoints on their matching suites
run "$MODELS/hs_competition_nofilter/step-1999" hs_competition_nofilter math
run "$MODELS/coding_depth4v3_nofilter/step-1999" coding_depth4v3_nofilter code
echo "=========== ALL EVALS DONE $(date -u) ==========="
echo "results -> $RESULTS"