File size: 3,591 Bytes
58258b8
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
#!/usr/bin/env bash
# Run the Qwen3-8B baseline + the completed SFT checkpoints through evalchemy, sequentially,
# writing results to HDFS so a machine kill doesn't lose completed evals.
#
#   source /opt/tiger/lmc_muon/gpu-eval/activate.sh   # (this script also sources it)
#   nohup bash gpu-sft/scripts/gpu_eval/run_all_evals.sh \
#       > /mnt/hdfs/fangzhao_writable/marin-eval/eval_all.log 2>&1 &
#
# Each eval uses the full node (tp defaults to #GPUs). Jobs run one at a time. Results go to
# $RESULTS on HDFS; a completed task's JSON is durable immediately, so a kill only costs the
# in-flight task (re-run the script — the driver skips finished tasks unless --force).
source /opt/tiger/lmc_muon/gpu-eval/activate.sh 2>/dev/null || true
set -uo pipefail

DRIVER=/opt/tiger/lmc_muon/self-verified-distillation/gpu-sft/scripts/gpu_eval/gpu_eval_driver.py
RESULTS="${RESULTS:-/mnt/hdfs/fangzhao_writable/marin-eval/results}"
HDFS_SFT=/mnt/hdfs/fangzhao_writable/marin-sft
MODELS=/opt/tiger/lmc_muon/models
PROFILE="${PROFILE:-h100-80g}"
mkdir -p "$RESULTS" "$MODELS"

run() {  # run <checkpoint_dir> <experiment> <suite>
  echo "=========== eval $2 / $3   $(date -u) ==========="
  python "$DRIVER" --checkpoint "$1" --experiment "$2" --suite "$3" \
    --gpu-profile "$PROFILE" --results-root "$RESULTS" \
    || echo "!! $2/$3 FAILED (continuing to next)"
}

# Stage the completed checkpoints from HDFS -> local, keeping a 'step-1999' leaf so the driver
# parses the step number (16 GB each, ~30 s at 574 MB/s).
for e in coding_depth4v3_nofilter hs_competition_nofilter; do
  DST="$MODELS/$e/step-1999"
  if [ ! -f "$DST/config.json" ]; then
    echo "staging $e/hf/step-1999 from HDFS -> $DST"
    mkdir -p "$MODELS/$e" && cp -r "$HDFS_SFT/$e/hf/step-1999" "$DST"
  fi
done

if [ "${PARALLEL:-0}" = "1" ]; then
  LOGDIR="$RESULTS/_logs"; mkdir -p "$LOGDIR"
  echo "PARALLEL mode: 4 jobs x tp=2 across GPU pairs (per-job logs in $LOGDIR)"
  CUDA_VISIBLE_DEVICES=0,1 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_math.log" 2>&1 &
  CUDA_VISIBLE_DEVICES=2,3 python "$DRIVER" --checkpoint "$MODELS/Qwen3-8B" --experiment qwen3_8b_base --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/base_code.log" 2>&1 &
  CUDA_VISIBLE_DEVICES=4,5 python "$DRIVER" --checkpoint "$MODELS/hs_competition_nofilter/step-1999" --experiment hs_competition_nofilter --suite math --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/hs_math.log" 2>&1 &
  CUDA_VISIBLE_DEVICES=6,7 python "$DRIVER" --checkpoint "$MODELS/coding_depth4v3_nofilter/step-1999" --experiment coding_depth4v3_nofilter --suite code --tensor-parallel-size 2 --gpu-profile "$PROFILE" --results-root "$RESULTS" > "$LOGDIR/coding_code.log" 2>&1 &
  wait
  echo "=========== ALL PARALLEL EVALS DONE   $(date -u) ==========="
  echo "results -> $RESULTS ; per-job logs -> $LOGDIR"
  exit 0
fi

# 1) baseline (the denominator for every (+delta)) on both suites
run "$MODELS/Qwen3-8B"                       qwen3_8b_base            math
run "$MODELS/Qwen3-8B"                       qwen3_8b_base            code
# 2) the two completed checkpoints on their matching suites
run "$MODELS/hs_competition_nofilter/step-1999"   hs_competition_nofilter   math
run "$MODELS/coding_depth4v3_nofilter/step-1999"  coding_depth4v3_nofilter  code

echo "=========== ALL EVALS DONE   $(date -u) ==========="
echo "results -> $RESULTS"