GroundFlow / scripts /eval_32b_hrbench_cluster.sh
TerryPei's picture
32B full results: nips.tex (Vanilla/SD-RPN/GroundFlow), trainer_state for run2 + SD-RPN, textvqa patch tile figures (PNG), eval/sync scripts
d6e1c8a verified
Raw History Blame Contribute Delete
4.59 kB
#!/usr/bin/env bash
# ============================================================
# 32B Dense — Missing HRBench Eval (cluster-ready)
#
# Self-contained: re-creates the env on a FRESH cluster node, then runs
# ours/hrbench4k, ours/hrbench8k (32B GroundFlow Confluent+IWA)
# baseline/hrbench4k, baseline/hrbench8k (32B Vanilla Qwen3-VL-32B-Instruct)
# in parallel on GPUs 0-3 (one task per 80GB GPU).
#
# Idempotent: skips any (tag,task) that already has *results.json on NAS.
#
# Usage on a fresh cluster node:
# bash /opt/tiger/thothvl_pretrain/scripts/eval_32b_hrbench_cluster.sh
#
# If the repo is missing, bootstrap it first via:
# source /opt/tiger/thothvl_pretrain/bootstrap_tie.sh
# bash /opt/tiger/thothvl_pretrain/scripts/eval_32b_hrbench_cluster.sh
# ============================================================
set -u
REPO=/opt/tiger/thothvl_pretrain
NAS=/mnt/bn/leonworkspace
# ---------- 1. Repo + deps (idempotent) ----------
if [ ! -d "$REPO/QWENVL-PRIVATE" ] || [ ! -d "$REPO/lmms-eval" ]; then
echo "[setup] Repo missing. Source bootstrap_tie.sh first:"
echo " source $REPO/bootstrap_tie.sh"
exit 1
fi
cd "$REPO"
# Ensure deps installed; cheap if already there
bash "$REPO/QWENVL-PRIVATE/tools/install.sh" > /tmp/install_qwenvl.log 2>&1 || \
echo "[setup] install.sh warnings — see /tmp/install_qwenvl.log"
(cd "$REPO/lmms-eval" && pip3 install -e . > /dev/null 2>&1) || true
pip3 install loguru > /dev/null 2>&1 || true
# ---------- 2. Env (must match current node) ----------
export HF_HOME=$NAS/HF_HOME
export HF_DATASETS_CACHE=$HF_HOME/datasets
export HF_TOKEN=<HF_TOKEN>
export HF_HUB_ENABLE_HF_TRANSFER=1
# OpenAI proxy for HRBench GPT-judge (api.openai.com is blocked from cluster)
export OPENAI_API_KEY=<OPENAI_API_KEY>
export OPENAI_API_URL=https://search-va.byteintl.net/gpt/openapi/online/v2/crawl
export MODEL_VERSION=gpt-4o-2024-11-20
export PYTHONPATH=$REPO/QWENVL-PRIVATE:${PYTHONPATH:-}
unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY
# ---------- 3. Eval config ----------
cd "$REPO/lmms-eval"
OURS_CKPT=$NAS/terry/model/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.667-t1.5-iwa-0505_0321
BASE_CKPT=$NAS/terry/model/Qwen3-VL-32B-Instruct
R=$NAS/terry/results
ATTN=flash_attention_2
CONF=0.15
LOG_DIR=$NAS/terry/logs/eval_32b_hrbench_$(date +%Y%m%d_%H%M)
mkdir -p "$LOG_DIR"
SUMMARY="$LOG_DIR/summary.log"
: > "$SUMMARY"
ours_args() {
echo "pretrained=$OURS_CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=$CONF,high_res_thresh=0.1,attn_implementation=$ATTN"
}
base_args() {
echo "pretrained=$BASE_CKPT,device_map=auto,attn_implementation=$ATTN"
}
run_one() {
local gpu=$1 cli=$2 args=$3 task=$4 tag=$5
local out_dir="$R/$tag/$task"
if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
echo "[$(date '+%F %T')] SKIP gpu$gpu $tag/$task (done)" | tee -a "$SUMMARY"
return
fi
mkdir -p "$out_dir"
local log="$LOG_DIR/${tag}_${task}.log"
echo "[$(date '+%F %T')] START gpu$gpu $tag/$task" | tee -a "$SUMMARY"
CUDA_VISIBLE_DEVICES=$gpu python3 -m lmms_eval \
--model "$cli" --model_args "$args" \
--tasks "$task" --batch_size 1 \
--output_path "$out_dir" \
--log_samples --log_samples_suffix "$tag" \
> "$log" 2>&1
local rc=$?
if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
echo "[$(date '+%F %T')] DONE gpu$gpu $tag/$task" | tee -a "$SUMMARY"
else
echo "[$(date '+%F %T')] ERR gpu$gpu $tag/$task rc=$rc — see $log" | tee -a "$SUMMARY"
fi
}
# ---------- 4. Launch 4 evals in parallel ----------
run_one 0 qwen3_vl_hybrid "$(ours_args)" hrbench4k publish-32b-ours-576 &
run_one 1 qwen3_vl_hybrid "$(ours_args)" hrbench8k publish-32b-ours-576 &
run_one 2 qwen3_vl "$(base_args)" hrbench4k 32b-vanilla-576 &
run_one 3 qwen3_vl "$(base_args)" hrbench8k 32b-vanilla-576 &
wait
# ---------- 5. Report ----------
echo "[$(date '+%F %T')] ALL DONE" | tee -a "$SUMMARY"
echo "Logs: $LOG_DIR"
echo
echo "=== Final scores ==="
for d in "$R/publish-32b-ours-576/hrbench4k" "$R/publish-32b-ours-576/hrbench8k" \
"$R/32b-vanilla-576/hrbench4k" "$R/32b-vanilla-576/hrbench8k"; do
f=$(find "$d" -name "*results.json" 2>/dev/null | head -1)
if [ -n "$f" ]; then
echo "$d:"
python3 -c "import json; r=json.load(open('$f'))['results']; [print(f' {k}: {v}') for k,v in r.items()]" 2>/dev/null || cat "$f"
else
echo "$d: (no results)"
fi
done