#!/usr/bin/env bash # 32B GroundFlow full eval @ 576: ours (10 tasks) + baseline HRBench (2 tasks). # Distributes across GPUs 0,1,2,3,4,5,7 (GPU 6 held by backup/sm.py). # Idempotent: skips any (tag,task) already with *results.json. set -u REPO=/opt/tiger/thothvl_pretrain M=/mnt/bn/leonworkspace/terry/model R=/mnt/bn/leonworkspace/terry/results LOG_DIR=/mnt/bn/leonworkspace/terry/logs/eval_32b_$(date +%Y%m%d_%H%M) mkdir -p "$LOG_DIR" SUMMARY="$LOG_DIR/summary.log" : > "$SUMMARY" cd "$REPO/lmms-eval" export PYTHONPATH=$REPO/QWENVL-PRIVATE:$PYTHONPATH export HF_TOKEN= export HF_HOME=/mnt/bn/leonworkspace/HF_HOME export HF_DATASETS_CACHE=$HF_HOME/datasets export OPENAI_API_KEY= export OPENAI_API_URL=https://search-va.byteintl.net/gpt/openapi/online/v2/crawl export MODEL_VERSION=gpt-4o-2024-11-20 unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY OURS_CKPT=$M/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.667-t1.5-iwa-0505_0321 BASE_CKPT=$M/Qwen3-VL-32B-Instruct ATTN=flash_attention_2 CONF=0.15 ours_args() { echo "pretrained=$OURS_CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=$CONF,high_res_thresh=0.1,attn_implementation=$ATTN"; } base_args() { echo "pretrained=$BASE_CKPT,device_map=auto,attn_implementation=$ATTN"; } run_one() { local gpu=$1 cli=$2 args=$3 task=$4 tag=$5 local out_dir="$R/$tag/$task" if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then echo "[$(date '+%F %T')] SKIP gpu$gpu $tag/$task (done)" | tee -a "$SUMMARY" return fi mkdir -p "$out_dir" local log="$LOG_DIR/${tag}_${task}.log" echo "[$(date '+%F %T')] START gpu$gpu $tag/$task" | tee -a "$SUMMARY" CUDA_VISIBLE_DEVICES=$gpu python3 -m lmms_eval \ --model "$cli" --model_args "$args" \ --tasks "$task" --batch_size 1 \ --output_path "$out_dir" \ --log_samples --log_samples_suffix "$tag" \ > "$log" 2>&1 local rc=$? if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then echo "[$(date '+%F %T')] DONE gpu$gpu $tag/$task" | tee -a "$SUMMARY" else echo "[$(date '+%F %T')] ERR gpu$gpu $tag/$task rc=$rc — see $log" | tee -a "$SUMMARY" fi } # Per-GPU queues. Dispatch each queue in background. queue_gpu0() { # mmerealworld is the heaviest (23k samples) — solo on GPU 0 run_one 0 qwen3_vl_hybrid "$(ours_args)" mmerealworld publish-32b-ours-576 } queue_gpu1() { # docvqa_val 5349 run_one 1 qwen3_vl_hybrid "$(ours_args)" docvqa_val publish-32b-ours-576 } queue_gpu2() { # textvqa_val 5000 run_one 2 qwen3_vl_hybrid "$(ours_args)" textvqa_val publish-32b-ours-576 } queue_gpu3() { # infovqa_val 2801 then hrbench8k (~1k + GPT judge) run_one 3 qwen3_vl_hybrid "$(ours_args)" infovqa_val publish-32b-ours-576 run_one 3 qwen3_vl_hybrid "$(ours_args)" hrbench8k publish-32b-ours-576 } queue_gpu4() { # chartqa 2500 then hrbench4k run_one 4 qwen3_vl_hybrid "$(ours_args)" chartqa publish-32b-ours-576 run_one 4 qwen3_vl_hybrid "$(ours_args)" hrbench4k publish-32b-ours-576 } queue_gpu5() { # smaller tasks run_one 5 qwen3_vl_hybrid "$(ours_args)" ocrbench publish-32b-ours-576 run_one 5 qwen3_vl_hybrid "$(ours_args)" realworldqa publish-32b-ours-576 run_one 5 qwen3_vl_hybrid "$(ours_args)" vstar_bench publish-32b-ours-576 } queue_gpu7() { # baseline HRBench (32B Vanilla — needs both) run_one 7 qwen3_vl "$(base_args)" hrbench4k 32b-vanilla-576 run_one 7 qwen3_vl "$(base_args)" hrbench8k 32b-vanilla-576 } # Stagger model loads. 32B = 64 GB per process. Loading 6 copies in parallel # from NAS corrupts weights on some processes (gibberish output, see memory # entry 'chat_template_post_train_fix.md'). 90s gap lets each load finish # before the next starts; once loaded, all GPUs run in parallel as intended. queue_gpu0 & sleep 90; queue_gpu1 & sleep 90; queue_gpu2 & sleep 90; queue_gpu3 & sleep 90; queue_gpu4 & sleep 90; queue_gpu5 & sleep 90; queue_gpu7 & wait echo "[$(date '+%F %T')] ALL EVAL QUEUES DONE" | tee -a "$SUMMARY" echo "Logs: $LOG_DIR"