#!/usr/bin/env bash # 32B run2 (α=0.8 τ=2.0 IWA, ckpt iwa-0505_1840) — full eval, sequential on gpu3. # Apply post-train fixes (chat_template + enable_iwa=False) BEFORE launching. # Runs alongside run1's main + extended evals. set -u REPO=/opt/tiger/thothvl_pretrain M=/mnt/bn/leonworkspace/terry/model R=/mnt/bn/leonworkspace/terry/results NAS_LOGS=/mnt/bn/leonworkspace/terry/logs LOG_DIR=$NAS_LOGS/eval_32b_run2_$(date +%Y%m%d_%H%M) mkdir -p "$LOG_DIR" SUMMARY="$LOG_DIR/summary.log" : > "$SUMMARY" cd "$REPO/lmms-eval" export PYTHONPATH=$REPO/QWENVL-PRIVATE:$PYTHONPATH export HF_TOKEN= export HF_HOME=/mnt/bn/leonworkspace/HF_HOME export HF_DATASETS_CACHE=$HF_HOME/datasets export OPENAI_API_KEY= export OPENAI_API_URL=https://search-va.byteintl.net/gpt/openapi/online/v2/crawl export MODEL_VERSION=gpt-4o-2024-11-20 unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY CKPT=$M/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.8-t2.0-iwa-0505_1840 ATTN=flash_attention_2 TAG=publish-32b-run2-a0.8-576 GPU=${GPU:-3} hybrid_args() { echo "pretrained=$CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=0.15,high_res_thresh=0.1,attn_implementation=$ATTN"; } simple_args() { echo "pretrained=$CKPT,device_map=auto,attn_implementation=$ATTN"; } run_one() { local cli=$1 args=$2 task=$3 local out_dir="$R/$TAG/$task" if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then echo "[$(date '+%F %T')] SKIP gpu$GPU $task (done)" | tee -a "$SUMMARY" return fi mkdir -p "$out_dir" local log="$LOG_DIR/${task}.log" echo "[$(date '+%F %T')] START gpu$GPU $task ($cli)" | tee -a "$SUMMARY" CUDA_VISIBLE_DEVICES=$GPU python3 -m lmms_eval \ --model "$cli" --model_args "$args" \ --tasks "$task" --batch_size 1 \ --output_path "$out_dir" \ --log_samples --log_samples_suffix "$TAG" \ > "$log" 2>&1 local rc=$? if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then echo "[$(date '+%F %T')] DONE gpu$GPU $task" | tee -a "$SUMMARY" else echo "[$(date '+%F %T')] ERR gpu$GPU $task rc=$rc — $log" | tee -a "$SUMMARY" fi } # Order: small fast tasks first for early sanity check, then the larger ones. run_one qwen3_vl_hybrid "$(hybrid_args)" vstar_bench # 191 run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench4k # 800 run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench8k # 800 run_one qwen3_vl_hybrid "$(hybrid_args)" realworldqa # 765 run_one qwen3_vl_hybrid "$(hybrid_args)" ocrbench # 1000 run_one qwen3_vl_hybrid "$(hybrid_args)" mme # 2374 run_one qwen3_vl_hybrid "$(hybrid_args)" chartqa # 2500 run_one qwen3_vl_hybrid "$(hybrid_args)" infovqa_val # 2801 run_one qwen3_vl_hybrid "$(hybrid_args)" pope # ~9000 run_one qwen3_vl_hybrid "$(hybrid_args)" scienceqa # ~4000-21000 run_one qwen3_vl_hybrid "$(hybrid_args)" docvqa_val # 5349 run_one qwen3_vl_hybrid "$(hybrid_args)" textvqa_val # 5000 run_one qwen3_vl "$(simple_args)" gqa # 12578 — simple (multi-image ROI bug) run_one qwen3_vl "$(simple_args)" seedbench # ~17000 — simple run_one qwen3_vl_hybrid "$(hybrid_args)" mmerealworld # 23609 — last, longest echo "[$(date '+%F %T')] ALL RUN2 EVALS DONE" | tee -a "$SUMMARY" echo "Logs: $LOG_DIR"