File size: 3,415 Bytes
d6e1c8a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 | #!/usr/bin/env bash
# 32B run2 (α=0.8 τ=2.0 IWA, ckpt iwa-0505_1840) — full eval, sequential on gpu3.
# Apply post-train fixes (chat_template + enable_iwa=False) BEFORE launching.
# Runs alongside run1's main + extended evals.
set -u
REPO=/opt/tiger/thothvl_pretrain
M=/mnt/bn/leonworkspace/terry/model
R=/mnt/bn/leonworkspace/terry/results
NAS_LOGS=/mnt/bn/leonworkspace/terry/logs
LOG_DIR=$NAS_LOGS/eval_32b_run2_$(date +%Y%m%d_%H%M)
mkdir -p "$LOG_DIR"
SUMMARY="$LOG_DIR/summary.log"
: > "$SUMMARY"
cd "$REPO/lmms-eval"
export PYTHONPATH=$REPO/QWENVL-PRIVATE:$PYTHONPATH
export HF_TOKEN=<HF_TOKEN>
export HF_HOME=/mnt/bn/leonworkspace/HF_HOME
export HF_DATASETS_CACHE=$HF_HOME/datasets
export OPENAI_API_KEY=<OPENAI_API_KEY>
export OPENAI_API_URL=https://search-va.byteintl.net/gpt/openapi/online/v2/crawl
export MODEL_VERSION=gpt-4o-2024-11-20
unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY
CKPT=$M/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.8-t2.0-iwa-0505_1840
ATTN=flash_attention_2
TAG=publish-32b-run2-a0.8-576
GPU=${GPU:-3}
hybrid_args() { echo "pretrained=$CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=0.15,high_res_thresh=0.1,attn_implementation=$ATTN"; }
simple_args() { echo "pretrained=$CKPT,device_map=auto,attn_implementation=$ATTN"; }
run_one() {
local cli=$1 args=$2 task=$3
local out_dir="$R/$TAG/$task"
if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
echo "[$(date '+%F %T')] SKIP gpu$GPU $task (done)" | tee -a "$SUMMARY"
return
fi
mkdir -p "$out_dir"
local log="$LOG_DIR/${task}.log"
echo "[$(date '+%F %T')] START gpu$GPU $task ($cli)" | tee -a "$SUMMARY"
CUDA_VISIBLE_DEVICES=$GPU python3 -m lmms_eval \
--model "$cli" --model_args "$args" \
--tasks "$task" --batch_size 1 \
--output_path "$out_dir" \
--log_samples --log_samples_suffix "$TAG" \
> "$log" 2>&1
local rc=$?
if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then
echo "[$(date '+%F %T')] DONE gpu$GPU $task" | tee -a "$SUMMARY"
else
echo "[$(date '+%F %T')] ERR gpu$GPU $task rc=$rc — $log" | tee -a "$SUMMARY"
fi
}
# Order: small fast tasks first for early sanity check, then the larger ones.
run_one qwen3_vl_hybrid "$(hybrid_args)" vstar_bench # 191
run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench4k # 800
run_one qwen3_vl_hybrid "$(hybrid_args)" hrbench8k # 800
run_one qwen3_vl_hybrid "$(hybrid_args)" realworldqa # 765
run_one qwen3_vl_hybrid "$(hybrid_args)" ocrbench # 1000
run_one qwen3_vl_hybrid "$(hybrid_args)" mme # 2374
run_one qwen3_vl_hybrid "$(hybrid_args)" chartqa # 2500
run_one qwen3_vl_hybrid "$(hybrid_args)" infovqa_val # 2801
run_one qwen3_vl_hybrid "$(hybrid_args)" pope # ~9000
run_one qwen3_vl_hybrid "$(hybrid_args)" scienceqa # ~4000-21000
run_one qwen3_vl_hybrid "$(hybrid_args)" docvqa_val # 5349
run_one qwen3_vl_hybrid "$(hybrid_args)" textvqa_val # 5000
run_one qwen3_vl "$(simple_args)" gqa # 12578 — simple (multi-image ROI bug)
run_one qwen3_vl "$(simple_args)" seedbench # ~17000 — simple
run_one qwen3_vl_hybrid "$(hybrid_args)" mmerealworld # 23609 — last, longest
echo "[$(date '+%F %T')] ALL RUN2 EVALS DONE" | tee -a "$SUMMARY"
echo "Logs: $LOG_DIR"
|