Download scripts/eval_32b_extended.sh from TerryPei/GroundFlow: direct link, hf CLI and curl.
- Browser
- Download file 2.62 kB
-
https://huggingface.co/TerryPei/GroundFlow/resolve/main/scripts/eval_32b_extended.sh
- Command line
-
hf download hf://TerryPei/GroundFlow/scripts/eval_32b_extended.sh
-
curl -L -o eval_32b_extended.sh https://huggingface.co/TerryPei/GroundFlow/resolve/main/scripts/eval_32b_extended.sh
2.62 kB
| # 32B GroundFlow extended benchmarks: pope, scienceqa, mme, gqa, seedbench. | |
| # Runs on idle GPUs (4 + 7) alongside the in-progress main eval. | |
| # seedbench/gqa use qwen3_vl simple (multi-image ROI crashes with hybrid). | |
| set -u | |
| REPO=/opt/tiger/thothvl_pretrain | |
| M=/mnt/bn/leonworkspace/terry/model | |
| R=/mnt/bn/leonworkspace/terry/results | |
| NAS_LOGS=/mnt/bn/leonworkspace/terry/logs | |
| LOG_DIR=$NAS_LOGS/eval_32b_ext_$(date +%Y%m%d_%H%M) | |
| mkdir -p "$LOG_DIR" | |
| SUMMARY="$LOG_DIR/summary.log" | |
| : > "$SUMMARY" | |
| cd "$REPO/lmms-eval" | |
| export PYTHONPATH=$REPO/QWENVL-PRIVATE:$PYTHONPATH | |
| export HF_TOKEN=<HF_TOKEN> | |
| export HF_HOME=/mnt/bn/leonworkspace/HF_HOME | |
| export HF_DATASETS_CACHE=$HF_HOME/datasets | |
| unset http_proxy https_proxy HTTP_PROXY HTTPS_PROXY | |
| OURS_CKPT=$M/qwen3vl-32b-dense-roi-K49T3-150k-confluent-a0.667-t1.5-iwa-0505_0321 | |
| ATTN=flash_attention_2 | |
| TAG=publish-32b-ours-576 | |
| hybrid_args() { echo "pretrained=$OURS_CKPT,device_map=auto,two_stage_roi=True,roi_baseline=True,roi_conf_thresh=0.15,high_res_thresh=0.1,attn_implementation=$ATTN"; } | |
| simple_args() { echo "pretrained=$OURS_CKPT,device_map=auto,attn_implementation=$ATTN"; } | |
| run_one() { | |
| local gpu=$1 cli=$2 args=$3 task=$4 | |
| local out_dir="$R/$TAG/$task" | |
| if find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then | |
| echo "[$(date '+%F %T')] SKIP gpu$gpu $task (done)" | tee -a "$SUMMARY" | |
| return | |
| fi | |
| mkdir -p "$out_dir" | |
| local log="$LOG_DIR/${task}_gpu${gpu}.log" | |
| echo "[$(date '+%F %T')] START gpu$gpu $task ($cli)" | tee -a "$SUMMARY" | |
| CUDA_VISIBLE_DEVICES=$gpu python3 -m lmms_eval \ | |
| --model "$cli" --model_args "$args" \ | |
| --tasks "$task" --batch_size 1 \ | |
| --output_path "$out_dir" \ | |
| --log_samples --log_samples_suffix "$TAG" \ | |
| > "$log" 2>&1 | |
| local rc=$? | |
| if [ $rc -eq 0 ] && find "$out_dir" -name "*results.json" 2>/dev/null | grep -q .; then | |
| echo "[$(date '+%F %T')] DONE gpu$gpu $task" | tee -a "$SUMMARY" | |
| else | |
| echo "[$(date '+%F %T')] ERR gpu$gpu $task rc=$rc — $log" | tee -a "$SUMMARY" | |
| fi | |
| } | |
| # gpu4: hybrid-compatible (pope, scienceqa, mme) — sequential | |
| queue_gpu4() { | |
| run_one 4 qwen3_vl_hybrid "$(hybrid_args)" pope | |
| run_one 4 qwen3_vl_hybrid "$(hybrid_args)" scienceqa | |
| run_one 4 qwen3_vl_hybrid "$(hybrid_args)" mme | |
| } | |
| # gpu7: simple-only (gqa, seedbench) — sequential | |
| queue_gpu7() { | |
| run_one 7 qwen3_vl "$(simple_args)" gqa | |
| run_one 7 qwen3_vl "$(simple_args)" seedbench | |
| } | |
| queue_gpu4 & | |
| queue_gpu7 & | |
| wait | |
| echo "[$(date '+%F %T')] ALL EXTENDED DONE" | tee -a "$SUMMARY" | |
| echo "Logs: $LOG_DIR" | |