Download code/run_stage2_instruct.sh from LeoMaxwell/vlm-twin-spec-decoding: direct link, hf CLI and curl.
- Browser
- Download file 3.18 kB
-
https://huggingface.co/LeoMaxwell/vlm-twin-spec-decoding/resolve/main/code/run_stage2_instruct.sh
- Command line
-
hf download hf://LeoMaxwell/vlm-twin-spec-decoding/code/run_stage2_instruct.sh
-
curl -L -o run_stage2_instruct.sh https://huggingface.co/LeoMaxwell/vlm-twin-spec-decoding/resolve/main/code/run_stage2_instruct.sh
3.18 kB
| # Chain I (GPU4): Instruct 8B + Instruct int4 twin on MM-Vet (n=30, seed42) + MathVista bridge | |
| S=/testessfs10/users/zeyu.zhang/wangyu_ssd | |
| ST=$S/stage0 | |
| PY=$S/vllm_env/bin/python | |
| cd $ST | |
| export PYTHONPATH=$ST PYTHONDONTWRITEBYTECODE=1 | |
| export PATH=$S/vllm_env/bin:$PATH | |
| export HF_HOME=$S/hf_cache HF_ENDPOINT=https://hf-mirror.com | |
| export PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True | |
| MI=Qwen/Qwen3-VL-8B-Instruct | |
| DI=cyankiwi/Qwen3-VL-8B-Instruct-AWQ-4bit | |
| CMV="--parquet mmvet.parquet --pids-from mmvet_pids.jsonl --n 30 --max-tokens 512 --min-tokens 64 --gpu-mem-util 0.60 --max-model-len 3584" | |
| CBR="--parquet testmini.parquet --pids-from res05_qwen_thinking.jsonl --n 20 --max-tokens 1024 --min-tokens 64 --gpu-mem-util 0.60 --max-model-len 3584" | |
| SMK="--parquet mmvet.parquet --pids-from mmvet_pids.jsonl --n 4 --max-tokens 128 --min-tokens 64 --gpu-mem-util 0.60 --max-model-len 3584" | |
| B=$S/stage1 | |
| rm -f $B/ogi_*.jsonl $B/bri_*.jsonl $B/DONE_ogi $B/FAIL_ogi | |
| gate() { | |
| A=$(grep -o "Draft acceptance rate: [0-9.]*" "$1" | tail -1 | sed "s/.*: //") | |
| awk -v a="${A:-0}" -v t="$2" "BEGIN{exit (a>=t)?0:1}" | |
| } | |
| nohup bash -c " | |
| echo \"from huggingface_hub import snapshot_download; snapshot_download(\\\"$DI\\\")\" > /tmp/ensure_iawq.py | |
| $PY /tmp/ensure_iawq.py | |
| $(declare -f gate) | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 6 $SMK --out $B/ogi_smoke.jsonl > $B/logs/ogi_smoke.log 2>&1 | |
| if gate $B/logs/ogi_smoke.log 40; then EAG=; else EAG=--enforce-eager; touch $B/FAIL_ogi; fi | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode vanilla --model $MI $CMV \$EAG --out $B/ogi_van1.jsonl > $B/logs/ogi_van1.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 6 $CMV \$EAG --out $B/ogi_strict_g6.jsonl > $B/logs/ogi_strict_g6.log 2>&1 | |
| VLLM_SPEC_RELAX_LOGTAU=0.693 CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 6 $CMV \$EAG --out $B/ogi_relax_g6.jsonl > $B/logs/ogi_relax_g6.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode vanilla --model $MI $CMV \$EAG --out $B/ogi_vanm.jsonl > $B/logs/ogi_vanm.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 4 $CMV \$EAG --out $B/ogi_strict_g4.jsonl > $B/logs/ogi_strict_g4.log 2>&1 | |
| VLLM_SPEC_RELAX_LOGTAU=0.693 CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 4 $CMV \$EAG --out $B/ogi_relax_g4.jsonl > $B/logs/ogi_relax_g4.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode vanilla --model $MI $CMV \$EAG --out $B/ogi_van2.jsonl > $B/logs/ogi_van2.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode vanilla --model $MI $CBR \$EAG --out $B/bri_van1.jsonl > $B/logs/bri_van1.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode spec --model $MI --draft $DI --gamma 6 $CBR \$EAG --out $B/bri_strict_g6.jsonl > $B/logs/bri_strict_g6.log 2>&1 | |
| CUDA_VISIBLE_DEVICES=4 $PY stage1_spec_smoke.py --mode vanilla --model $MI $CBR \$EAG --out $B/bri_van2.jsonl > $B/logs/bri_van2.log 2>&1 | |
| touch $B/DONE_ogi | |
| " > /dev/null 2>&1 & | |
| disown | |
| echo LAUNCHED_INSTRUCT | |