Download code/scripts/teacher_window.sh from darioooooo0o/tiny-agent-112m: direct link, hf CLI and curl.
- Browser
- Download file 3.88 kB
-
https://huggingface.co/darioooooo0o/tiny-agent-112m/resolve/main/code/scripts/teacher_window.sh
- Command line
-
hf download hf://darioooooo0o/tiny-agent-112m/code/scripts/teacher_window.sh
-
curl -L -o teacher_window.sh https://huggingface.co/darioooooo0o/tiny-agent-112m/resolve/main/code/scripts/teacher_window.sh
3.88 kB
| # Pause the main run just before its LR decay, generate Qwen3.8-27B teacher trajectories on the | |
| # freed GPU, build them into the synth_teacher source, then resume the run (which then decays with | |
| # the teacher data in its mix). If vLLM fails to come up, the run is resumed anyway. | |
| # The pre-decay checkpoint is kept, so pretraining can be continued later from it; on resume the | |
| # decay is forced right away and lasts --decay_frac of the steps done (no wall-clock re-fit). | |
| # PAUSE_AT=14:00 GEN_HOURS=10 setsid nohup bash scripts/teacher_window.sh & | |
| set -u | |
| cd "$(dirname "$0")/.." | |
| source env.sh | |
| RUN=$TA_DATA/runs/main | |
| PAUSE_AT=${PAUSE_AT:-14:00} | |
| GEN_HOURS=${GEN_HOURS:-10} | |
| L=$TA_DATA/logs/teacher_window.log | |
| log() { echo "[$(date '+%F %T')] $*" | tee -a "$L"; } | |
| COOKBOOK=${COOKBOOK:?set COOKBOOK to a dir with the vLLM-XPU patches} | |
| MODEL_DIR=${MODEL_DIR:?set MODEL_DIR to a local Qwen3.8-27B GPTQ copy} | |
| IMAGE='vllm/vllm-openai-xpu@sha256:f01e24f6c7ff01f1e0662234255a1372297d1dbd89d003cf13c8fad3eab1ba4f' | |
| TRAIN_CMD=($TA_PY scripts/train.py --size L --engram 1 --lr 3e-3 --tokens 1e10 --max_minutes 0 --decay_T 4096 | |
| --eval_every 250 --snapshot_tokens 5e8 --out "$RUN") | |
| target=$(date -d "today $PAUSE_AT" +%s) | |
| now=$(date +%s) | |
| [ "$target" -gt "$now" ] && { log "waiting until $PAUSE_AT"; sleep $((target - now)); } | |
| PID=$(ps -eo pid,args | awk '$3=="scripts/train.py" && $0 ~ /runs\/main/ {print $1}' | head -1) | |
| if [ -z "$PID" ]; then log "main run not running; nothing to pause"; exit 1; fi | |
| log "pausing main run (pid $PID)" | |
| touch "$RUN/STOP" | |
| while kill -0 "$PID" 2>/dev/null; do sleep 10; done | |
| log "main run stopped: $(tail -1 "$RUN/log.jsonl" | cut -c1-120)" | |
| TOK=$($TA_PY -c "import json;print(f\"{json.loads(open('$RUN/log.jsonl').readlines()[-1])['tokens']/1e9:.2f}B\")") | |
| cp "$RUN/ckpt.pt" "$RUN/ckpt_predecay_$TOK.pt" && log "kept pre-decay checkpoint ckpt_predecay_$TOK.pt" | |
| if [ ! -f "$MODEL_DIR/quantize_config.json" ]; then | |
| log "teacher model not downloaded; skipping generation" | |
| else | |
| docker rm -f tinyagent-teacher >/dev/null 2>&1 || true | |
| RGID="$(stat -c '%g' /dev/dri/render* | sort -u | sed -n '1p')" | |
| docker run -d --name tinyagent-teacher -p 8000:8000 --device /dev/dri --group-add "$RGID" \ | |
| -v /dev/dri:/dev/dri:ro -v "$MODEL_DIR:/model:ro" \ | |
| -v "$COOKBOOK/patches/patch_mtp_nightly.py:/patch_mtp.py:ro" \ | |
| -v "$COOKBOOK/patches/patch_mtp_boundary.py:/patch_boundary.py:ro" \ | |
| -e VLLM_TARGET_DEVICE=xpu -e ZE_FLAT_DEVICE_HIERARCHY=COMPOSITE -e ZE_AFFINITY_MASK=0 \ | |
| -e B70_MTP_BF16_DRAFT=1 -e VLLM_XPU_ENABLE_XPU_GRAPH=1 -e PYTORCH_ALLOC_CONF=expandable_segments:True \ | |
| --entrypoint bash "$IMAGE" -lc \ | |
| "set -e; python /patch_mtp.py; python /patch_boundary.py; exec vllm serve /model --quantization gptq --dtype float16 --max-model-len 32768 --gpu-memory-utilization 0.90 --kv-cache-dtype fp8 --port 8000 --max-num-seqs 64 --max-num-batched-tokens 8192 --no-enable-prefix-caching --served-model-name qwen38 --language-model-only" \ | |
| >> "$L" 2>&1 | |
| ok=0 | |
| for i in $(seq 1 120); do | |
| if curl -sf http://127.0.0.1:8000/health >/dev/null; then ok=1; break; fi | |
| sleep 10 | |
| done | |
| if [ "$ok" = 1 ]; then | |
| log "vLLM up; generating teacher trajectories for $GEN_HOURS h" | |
| $TA_PY scripts/teacher.py --hours "$GEN_HOURS" --concurrency 48 \ | |
| --out "$TA_DATA/teacher/qwen38_$(date +%Y%m%d).jsonl" >> "$TA_DATA/logs/teacher.log" 2>&1 | |
| log "teacher done: $(tail -1 "$TA_DATA/logs/teacher.log" | cut -c1-300)" | |
| else | |
| log "vLLM did not become healthy; last docker logs:" | |
| docker logs --tail 40 tinyagent-teacher >> "$L" 2>&1 | |
| fi | |
| docker rm -f tinyagent-teacher >/dev/null 2>&1 || true | |
| $TA_PY scripts/gen_synth.py --only teacher >> "$L" 2>&1 | |
| fi | |
| log "resuming main run with the decay forced now" | |
| touch "$RUN/DECAY" | |
| nohup "${TRAIN_CMD[@]}" >> "$TA_DATA/runs/main.out" 2>&1 & | |
| log "resumed (pid $!)" | |