#!/bin/bash # Wait for the main run to finish its decay (final.pt written and the trainer gone), run the agent # eval gate on final.pt, then start GRPO from it. Logs: $TA_DATA/logs/rl_after_decay.log # setsid nohup bash scripts/rl_after_decay.sh >/dev/null 2>&1 < /dev/null & set -u cd "$(dirname "$0")/.." source env.sh RUN=$TA_DATA/runs/main OUT=${RL_OUT:-$TA_DATA/rl/r1} L=$TA_DATA/logs/rl_after_decay.log log() { echo "[$(date '+%F %T')] $*" | tee -a "$L"; } trainer() { ps -eo pid,args | awk '$3=="scripts/train.py" && $0 ~ /runs\/main/ {print $1}' | head -1; } log "waiting for $RUN/final.pt" while [ ! -f "$RUN/final.pt" ] || [ -n "$(trainer)" ]; do sleep 60; done # the orchestrator also must be done (it starts the trainer after teacher generation) while ps -eo args | awk '$2=="scripts/teacher_window.sh"' | grep -q .; do sleep 60; done log "decay finished: $(tail -1 "$RUN/log.jsonl" | cut -c1-200)" log "eval gate on final.pt" $TA_PY scripts/eval_agent.py --ckpt "$RUN/final.pt" --tasks 150 --k 8 --verbose 0 > "$TA_DATA/logs/eval_final.json" 2>> "$L" log "eval: $($TA_PY -c "import json;r=json.load(open('$TA_DATA/logs/eval_final.json'));print({k:r[k] for k in ('_in_dist','_held_out','_all')})" 2>&1 | cut -c1-400)" log "starting GRPO -> $OUT (continuous engine)" mkdir -p "$OUT" $TA_PY scripts/grpo.py --ckpt "$RUN/final.pt" --out "$OUT" --steps ${RL_STEPS:-300} >> "$OUT/stdout.log" 2>&1 rc=$? log "GRPO exited with $rc: $(tail -1 "$OUT/log.jsonl" 2>/dev/null | cut -c1-200)" if [ $rc -ne 0 ]; then # fallback: lockstep rollouts, smaller batches (OOM or an engine bug); a fresh run from final.pt log "first run failed; stdout tail:" tail -25 "$OUT/stdout.log" >> "$L" OUT2=${OUT}_lockstep mkdir -p "$OUT2" log "starting fallback GRPO -> $OUT2 (lockstep, micro 2, 128 rows)" $TA_PY scripts/grpo.py --ckpt "$RUN/final.pt" --out "$OUT2" --steps ${RL_STEPS:-300} --engine lockstep \ --micro 2 --n_tasks 16 >> "$OUT2/stdout.log" 2>&1 log "fallback GRPO exited with $?: $(tail -1 "$OUT2/log.jsonl" 2>/dev/null | cut -c1-200)" fi