Download harness/scripts/eval_step150_when_ready.sh from agentic-ptb/sol-max-record: direct link, hf CLI and curl.
- Browser
- Download file 6.44 kB
-
https://huggingface.co/agentic-ptb/sol-max-record/resolve/main/harness/scripts/eval_step150_when_ready.sh
- Command line
-
hf download hf://agentic-ptb/sol-max-record/harness/scripts/eval_step150_when_ready.sh
-
curl -L -o eval_step150_when_ready.sh https://huggingface.co/agentic-ptb/sol-max-record/resolve/main/harness/scripts/eval_step150_when_ready.sh
6.44 kB
| set -euo pipefail | |
| workspace="/mnt/pvc/users/simon/agentptb/runs/sol-max-s1/workspace" | |
| weights="$workspace/outputs/sft-agent-mix-clean-full-v1/weights/step_150" | |
| train_session="agentptb-sft-clean-resume50b" | |
| inference_pid="" | |
| cleanup() { | |
| if [[ -n "$inference_pid" ]]; then | |
| kill -TERM -- "-$inference_pid" 2>/dev/null || true | |
| for _ in {1..12}; do | |
| kill -0 -- "-$inference_pid" 2>/dev/null || break | |
| sleep 5 | |
| done | |
| kill -KILL -- "-$inference_pid" 2>/dev/null || true | |
| wait "$inference_pid" 2>/dev/null || true | |
| fi | |
| } | |
| trap cleanup EXIT INT TERM | |
| while [[ ! -f "$weights/STABLE" ]]; do | |
| sleep 30 | |
| done | |
| while tmux has-session -t "$train_session" 2>/dev/null; do | |
| sleep 10 | |
| done | |
| python "$workspace/scripts/ensure_chat_stop_metadata.py" "$weights" \ | |
| >"$workspace/logs/chat-stop-metadata-step150.json" | |
| mkdir -p "$workspace/candidates" | |
| if [[ ! -e "$workspace/candidates/step_150" ]]; then | |
| # Archive mode attempts a chown/chmod after linking and fails under the | |
| # shared filesystem's root-squash policy. Recursive hard links preserve | |
| # the exact file inodes without metadata mutations. | |
| cp -Rl "$weights" "$workspace/candidates/step_150" | |
| fi | |
| python "$workspace/scripts/ensure_chat_stop_metadata.py" \ | |
| "$workspace/candidates/step_150" --check >/dev/null | |
| python "$workspace/scripts/check_checkpoint.py" \ | |
| "$weights" "$workspace/outputs/sft-agent-mix-clean-full-v1/metrics.jsonl" \ | |
| --expected-step 150 >"$workspace/logs/checkpoint-health-step150.json" | |
| cd /root/work/a/prime-rl | |
| setsid bash -c 'exec "$@"' _ \ | |
| env CUDA_VISIBLE_DEVICES=4,5,6,7 \ | |
| TMPDIR=/tmp/agentptb-s1 \ | |
| PYTHONUNBUFFERED=1 \ | |
| uv run --no-sync inference @ "$workspace/configs/inference-candidate.toml" \ | |
| >"$workspace/logs/inference-step150-launch.log" 2>&1 & | |
| inference_pid=$! | |
| ready=false | |
| for _ in {1..120}; do | |
| # The DP router's generic /v1/models forwarding can rewrite the backend | |
| # address to 0.0.0.0 even while inference is fully healthy. Readiness only | |
| # needs the router health endpoint; the explicit-model smoke below verifies | |
| # an actual structured completion. | |
| if curl --fail --silent http://127.0.0.1:8200/health >/dev/null; then | |
| ready=true | |
| break | |
| fi | |
| kill -0 "$inference_pid" | |
| sleep 10 | |
| done | |
| [[ "$ready" == true ]] | |
| cd "$workspace" | |
| python scripts/smoke_tool_call.py --model "$weights" \ | |
| >logs/smoke-tool-call-step150.log 2>&1 | |
| # Keep each fixed-sample wave at no more than 24 live sandboxes. The broker is | |
| # shared, and a prior 120-wide wave spent most of its wall time pending despite | |
| # ample model capacity. Grouping by suite and sampling temperature preserves | |
| # every paired comparison while avoiding shared-pool head-of-line blocking. | |
| scripts/run_paired_evals.sh \ | |
| configs/eval-stock-tb2-32.toml \ | |
| configs/eval-custom-noreview-tb2-32.toml \ | |
| configs/eval-custom-tb2-32.toml | |
| scripts/run_paired_evals.sh \ | |
| configs/eval-stock-swe-32.toml \ | |
| configs/eval-custom-noreview-swe-32.toml \ | |
| configs/eval-custom-swe-32.toml | |
| scripts/run_paired_evals.sh \ | |
| configs/eval-custom-noreview-t06-tb2-32.toml \ | |
| configs/eval-custom-t06-tb2-32.toml \ | |
| configs/eval-stock-t06-tb2-32.toml | |
| scripts/run_paired_evals.sh \ | |
| configs/eval-custom-noreview-t06-swe-32.toml \ | |
| configs/eval-custom-t06-swe-32.toml \ | |
| configs/eval-stock-t06-swe-32.toml | |
| # Repair only explicit sandbox/network failures after both fixed waves have | |
| # exited. This remains part of the precommitted protocol, but running it in the | |
| # same process makes GPU and server ownership unambiguous. | |
| scripts/repair_step150_infra.sh | |
| for label in \ | |
| stock-tb2 custom-noreview-tb2 custom-tb2 \ | |
| stock-swe custom-noreview-swe custom-swe \ | |
| custom-noreview-t06-tb2 custom-t06-tb2 \ | |
| custom-noreview-t06-swe custom-t06-swe \ | |
| stock-t06-tb2 stock-t06-swe; do | |
| python scripts/summarize_eval.py \ | |
| "evals/step150-${label}-32-run2" \ | |
| >"logs/summary-step150-${label}-32-run2.json" | |
| done | |
| python scripts/compare_evals.py \ | |
| evals/step150-stock-tb2-32-run2 evals/step150-custom-noreview-tb2-32-run2 \ | |
| --label-a stock --label-b aligned \ | |
| >logs/compare-step150-stock-vs-aligned-tb2.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-tb2-32-run2 evals/step150-custom-tb2-32-run2 \ | |
| --label-a aligned --label-b review \ | |
| >logs/compare-step150-aligned-vs-review-tb2.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-stock-swe-32-run2 evals/step150-custom-noreview-swe-32-run2 \ | |
| --label-a stock --label-b aligned \ | |
| >logs/compare-step150-stock-vs-aligned-swe.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-swe-32-run2 evals/step150-custom-swe-32-run2 \ | |
| --label-a aligned --label-b review \ | |
| >logs/compare-step150-aligned-vs-review-swe.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-tb2-32-run2 \ | |
| evals/step150-custom-noreview-t06-tb2-32-run2 \ | |
| --label-a t02 --label-b t06 \ | |
| >logs/compare-step150-t02-vs-t06-tb2.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-swe-32-run2 \ | |
| evals/step150-custom-noreview-t06-swe-32-run2 \ | |
| --label-a t02 --label-b t06 \ | |
| >logs/compare-step150-t02-vs-t06-swe.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-t06-tb2-32-run2 \ | |
| evals/step150-custom-t06-tb2-32-run2 \ | |
| --label-a aligned --label-b review \ | |
| >logs/compare-step150-t06-aligned-vs-review-tb2.json | |
| python scripts/compare_evals.py \ | |
| evals/step150-custom-noreview-t06-swe-32-run2 \ | |
| evals/step150-custom-t06-swe-32-run2 \ | |
| --label-a aligned --label-b review \ | |
| >logs/compare-step150-t06-aligned-vs-review-swe.json | |
| python scripts/select_harness.py \ | |
| evals/step150-custom-noreview-tb2-32-run2 \ | |
| evals/step150-custom-tb2-32-run2 \ | |
| evals/step150-custom-noreview-t06-tb2-32-run2 \ | |
| evals/step150-custom-t06-tb2-32-run2 \ | |
| evals/step150-custom-noreview-swe-32-run2 \ | |
| evals/step150-custom-swe-32-run2 \ | |
| evals/step150-custom-noreview-t06-swe-32-run2 \ | |
| evals/step150-custom-t06-swe-32-run2 \ | |
| evals/step150-stock-tb2-32-run2 \ | |
| evals/step150-stock-t06-tb2-32-run2 \ | |
| evals/step150-stock-swe-32-run2 \ | |
| evals/step150-stock-t06-swe-32-run2 \ | |
| >logs/select-harness-step150.json | |
| python -c \ | |
| 'import json; print(json.load(open("logs/select-harness-step150.json"))["choice"])' \ | |
| >state/harness-choice.txt | |
| python -c \ | |
| 'import json; print(json.load(open("logs/select-harness-step150.json"))["temperature"])' \ | |
| >state/sampling-temperature.txt | |
| touch state/step150-eval-complete | |