#!/bin/bash # Run once on a fresh training instance, after: # git clone https://github.com/karvachiik-lgtm/threejstraining.git coder-training # scp -r you@local:.../coder-training/{archive,datasets} coder-training/ # data, not in git — see README.md set -euo pipefail cd "$(dirname "$0")/.." # Fresh minimal cloud/Docker images often lack python3-venv (→ `ensurepip is # not available`) and build tooling bitsandbytes/flash-attn wheels sometimes # need — best-effort install, skipped quietly if apt-get isn't usable (no # root, non-Debian image, etc.) rather than failing the whole script over it. if command -v apt-get >/dev/null 2>&1 && [ "$(id -u)" = "0" ]; then echo "=== Ensuring python3-venv/build-essential (apt) ===" apt-get update -qq && apt-get install -y -qq python3-venv python3-pip build-essential >/dev/null \ || echo " (apt-get install failed or partially failed — continuing; venv creation below will surface any real problem)" fi python3 -m venv .venv source .venv/bin/activate pip install --upgrade pip pip install -r requirements.txt echo echo "=== Pre-flight checks (see README.md's checklist for what to do if these fail) ===" python3 - <<'PY' import importlib import sys checks = ["torch", "transformers", "trl", "peft", "bitsandbytes", "datasets", "wandb"] for mod in checks: try: m = importlib.import_module(mod) print(f" OK {mod:<14} {getattr(m, '__version__', '?')}") except ImportError as e: print(f" FAIL {mod:<14} {e}") sys.exit(1) import torch print(f" CUDA available: {torch.cuda.is_available()} device_count: {torch.cuda.device_count()}") if not torch.cuda.is_available(): print(" WARN no CUDA GPU visible — QLoRA/bitsandbytes below needs one; check `nvidia-smi` on this instance.") try: from transformers import AutoConfig # Cheap check: does this transformers version know about the qwen3_5 # architecture Tooony133/Qwen-3.6-27B-* and Qwen/Qwen3.6-27B report? # See requirements.txt's note — this is the biggest unverified risk here. AutoConfig.for_model("qwen3_5") print(" OK transformers recognizes model_type=qwen3_5") except Exception as e: print(f" WARN transformers may not support model_type=qwen3_5 yet: {e}") print(" You likely need `pip install git+https://github.com/huggingface/transformers.git`") PY echo if [ -f "datasets/preference.jsonl" ] && [ -f "datasets/sft.jsonl" ] && [ -f "datasets/kto.jsonl" ]; then echo "=== datasets/ already present (scp'd) — data pipeline (Phase 0/1) already done, skip it ===" echo "Next: source .venv/bin/activate && python train_sft.py --config configs/sft.yaml" else echo "=== datasets/ not found — either scp it over, or build it fresh: ===" echo " 1) scp -r you@local:.../coder-training/{archive,datasets} . (fastest — data's already built)" echo " or" echo " 2) python data/scrape_duels.py --rounds-dir --out-dir ./archive" echo " python data/build_datasets.py --manifest ./archive/duels_index.jsonl --out-dir ./datasets" fi echo "See README.md for the full pipeline; TRAINING_GUIDE.md for the algorithm/hyperparameter choices."