File size: 3,232 Bytes
78cb49e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
#!/bin/bash
# Run once on a fresh training instance, after:
#   git clone https://github.com/karvachiik-lgtm/threejstraining.git coder-training
#   scp -r you@local:.../coder-training/{archive,datasets} coder-training/   # data, not in git β€” see README.md
set -euo pipefail

cd "$(dirname "$0")/.."

# Fresh minimal cloud/Docker images often lack python3-venv (β†’ `ensurepip is
# not available`) and build tooling bitsandbytes/flash-attn wheels sometimes
# need β€” best-effort install, skipped quietly if apt-get isn't usable (no
# root, non-Debian image, etc.) rather than failing the whole script over it.
if command -v apt-get >/dev/null 2>&1 && [ "$(id -u)" = "0" ]; then
    echo "=== Ensuring python3-venv/build-essential (apt) ==="
    apt-get update -qq && apt-get install -y -qq python3-venv python3-pip build-essential >/dev/null \
        || echo "  (apt-get install failed or partially failed β€” continuing; venv creation below will surface any real problem)"
fi

python3 -m venv .venv
source .venv/bin/activate
pip install --upgrade pip
pip install -r requirements.txt

echo
echo "=== Pre-flight checks (see README.md's checklist for what to do if these fail) ==="
python3 - <<'PY'
import importlib
import sys

checks = ["torch", "transformers", "trl", "peft", "bitsandbytes", "datasets", "wandb"]
for mod in checks:
    try:
        m = importlib.import_module(mod)
        print(f"  OK  {mod:<14} {getattr(m, '__version__', '?')}")
    except ImportError as e:
        print(f"  FAIL {mod:<14} {e}")
        sys.exit(1)

import torch
print(f"  CUDA available: {torch.cuda.is_available()}  device_count: {torch.cuda.device_count()}")
if not torch.cuda.is_available():
    print("  WARN no CUDA GPU visible β€” QLoRA/bitsandbytes below needs one; check `nvidia-smi` on this instance.")

try:
    from transformers import AutoConfig
    # Cheap check: does this transformers version know about the qwen3_5
    # architecture Tooony133/Qwen-3.6-27B-* and Qwen/Qwen3.6-27B report?
    # See requirements.txt's note β€” this is the biggest unverified risk here.
    AutoConfig.for_model("qwen3_5")
    print("  OK  transformers recognizes model_type=qwen3_5")
except Exception as e:
    print(f"  WARN transformers may not support model_type=qwen3_5 yet: {e}")
    print("       You likely need `pip install git+https://github.com/huggingface/transformers.git`")
PY

echo
if [ -f "datasets/preference.jsonl" ] && [ -f "datasets/sft.jsonl" ] && [ -f "datasets/kto.jsonl" ]; then
    echo "=== datasets/ already present (scp'd) β€” data pipeline (Phase 0/1) already done, skip it ==="
    echo "Next: source .venv/bin/activate && python train_sft.py --config configs/sft.yaml"
else
    echo "=== datasets/ not found β€” either scp it over, or build it fresh: ==="
    echo "  1) scp -r you@local:.../coder-training/{archive,datasets} .   (fastest β€” data's already built)"
    echo "  or"
    echo "  2) python data/scrape_duels.py --rounds-dir <path-to-404-active-competition/rounds> --out-dir ./archive"
    echo "     python data/build_datasets.py --manifest ./archive/duels_index.jsonl --out-dir ./datasets"
fi
echo "See README.md for the full pipeline; TRAINING_GUIDE.md for the algorithm/hyperparameter choices."