Download scripts/setup_env.sh from halle01/coder-fake: direct link, hf CLI and curl.
- Browser
- Download file 3.23 kB
-
https://huggingface.co/halle01/coder-fake/resolve/main/scripts/setup_env.sh
- Command line
-
hf download hf://halle01/coder-fake/scripts/setup_env.sh
-
curl -L -o setup_env.sh https://huggingface.co/halle01/coder-fake/resolve/main/scripts/setup_env.sh
3.23 kB
| # Run once on a fresh training instance, after: | |
| # git clone https://github.com/karvachiik-lgtm/threejstraining.git coder-training | |
| # scp -r you@local:.../coder-training/{archive,datasets} coder-training/ # data, not in git β see README.md | |
| set -euo pipefail | |
| cd "$(dirname "$0")/.." | |
| # Fresh minimal cloud/Docker images often lack python3-venv (β `ensurepip is | |
| # not available`) and build tooling bitsandbytes/flash-attn wheels sometimes | |
| # need β best-effort install, skipped quietly if apt-get isn't usable (no | |
| # root, non-Debian image, etc.) rather than failing the whole script over it. | |
| if command -v apt-get >/dev/null 2>&1 && [ "$(id -u)" = "0" ]; then | |
| echo "=== Ensuring python3-venv/build-essential (apt) ===" | |
| apt-get update -qq && apt-get install -y -qq python3-venv python3-pip build-essential >/dev/null \ | |
| || echo " (apt-get install failed or partially failed β continuing; venv creation below will surface any real problem)" | |
| fi | |
| python3 -m venv .venv | |
| source .venv/bin/activate | |
| pip install --upgrade pip | |
| pip install -r requirements.txt | |
| echo | |
| echo "=== Pre-flight checks (see README.md's checklist for what to do if these fail) ===" | |
| python3 - <<'PY' | |
| import importlib | |
| import sys | |
| checks = ["torch", "transformers", "trl", "peft", "bitsandbytes", "datasets", "wandb"] | |
| for mod in checks: | |
| try: | |
| m = importlib.import_module(mod) | |
| print(f" OK {mod:<14} {getattr(m, '__version__', '?')}") | |
| except ImportError as e: | |
| print(f" FAIL {mod:<14} {e}") | |
| sys.exit(1) | |
| import torch | |
| print(f" CUDA available: {torch.cuda.is_available()} device_count: {torch.cuda.device_count()}") | |
| if not torch.cuda.is_available(): | |
| print(" WARN no CUDA GPU visible β QLoRA/bitsandbytes below needs one; check `nvidia-smi` on this instance.") | |
| try: | |
| from transformers import AutoConfig | |
| # Cheap check: does this transformers version know about the qwen3_5 | |
| # architecture Tooony133/Qwen-3.6-27B-* and Qwen/Qwen3.6-27B report? | |
| # See requirements.txt's note β this is the biggest unverified risk here. | |
| AutoConfig.for_model("qwen3_5") | |
| print(" OK transformers recognizes model_type=qwen3_5") | |
| except Exception as e: | |
| print(f" WARN transformers may not support model_type=qwen3_5 yet: {e}") | |
| print(" You likely need `pip install git+https://github.com/huggingface/transformers.git`") | |
| PY | |
| echo | |
| if [ -f "datasets/preference.jsonl" ] && [ -f "datasets/sft.jsonl" ] && [ -f "datasets/kto.jsonl" ]; then | |
| echo "=== datasets/ already present (scp'd) β data pipeline (Phase 0/1) already done, skip it ===" | |
| echo "Next: source .venv/bin/activate && python train_sft.py --config configs/sft.yaml" | |
| else | |
| echo "=== datasets/ not found β either scp it over, or build it fresh: ===" | |
| echo " 1) scp -r you@local:.../coder-training/{archive,datasets} . (fastest β data's already built)" | |
| echo " or" | |
| echo " 2) python data/scrape_duels.py --rounds-dir <path-to-404-active-competition/rounds> --out-dir ./archive" | |
| echo " python data/build_datasets.py --manifest ./archive/duels_index.jsonl --out-dir ./datasets" | |
| fi | |
| echo "See README.md for the full pipeline; TRAINING_GUIDE.md for the algorithm/hyperparameter choices." | |