#!/bin/bash # FALLBACK training path — install Unsloth. Use this only if the primary # TRL + flash-linear-attention path (scripts/setup_fast_kernels.sh) hits # trouble on the new instance. Unsloth ships its own fused Gated-DeltaNet / # MoE Triton kernels (so it doesn't depend on fla) and claims ~1.5-1.7x # faster / 50-60% less VRAM vs FA2. See SPEEDUP_FINDINGS.md's "TRL vs Unsloth". # # READ THIS BEFORE USING — two real caveats specific to our model on 48GB: # 1. Unsloth EXPLICITLY recommends AGAINST 4-bit QLoRA for Qwen3.5 ("not # recommended to do QLoRA (4-bit) ... due to higher-than-normal # quantization differences", and "MoE QLoRA 4-bit not recommended due to # BitsandBytes limitations"). It steers you to bf16 LoRA. # 2. But their own VRAM table puts 27B bf16 LoRA at ~56GB — OVER our 48GB. # So on a single 48GB card you're forced to either (a) run 4-bit anyway, # against their advice, and validate output quality hard on the eval set, or # (b) drop to a smaller Qwen3.5 variant that fits bf16 (their table: 9B=22GB, # 4B=10GB). The configs/*_unsloth.yaml default to 4-bit with that warning; # flip load_in_4bit->false + pick a smaller base if quality is unacceptable. # # Run AFTER scripts/setup_env.sh, in the same venv. set -euo pipefail cd "$(dirname "$0")/.." source .venv/bin/activate echo "=== Installing Unsloth (this reinstalls torch/deps to Unsloth-matched versions) ===" # Unsloth's documented install for the latest models. --force-reinstall because # Unsloth pins specific torch/xformers/trl builds; on a fresh instance that's # fine, but be aware it may move torch off the version fla was built against — # which is exactly why this is a SEPARATE, opt-in path, not folded into # setup_env.sh. Do NOT run this and setup_fast_kernels.sh into the same venv # expecting both to stay consistent; pick one path per venv. pip install --upgrade --force-reinstall --no-cache-dir unsloth unsloth_zoo echo echo "=== Verify Unsloth imports and sees the GPU ===" python - <<'PY' import torch print("torch:", torch.__version__, "| cuda:", torch.cuda.is_available()) try: from unsloth import FastVisionModel, FastModel # noqa: F401 print("Unsloth import OK (FastVisionModel / FastModel available).") except Exception as e: print(f"!! Unsloth import FAILED: {e!r}") raise SystemExit(1) print("\nOK. Next: python train_sft_unsloth.py --config configs/sft_unsloth.yaml") print("(Re-run the smoke test first — same as the TRL path — to compare s/example.)") PY