coder-fake / scripts /setup_unsloth.sh
halle01's picture
Add files using upload-large-folder tool
78cb49e verified
Raw History Blame Contribute Delete
2.52 kB
#!/bin/bash
# FALLBACK training path β€” install Unsloth. Use this only if the primary
# TRL + flash-linear-attention path (scripts/setup_fast_kernels.sh) hits
# trouble on the new instance. Unsloth ships its own fused Gated-DeltaNet /
# MoE Triton kernels (so it doesn't depend on fla) and claims ~1.5-1.7x
# faster / 50-60% less VRAM vs FA2. See SPEEDUP_FINDINGS.md's "TRL vs Unsloth".
#
# READ THIS BEFORE USING β€” two real caveats specific to our model on 48GB:
# 1. Unsloth EXPLICITLY recommends AGAINST 4-bit QLoRA for Qwen3.5 ("not
# recommended to do QLoRA (4-bit) ... due to higher-than-normal
# quantization differences", and "MoE QLoRA 4-bit not recommended due to
# BitsandBytes limitations"). It steers you to bf16 LoRA.
# 2. But their own VRAM table puts 27B bf16 LoRA at ~56GB β€” OVER our 48GB.
# So on a single 48GB card you're forced to either (a) run 4-bit anyway,
# against their advice, and validate output quality hard on the eval set, or
# (b) drop to a smaller Qwen3.5 variant that fits bf16 (their table: 9B=22GB,
# 4B=10GB). The configs/*_unsloth.yaml default to 4-bit with that warning;
# flip load_in_4bit->false + pick a smaller base if quality is unacceptable.
#
# Run AFTER scripts/setup_env.sh, in the same venv.
set -euo pipefail
cd "$(dirname "$0")/.."
source .venv/bin/activate
echo "=== Installing Unsloth (this reinstalls torch/deps to Unsloth-matched versions) ==="
# Unsloth's documented install for the latest models. --force-reinstall because
# Unsloth pins specific torch/xformers/trl builds; on a fresh instance that's
# fine, but be aware it may move torch off the version fla was built against β€”
# which is exactly why this is a SEPARATE, opt-in path, not folded into
# setup_env.sh. Do NOT run this and setup_fast_kernels.sh into the same venv
# expecting both to stay consistent; pick one path per venv.
pip install --upgrade --force-reinstall --no-cache-dir unsloth unsloth_zoo
echo
echo "=== Verify Unsloth imports and sees the GPU ==="
python - <<'PY'
import torch
print("torch:", torch.__version__, "| cuda:", torch.cuda.is_available())
try:
from unsloth import FastVisionModel, FastModel # noqa: F401
print("Unsloth import OK (FastVisionModel / FastModel available).")
except Exception as e:
print(f"!! Unsloth import FAILED: {e!r}")
raise SystemExit(1)
print("\nOK. Next: python train_sft_unsloth.py --config configs/sft_unsloth.yaml")
print("(Re-run the smoke test first β€” same as the TRL path β€” to compare s/example.)")
PY