File size: 3,463 Bytes
6c223ff
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
#!/usr/bin/env bash
# Fractus 1B GPU Deployment Script
# Run this on the GPU machine after SSH'ing in.
#
# Usage:
#   bash deploy_gpu.sh
#
# What it does:
#   1. Installs dependencies
#   2. Clones fractus-cte + downloads datasets from HF
#   3. Trains paliers 0-3 on CPU (if no checkpoint)
#   4. Grows to 1B + trains on GPU
#
set -e

echo "============================================"
echo "  FRACTUS 1B GPU DEPLOYMENT"
echo "============================================"

# 1. Check GPU
echo ""
echo "=== GPU Check ==="
nvidia-smi --query-gpu=name,memory.total --format=csv,noheader
echo ""

# 1b. Preflight: disk + HF token (catch problems before the long download/train).
echo "=== Preflight ==="
FREE_GB=$(df -BG . 2>/dev/null | awk 'NR==2{print $4}' | tr -d 'G')
if [ -n "$FREE_GB" ] && [ "$FREE_GB" -lt 20 ]; then
    echo "WARNING: only ${FREE_GB}GB free on disk — need ~20GB (3GB datasets + 4GB corpus + ~8GB checkpoints)."
    echo "         Continuing, but the run may fail mid-training."
else
    echo "  disk: ${FREE_GB:-?}GB free (need ~20GB) — OK"
fi
if [ -z "$HF_TOKEN" ]; then
    echo "  HF_TOKEN: NOT SET — the 1B checkpoint upload to HF will be skipped."
    echo "           Set it with: export HF_TOKEN=hf_xxx (from https://huggingface.co/settings/tokens)"
else
    echo "  HF_TOKEN: set — 1B will upload to thefinalboss/Fractus-1B"
fi
python -c "import sys; print(f'  python: {sys.version.split()[0]}')" || echo "  python: MISSING"
echo ""

# 2. Install dependencies
echo "=== Installing dependencies ==="
pip install torch numpy tokenizers matplotlib huggingface_hub --quiet
echo "Done"

# 3. Clone fractus-cte
echo ""
echo "=== Cloning fractus-cte ==="
git clone https://github.com/AFKmoney/fractus-cte.git
cd fractus-cte

# 4. Download datasets from HF
echo ""
echo "=== Downloading datasets from HF ==="
python -c "
from huggingface_hub import snapshot_download
import os
os.makedirs('data', exist_ok=True)
snapshot_download(
    repo_id='thefinalboss/fractus-datasets',
    repo_type='dataset',
    local_dir='data/hf_datasets')
print('Datasets downloaded')
"

# 5. Build combined corpus from ALL datasets (streaming tokenize of every .jsonl)
echo ""
echo "=== Building training corpus ==="
python scripts/build_corpus.py --src data/hf_datasets --out data/training_corpus.pt --cap 1000000000

# 6. Train paliers 0-3 (GPU if available, else CPU — auto-detected)
if [ ! -f "checkpoints/fractus_palier3.pt" ]; then
    echo ""
    echo "=== Training paliers 0-3 (auto: GPU if present) ==="
    export FRACTUS_CORPUS="data/training_corpus.pt"
    python scripts/train_progressive.py --paliers 0,1,2,3 --accumulation-steps 8
else
    echo ""
    echo "=== Checkpoint exists, skipping paliers 0-3 ==="
fi

# 7. Grow to 1B + train on GPU
echo ""
echo "=== GPU Training: 1B ==="
echo "Loading palier 3 checkpoint, growing to 1B..."
python scripts/train_1b_gpu.py \
    --checkpoint checkpoints/fractus_palier3.pt \
    --tokens 2000000000 \
    --batch-size 8 \
    --bf16 \
    --accumulation-steps 4 \
    --corpus data/training_corpus.pt

echo ""
echo "============================================"
echo "  FRACTUS 1B TRAINING COMPLETE"
echo "============================================"
echo ""
echo "Checkpoint: checkpoints/fractus_1b_gpu.pt"
echo "Run 'python -c \"from fractus.continuous_engine import ContinuousThoughtEngine; e = ContinuousThoughtEngine.from_pretrained(\\\"checkpoints/fractus_1b_gpu.pt\\\")\"' to load."