File size: 2,823 Bytes
4c70717
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
0028524
4c70717
0028524
4c70717
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
"""MicroSupra-10k — treino (LLaMA 10.000 params, CPU)."""
import math
import os
import time
import numpy as np
import torch
from transformers import LlamaConfig, LlamaForCausalLM

SEQ = 256
BS = int(os.environ.get("BS", 64))
EPOCHS = int(os.environ.get("EPOCHS", 2))
LR = 5e-3
BASE = os.path.dirname(os.path.abspath(__file__))
OUT = f"{BASE}/out"
os.makedirs(OUT, exist_ok=True)

data = np.memmap(f"{BASE}/data/train.bin", dtype=np.uint16, mode="r")
val = np.memmap(f"{BASE}/data/val.bin", dtype=np.uint16, mode="r")
tokens_per_step = BS * SEQ
steps = int(len(data) // tokens_per_step * EPOCHS)
print(f"[*] {len(data):,} train tokens | {len(val):,} val tokens | {steps} steps x {tokens_per_step} tok/step", flush=True)

cfg = LlamaConfig(
    vocab_size=1024, hidden_size=8, intermediate_size=69,
    num_hidden_layers=1, num_attention_heads=1, num_key_value_heads=1,
    head_dim=4, max_position_embeddings=256, tie_word_embeddings=True,
    rms_norm_eps=1e-6, bos_token_id=0, eos_token_id=2, pad_token_id=1,
    use_cache=False, attention_bias=False, mlp_bias=False, hidden_act="silu",
)
model = LlamaForCausalLM(cfg)
n = sum(p.numel() for p in model.parameters())
print(f"[*] Parâmetros: {n:,}", flush=True)
assert n == 10000, f"esperado 10000, veio {n}"

def get_batch(split):
    d = data if split == "train" else val
    ix = torch.randint(len(d) - SEQ - 1, (BS,))
    x = torch.stack([torch.from_numpy(np.array(d[i:i + SEQ], dtype=np.int64)) for i in ix])
    y = torch.stack([torch.from_numpy(np.array(d[i + 1:i + 1 + SEQ], dtype=np.int64)) for i in ix])
    return x, y

opt = torch.optim.AdamW(model.parameters(), lr=LR, betas=(0.9, 0.95), weight_decay=0.01)
warmup = 200
def lr_fn(step):
    if step < warmup:
        return step / max(1, warmup)
    p = (step - warmup) / max(1, steps - warmup)
    return 0.1 + 0.9 * 0.5 * (1 + math.cos(math.pi * min(p, 1.0)))
sched = torch.optim.lr_scheduler.LambdaLR(opt, lr_fn)

best = float("inf")
t0 = time.time()
for step in range(steps):
    x, y = get_batch("train")
    loss = model(input_ids=x, labels=y).loss
    loss.backward()
    torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)
    opt.step(); sched.step(); opt.zero_grad()
    if step % 100 == 0 or step == steps - 1:
        model.eval()
        with torch.no_grad():
            vx, vy = get_batch("val")
            vl = model(input_ids=vx, labels=vy).loss.item()
        model.train()
        print(f"step {step:5d}/{steps} | train {loss.item():.4f} | val {vl:.4f} | lr {sched.get_last_lr()[0]:.2e} | {time.time()-t0:.0f}s", flush=True)
        if vl < best:
            best = vl
            model.save_pretrained(OUT)

model.save_pretrained(OUT)  # checkpoint final garantido
print(f"[*] FINAL train loss ~{loss.item():.4f} | best val loss {best:.4f}", flush=True)
print("[*] DONE", flush=True)