MicroSupra-10k / scripts /train.py
DedeProGames's picture
fix: config de 10.000 params exatos (head_dim=4, intermediate=69)
0028524 verified
Raw History Blame Contribute Delete
2.82 kB
"""MicroSupra-10k — treino (LLaMA 10.000 params, CPU)."""
import math
import os
import time
import numpy as np
import torch
from transformers import LlamaConfig, LlamaForCausalLM
SEQ = 256
BS = int(os.environ.get("BS", 64))
EPOCHS = int(os.environ.get("EPOCHS", 2))
LR = 5e-3
BASE = os.path.dirname(os.path.abspath(__file__))
OUT = f"{BASE}/out"
os.makedirs(OUT, exist_ok=True)
data = np.memmap(f"{BASE}/data/train.bin", dtype=np.uint16, mode="r")
val = np.memmap(f"{BASE}/data/val.bin", dtype=np.uint16, mode="r")
tokens_per_step = BS * SEQ
steps = int(len(data) // tokens_per_step * EPOCHS)
print(f"[*] {len(data):,} train tokens | {len(val):,} val tokens | {steps} steps x {tokens_per_step} tok/step", flush=True)
cfg = LlamaConfig(
vocab_size=1024, hidden_size=8, intermediate_size=69,
num_hidden_layers=1, num_attention_heads=1, num_key_value_heads=1,
head_dim=4, max_position_embeddings=256, tie_word_embeddings=True,
rms_norm_eps=1e-6, bos_token_id=0, eos_token_id=2, pad_token_id=1,
use_cache=False, attention_bias=False, mlp_bias=False, hidden_act="silu",
)
model = LlamaForCausalLM(cfg)
n = sum(p.numel() for p in model.parameters())
print(f"[*] Parâmetros: {n:,}", flush=True)
assert n == 10000, f"esperado 10000, veio {n}"
def get_batch(split):
d = data if split == "train" else val
ix = torch.randint(len(d) - SEQ - 1, (BS,))
x = torch.stack([torch.from_numpy(np.array(d[i:i + SEQ], dtype=np.int64)) for i in ix])
y = torch.stack([torch.from_numpy(np.array(d[i + 1:i + 1 + SEQ], dtype=np.int64)) for i in ix])
return x, y
opt = torch.optim.AdamW(model.parameters(), lr=LR, betas=(0.9, 0.95), weight_decay=0.01)
warmup = 200
def lr_fn(step):
if step < warmup:
return step / max(1, warmup)
p = (step - warmup) / max(1, steps - warmup)
return 0.1 + 0.9 * 0.5 * (1 + math.cos(math.pi * min(p, 1.0)))
sched = torch.optim.lr_scheduler.LambdaLR(opt, lr_fn)
best = float("inf")
t0 = time.time()
for step in range(steps):
x, y = get_batch("train")
loss = model(input_ids=x, labels=y).loss
loss.backward()
torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)
opt.step(); sched.step(); opt.zero_grad()
if step % 100 == 0 or step == steps - 1:
model.eval()
with torch.no_grad():
vx, vy = get_batch("val")
vl = model(input_ids=vx, labels=vy).loss.item()
model.train()
print(f"step {step:5d}/{steps} | train {loss.item():.4f} | val {vl:.4f} | lr {sched.get_last_lr()[0]:.2e} | {time.time()-t0:.0f}s", flush=True)
if vl < best:
best = vl
model.save_pretrained(OUT)
model.save_pretrained(OUT) # checkpoint final garantido
print(f"[*] FINAL train loss ~{loss.item():.4f} | best val loss {best:.4f}", flush=True)
print("[*] DONE", flush=True)