Text Generation
Transformers
Safetensors
English
llama
micro
nano
small
supra
cpu
cpu-only
microsupra
text-generation-inference
Instructions to use SupraLabs/MicroSupra-10k with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SupraLabs/MicroSupra-10k with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="SupraLabs/MicroSupra-10k")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("SupraLabs/MicroSupra-10k") model = AutoModelForCausalLM.from_pretrained("SupraLabs/MicroSupra-10k", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use SupraLabs/MicroSupra-10k with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "SupraLabs/MicroSupra-10k" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/SupraLabs/MicroSupra-10k
- SGLang
How to use SupraLabs/MicroSupra-10k with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "SupraLabs/MicroSupra-10k" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "SupraLabs/MicroSupra-10k" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use SupraLabs/MicroSupra-10k with Docker Model Runner:
docker model run hf.co/SupraLabs/MicroSupra-10k
File size: 2,823 Bytes
4c70717 0028524 4c70717 0028524 4c70717 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 | """MicroSupra-10k — treino (LLaMA 10.000 params, CPU)."""
import math
import os
import time
import numpy as np
import torch
from transformers import LlamaConfig, LlamaForCausalLM
SEQ = 256
BS = int(os.environ.get("BS", 64))
EPOCHS = int(os.environ.get("EPOCHS", 2))
LR = 5e-3
BASE = os.path.dirname(os.path.abspath(__file__))
OUT = f"{BASE}/out"
os.makedirs(OUT, exist_ok=True)
data = np.memmap(f"{BASE}/data/train.bin", dtype=np.uint16, mode="r")
val = np.memmap(f"{BASE}/data/val.bin", dtype=np.uint16, mode="r")
tokens_per_step = BS * SEQ
steps = int(len(data) // tokens_per_step * EPOCHS)
print(f"[*] {len(data):,} train tokens | {len(val):,} val tokens | {steps} steps x {tokens_per_step} tok/step", flush=True)
cfg = LlamaConfig(
vocab_size=1024, hidden_size=8, intermediate_size=69,
num_hidden_layers=1, num_attention_heads=1, num_key_value_heads=1,
head_dim=4, max_position_embeddings=256, tie_word_embeddings=True,
rms_norm_eps=1e-6, bos_token_id=0, eos_token_id=2, pad_token_id=1,
use_cache=False, attention_bias=False, mlp_bias=False, hidden_act="silu",
)
model = LlamaForCausalLM(cfg)
n = sum(p.numel() for p in model.parameters())
print(f"[*] Parâmetros: {n:,}", flush=True)
assert n == 10000, f"esperado 10000, veio {n}"
def get_batch(split):
d = data if split == "train" else val
ix = torch.randint(len(d) - SEQ - 1, (BS,))
x = torch.stack([torch.from_numpy(np.array(d[i:i + SEQ], dtype=np.int64)) for i in ix])
y = torch.stack([torch.from_numpy(np.array(d[i + 1:i + 1 + SEQ], dtype=np.int64)) for i in ix])
return x, y
opt = torch.optim.AdamW(model.parameters(), lr=LR, betas=(0.9, 0.95), weight_decay=0.01)
warmup = 200
def lr_fn(step):
if step < warmup:
return step / max(1, warmup)
p = (step - warmup) / max(1, steps - warmup)
return 0.1 + 0.9 * 0.5 * (1 + math.cos(math.pi * min(p, 1.0)))
sched = torch.optim.lr_scheduler.LambdaLR(opt, lr_fn)
best = float("inf")
t0 = time.time()
for step in range(steps):
x, y = get_batch("train")
loss = model(input_ids=x, labels=y).loss
loss.backward()
torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)
opt.step(); sched.step(); opt.zero_grad()
if step % 100 == 0 or step == steps - 1:
model.eval()
with torch.no_grad():
vx, vy = get_batch("val")
vl = model(input_ids=vx, labels=vy).loss.item()
model.train()
print(f"step {step:5d}/{steps} | train {loss.item():.4f} | val {vl:.4f} | lr {sched.get_last_lr()[0]:.2e} | {time.time()-t0:.0f}s", flush=True)
if vl < best:
best = vl
model.save_pretrained(OUT)
model.save_pretrained(OUT) # checkpoint final garantido
print(f"[*] FINAL train loss ~{loss.item():.4f} | best val loss {best:.4f}", flush=True)
print("[*] DONE", flush=True) |