Text Generation
Transformers
Safetensors
English
llama
micro
nano
small
supra
cpu
cpu-only
microsupra
text-generation-inference
Instructions to use SupraLabs/MicroSupra-10k with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use SupraLabs/MicroSupra-10k with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="SupraLabs/MicroSupra-10k")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("SupraLabs/MicroSupra-10k") model = AutoModelForCausalLM.from_pretrained("SupraLabs/MicroSupra-10k", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use SupraLabs/MicroSupra-10k with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "SupraLabs/MicroSupra-10k" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/SupraLabs/MicroSupra-10k
- SGLang
How to use SupraLabs/MicroSupra-10k with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "SupraLabs/MicroSupra-10k" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "SupraLabs/MicroSupra-10k" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "SupraLabs/MicroSupra-10k", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use SupraLabs/MicroSupra-10k with Docker Model Runner:
docker model run hf.co/SupraLabs/MicroSupra-10k
Download scripts/train.py from SupraLabs/MicroSupra-10k: direct link, hf CLI and curl.
- Browser
- Download file 2.82 kB
-
https://huggingface.co/SupraLabs/MicroSupra-10k/resolve/main/scripts/train.py
- Command line
-
hf download hf://SupraLabs/MicroSupra-10k/scripts/train.py
-
curl -L -o train.py https://huggingface.co/SupraLabs/MicroSupra-10k/resolve/main/scripts/train.py
2.82 kB
| """MicroSupra-10k — treino (LLaMA 10.000 params, CPU).""" | |
| import math | |
| import os | |
| import time | |
| import numpy as np | |
| import torch | |
| from transformers import LlamaConfig, LlamaForCausalLM | |
| SEQ = 256 | |
| BS = int(os.environ.get("BS", 64)) | |
| EPOCHS = int(os.environ.get("EPOCHS", 2)) | |
| LR = 5e-3 | |
| BASE = os.path.dirname(os.path.abspath(__file__)) | |
| OUT = f"{BASE}/out" | |
| os.makedirs(OUT, exist_ok=True) | |
| data = np.memmap(f"{BASE}/data/train.bin", dtype=np.uint16, mode="r") | |
| val = np.memmap(f"{BASE}/data/val.bin", dtype=np.uint16, mode="r") | |
| tokens_per_step = BS * SEQ | |
| steps = int(len(data) // tokens_per_step * EPOCHS) | |
| print(f"[*] {len(data):,} train tokens | {len(val):,} val tokens | {steps} steps x {tokens_per_step} tok/step", flush=True) | |
| cfg = LlamaConfig( | |
| vocab_size=1024, hidden_size=8, intermediate_size=69, | |
| num_hidden_layers=1, num_attention_heads=1, num_key_value_heads=1, | |
| head_dim=4, max_position_embeddings=256, tie_word_embeddings=True, | |
| rms_norm_eps=1e-6, bos_token_id=0, eos_token_id=2, pad_token_id=1, | |
| use_cache=False, attention_bias=False, mlp_bias=False, hidden_act="silu", | |
| ) | |
| model = LlamaForCausalLM(cfg) | |
| n = sum(p.numel() for p in model.parameters()) | |
| print(f"[*] Parâmetros: {n:,}", flush=True) | |
| assert n == 10000, f"esperado 10000, veio {n}" | |
| def get_batch(split): | |
| d = data if split == "train" else val | |
| ix = torch.randint(len(d) - SEQ - 1, (BS,)) | |
| x = torch.stack([torch.from_numpy(np.array(d[i:i + SEQ], dtype=np.int64)) for i in ix]) | |
| y = torch.stack([torch.from_numpy(np.array(d[i + 1:i + 1 + SEQ], dtype=np.int64)) for i in ix]) | |
| return x, y | |
| opt = torch.optim.AdamW(model.parameters(), lr=LR, betas=(0.9, 0.95), weight_decay=0.01) | |
| warmup = 200 | |
| def lr_fn(step): | |
| if step < warmup: | |
| return step / max(1, warmup) | |
| p = (step - warmup) / max(1, steps - warmup) | |
| return 0.1 + 0.9 * 0.5 * (1 + math.cos(math.pi * min(p, 1.0))) | |
| sched = torch.optim.lr_scheduler.LambdaLR(opt, lr_fn) | |
| best = float("inf") | |
| t0 = time.time() | |
| for step in range(steps): | |
| x, y = get_batch("train") | |
| loss = model(input_ids=x, labels=y).loss | |
| loss.backward() | |
| torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0) | |
| opt.step(); sched.step(); opt.zero_grad() | |
| if step % 100 == 0 or step == steps - 1: | |
| model.eval() | |
| with torch.no_grad(): | |
| vx, vy = get_batch("val") | |
| vl = model(input_ids=vx, labels=vy).loss.item() | |
| model.train() | |
| print(f"step {step:5d}/{steps} | train {loss.item():.4f} | val {vl:.4f} | lr {sched.get_last_lr()[0]:.2e} | {time.time()-t0:.0f}s", flush=True) | |
| if vl < best: | |
| best = vl | |
| model.save_pretrained(OUT) | |
| model.save_pretrained(OUT) # checkpoint final garantido | |
| print(f"[*] FINAL train loss ~{loss.item():.4f} | best val loss {best:.4f}", flush=True) | |
| print("[*] DONE", flush=True) |