Text Generation
Transformers
Safetensors
cortex
tiny
spanish
bilingual
causal-lm
from-scratch
conversational
custom_code
Instructions to use Ilides/cortex-0.3-0.02b with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Ilides/cortex-0.3-0.02b with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="Ilides/cortex-0.3-0.02b", trust_remote_code=True) messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("Ilides/cortex-0.3-0.02b", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use Ilides/cortex-0.3-0.02b with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "Ilides/cortex-0.3-0.02b" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Ilides/cortex-0.3-0.02b", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/Ilides/cortex-0.3-0.02b
- SGLang
How to use Ilides/cortex-0.3-0.02b with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "Ilides/cortex-0.3-0.02b" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Ilides/cortex-0.3-0.02b", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "Ilides/cortex-0.3-0.02b" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "Ilides/cortex-0.3-0.02b", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use Ilides/cortex-0.3-0.02b with Docker Model Runner:
docker model run hf.co/Ilides/cortex-0.3-0.02b
Download modeling_cortex.py from Ilides/cortex-0.3-0.02b: direct link, hf CLI and curl.
- Browser
- Download file 6.64 kB
-
https://huggingface.co/Ilides/cortex-0.3-0.02b/resolve/main/modeling_cortex.py
- Command line
-
hf download hf://Ilides/cortex-0.3-0.02b/modeling_cortex.py
-
curl -L -o modeling_cortex.py https://huggingface.co/Ilides/cortex-0.3-0.02b/resolve/main/modeling_cortex.py
6.64 kB
| """Cortex-1 en PyTorch — implementación canónica equivalente al modelo NumPy. | |
| Mapeo de pesos (NumPy -> PyTorch), verificado por test de equivalencia de | |
| logits (max|Δ| < 1e-4 en float32): | |
| tok_emb.w -> model.embed_tokens.weight (vocab, d) | |
| pos_emb.w -> model.embed_positions.weight (ctx, d) | |
| blocks.i.attn_norm.g -> model.layers.i.input_layernorm.weight | |
| blocks.i.attn.wqkv -> model.layers.i.self_attn.qkv.weight (transpuesta) | |
| blocks.i.attn.wo -> model.layers.i.self_attn.proj.weight (transpuesta) | |
| blocks.i.mlp_norm.g -> model.layers.i.post_attention_layernorm.weight | |
| blocks.i.mlp.w1 -> model.layers.i.mlp.gate_proj.weight (transpuesta) | |
| blocks.i.mlp.w3 -> model.layers.i.mlp.up_proj.weight (transpuesta) | |
| blocks.i.mlp.w2 -> model.layers.i.mlp.down_proj.weight (transpuesta) | |
| final_norm.g -> model.norm.weight | |
| (lm_head atado a tok_emb — tie_word_embeddings=True) | |
| Arquitectura: GPT pre-norm, RMSNorm sin sesgos, atención causal multi-cabeza | |
| con QKV fusionado y SwiGLU — sin dropout ni stochasticidad: eval == generate. | |
| """ | |
| from __future__ import annotations | |
| import torch | |
| import torch.nn as nn | |
| import torch.nn.functional as F | |
| from .configuration_cortex import CortexConfig | |
| try: | |
| from transformers.modeling_utils import PreTrainedModel | |
| from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast | |
| from transformers import GenerationMixin | |
| except ImportError: # transformers antiguo | |
| from transformers import PreTrainedModel, GenerationMixin | |
| from transformers.modeling_outputs import BaseModelOutputWithPast, CausalLMOutputWithPast | |
| class RMSNorm(nn.Module): | |
| def __init__(self, dim: int, eps: float): | |
| super().__init__() | |
| self.eps = eps | |
| self.weight = nn.Parameter(torch.ones(dim)) | |
| def forward(self, x): | |
| norm = x.float() * torch.rsqrt(x.float().pow(2).mean(-1, keepdim=True) + self.eps) | |
| return (norm * self.weight.float()).to(x.dtype) | |
| class CortexAttention(nn.Module): | |
| """Atención causal con QKV fusionado (una GEMM), como el original NumPy.""" | |
| def __init__(self, cfg: CortexConfig): | |
| super().__init__() | |
| self.n_heads = cfg.num_attention_heads | |
| self.d_head = cfg.hidden_size // cfg.num_attention_heads | |
| self.scale = self.d_head ** -0.5 | |
| self.qkv = nn.Linear(cfg.hidden_size, 3 * cfg.hidden_size, bias=False) | |
| self.proj = nn.Linear(cfg.hidden_size, cfg.hidden_size, bias=False) | |
| def forward(self, x): | |
| B, T, d = x.shape | |
| qkv = self.qkv(x).view(B, T, 3, self.n_heads, self.d_head) | |
| q, k, v = qkv.unbind(dim=2) # (B, T, H, dh) | |
| q = q.transpose(1, 2) # (B, H, T, dh) | |
| k = k.transpose(1, 2) | |
| v = v.transpose(1, 2) | |
| att = (q @ k.transpose(-2, -1)) * self.scale # (B, H, T, T) | |
| mask = torch.triu(torch.full((T, T), float("-inf"), device=x.device, dtype=att.dtype), 1) | |
| att = att + mask | |
| att = att.softmax(dim=-1) | |
| y = (att @ v).transpose(1, 2).reshape(B, T, d) # reensambla cabezas | |
| return self.proj(y) | |
| class CortexMLP(nn.Module): | |
| """SwiGLU: down(silu(gate(x)) * up(x)) — w1=gate, w3=up, w2=down.""" | |
| def __init__(self, cfg: CortexConfig): | |
| super().__init__() | |
| self.gate_proj = nn.Linear(cfg.hidden_size, cfg.intermediate_size, bias=False) | |
| self.up_proj = nn.Linear(cfg.hidden_size, cfg.intermediate_size, bias=False) | |
| self.down_proj = nn.Linear(cfg.intermediate_size, cfg.hidden_size, bias=False) | |
| def forward(self, x): | |
| return self.down_proj(F.silu(self.gate_proj(x)) * self.up_proj(x)) | |
| class CortexBlock(nn.Module): | |
| def __init__(self, cfg: CortexConfig): | |
| super().__init__() | |
| self.input_layernorm = RMSNorm(cfg.hidden_size, cfg.rms_norm_eps) | |
| self.self_attn = CortexAttention(cfg) | |
| self.post_attention_layernorm = RMSNorm(cfg.hidden_size, cfg.rms_norm_eps) | |
| self.mlp = CortexMLP(cfg) | |
| def forward(self, x): | |
| x = x + self.self_attn(self.input_layernorm(x)) | |
| x = x + self.mlp(self.post_attention_layernorm(x)) | |
| return x | |
| class CortexModel(nn.Module): | |
| def __init__(self, cfg: CortexConfig): | |
| super().__init__() | |
| self.embed_tokens = nn.Embedding(cfg.vocab_size, cfg.hidden_size) | |
| self.embed_positions = nn.Embedding(cfg.max_position_embeddings, cfg.hidden_size) | |
| self.layers = nn.ModuleList(CortexBlock(cfg) for _ in range(cfg.num_hidden_layers)) | |
| self.norm = RMSNorm(cfg.hidden_size, cfg.rms_norm_eps) | |
| def forward(self, input_ids): | |
| T = input_ids.shape[1] | |
| h = self.embed_tokens(input_ids) + self.embed_positions(torch.arange(T, device=input_ids.device)) | |
| for layer in self.layers: | |
| h = layer(h) | |
| return self.norm(h) | |
| class CortexForCausalLM(PreTrainedModel, GenerationMixin): | |
| config_class = CortexConfig | |
| _tied_weights_keys = ["lm_head.weight"] | |
| _dynamic_tied_weights_keys = ["lm_head.weight"] | |
| def __init__(self, cfg: CortexConfig): | |
| super().__init__(cfg) | |
| self.model = CortexModel(cfg) | |
| self.lm_head = nn.Linear(cfg.hidden_size, cfg.vocab_size, bias=False) | |
| if cfg.tie_word_embeddings: | |
| self.lm_head.weight = self.model.embed_tokens.weight | |
| def forward(self, input_ids, attention_mask=None, labels=None, | |
| output_hidden_states=False, use_cache=False, **kwargs): | |
| h = self.model(input_ids) | |
| logits = self.lm_head(h) | |
| loss = None | |
| if labels is not None: | |
| shift_logits = logits[..., :-1, :].contiguous() | |
| shift_labels = labels[..., 1:].contiguous() | |
| loss = F.cross_entropy( | |
| shift_logits.view(-1, shift_logits.size(-1)), | |
| shift_labels.view(-1), | |
| ignore_index=-100, | |
| ) | |
| return CausalLMOutputWithPast( | |
| loss=loss, | |
| logits=logits, | |
| hidden_states=(h,) if output_hidden_states else None, | |
| ) | |
| def _init_weights(module): | |
| if isinstance(module, nn.Linear): | |
| nn.init.normal_(module.weight, mean=0.0, std=0.02) | |
| if module.bias is not None: | |
| nn.init.zeros_(module.bias) | |
| elif isinstance(module, nn.Embedding): | |
| nn.init.normal_(module.weight, mean=0.0, std=0.02) | |
| def prepare_inputs_for_generation(self, input_ids, **kwargs): | |
| return {"input_ids": input_ids} | |