Bobic 1.0 (11.6M, Char-Level)
Bobic 1.0 is the original foundation model of the Bobic series, featuring an ultra-lightweight 11.65 million parameter character-level autoregressive transformer.
Trained directly on Telegram chat export logs, synthetic physics reasoning questions, and basic mathematical operations.
Technical Specifications
| Parameter | Value |
|---|---|
| Parameters | 11,653,533 (~11.6M) |
| Tokenization | Character-level (vocab=669 unique characters) |
Layers (L) |
6 transformer blocks |
Embedding Size (N) |
384 hidden dimensions |
Context Window (BLOCK) |
256 characters |
| Attention Heads | 6 multi-head attention |
| Weight Format | Full Precision FP32 PyTorch (bobic_1.0.pt, unquantized) |
Quickstart & Usage (PyTorch)
import torch
import torch.nn as nn
import torch.nn.functional as F
# 1. Load Checkpoint
ck = torch.load('bobic_1.0.pt', map_location='cpu')
c2i, chars = ck['c2i'], ck['i2c']
V, BLOCK, N, L = ck['cfg']['V'], ck['cfg']['BLOCK'], ck['cfg']['N'], ck['cfg']['L']
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
# 2. Define Architecture
class Block(nn.Module):
def __init__(self):
super().__init__()
self.sa = nn.MultiheadAttention(N, 6, batch_first=True)
self.ff = nn.Sequential(nn.Linear(N, 4 * N), nn.GELU(), nn.Linear(4 * N, N))
self.ln1, self.ln2 = nn.LayerNorm(N), nn.LayerNorm(N)
self.register_buffer('mask', torch.triu(torch.full((BLOCK, BLOCK), float('-inf')), diagonal=1))
def forward(self, x):
h = self.ln1(x)
a, _ = self.sa(h, h, h, attn_mask=self.mask[:x.size(1), :x.size(1)], need_weights=False)
return x + a + self.ff(self.ln2(x + a))
class GPT(nn.Module):
def __init__(self):
super().__init__()
self.tok = nn.Embedding(V, N)
self.pos = nn.Embedding(BLOCK, N)
self.blocks = nn.ModuleList([Block() for _ in range(L)])
self.ln = nn.LayerNorm(N)
self.head = nn.Linear(N, V)
def forward(self, x):
h = self.tok(x) + self.pos(torch.arange(x.size(1), device=x.device))
for b in self.blocks: h = b(h)
return self.head(self.ln(h))
# 3. Instantiate and Run Inference
model = GPT().to(device)
model.load_state_dict(ck['model'])
model.eval()
prompt = "User: привет\nBobic:"
ids = [c2i.get(c, 0) for c in prompt][-BLOCK:]
for _ in range(35):
ctx = torch.tensor([ids[-BLOCK:]], device=device)
lg = model(ctx)[0, -1] / 0.4
nxt = int(lg.argmax())
ids.append(nxt)
print(''.join(chars[i] for i in ids))
License
Released under the MIT License. Created by Skebobic.