Upload folder using huggingface_hub
Browse files- README.md +92 -0
- __pycache__/model.cpython-311.pyc +0 -0
- config.json +18 -0
- generate.py +62 -0
- model.py +109 -0
- model.safetensors +3 -0
- requirements.txt +3 -0
- tokenizer.json +0 -0
README.md
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
library_name: pytorch
|
| 3 |
+
pipeline_tag: text-generation
|
| 4 |
+
language:
|
| 5 |
+
- en
|
| 6 |
+
datasets:
|
| 7 |
+
- roneneldan/TinyStories
|
| 8 |
+
tags:
|
| 9 |
+
- safetensors
|
| 10 |
+
- custom-code
|
| 11 |
+
---
|
| 12 |
+
|
| 13 |
+
# Tiny CED
|
| 14 |
+
|
| 15 |
+
Tiny CED is a small custom PyTorch encoder-decoder language model trained from
|
| 16 |
+
scratch on the English [TinyStories](https://huggingface.co/datasets/roneneldan/TinyStories)
|
| 17 |
+
dataset. This package contains inference-only weights in Safetensors format.
|
| 18 |
+
|
| 19 |
+
This is not a Transformers `PreTrainedModel`. Load it with the included
|
| 20 |
+
`model.py` and `safetensors.torch.load_model` as shown below.
|
| 21 |
+
|
| 22 |
+
## Model details
|
| 23 |
+
|
| 24 |
+
| Item | Value |
|
| 25 |
+
| --- | ---: |
|
| 26 |
+
| Independent parameters | 19,667,712 |
|
| 27 |
+
| Vocabulary size | 8,192 |
|
| 28 |
+
| Hidden size | 384 |
|
| 29 |
+
| Encoder layers | 4 |
|
| 30 |
+
| Decoder layers | 4 |
|
| 31 |
+
| Attention heads | 6 |
|
| 32 |
+
| Feed-forward size | 1,024 |
|
| 33 |
+
| Local attention window | 64 |
|
| 34 |
+
| Trained context length | 512 |
|
| 35 |
+
| Weight dtype | FP32 |
|
| 36 |
+
|
| 37 |
+
The token embedding and output head are tied. `model.safetensors` was exported
|
| 38 |
+
with `safetensors.torch.save_model` so the shared tensor is stored only once.
|
| 39 |
+
|
| 40 |
+
## Usage
|
| 41 |
+
|
| 42 |
+
Install the runtime dependencies:
|
| 43 |
+
|
| 44 |
+
```bash
|
| 45 |
+
pip install -r requirements.txt
|
| 46 |
+
```
|
| 47 |
+
|
| 48 |
+
Run generation on CPU:
|
| 49 |
+
|
| 50 |
+
```bash
|
| 51 |
+
python generate.py \
|
| 52 |
+
--device cpu \
|
| 53 |
+
--text "Once upon a time, a little rabbit" \
|
| 54 |
+
--max-new-tokens 100
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
Use `--device cuda` when CUDA is available. Set `--temperature 0` for greedy
|
| 58 |
+
decoding.
|
| 59 |
+
|
| 60 |
+
## Training and evaluation
|
| 61 |
+
|
| 62 |
+
The exported weights come from the existing `best.pt` checkpoint; no retraining
|
| 63 |
+
was performed during export.
|
| 64 |
+
|
| 65 |
+
- Optimizer: AdamW
|
| 66 |
+
- Peak learning rate: 3e-4
|
| 67 |
+
- Best-checkpoint training tokens: 40,004,782
|
| 68 |
+
- Best-checkpoint step: 5,856
|
| 69 |
+
- Validation loss: 1.896527
|
| 70 |
+
- Validation perplexity: 6.662715
|
| 71 |
+
|
| 72 |
+
Validation used the TinyStories validation split and the same 8,192-token BPE
|
| 73 |
+
tokenizer included in this repository.
|
| 74 |
+
|
| 75 |
+
## Files
|
| 76 |
+
|
| 77 |
+
- `model.safetensors`: inference weights
|
| 78 |
+
- `model.py`: exact custom PyTorch architecture
|
| 79 |
+
- `config.json`: architecture and token IDs
|
| 80 |
+
- `tokenizer.json`: BPE tokenizer
|
| 81 |
+
- `generate.py`: minimal generation example
|
| 82 |
+
|
| 83 |
+
## Limitations
|
| 84 |
+
|
| 85 |
+
This is a small research model trained only on synthetic English children's
|
| 86 |
+
stories. It may generate incorrect, repetitive, biased, unsafe, or otherwise
|
| 87 |
+
inappropriate text. It is not suitable for factual, safety-critical, or
|
| 88 |
+
production use without further evaluation.
|
| 89 |
+
|
| 90 |
+
The source project did not specify a license for its code or weights. Choose an
|
| 91 |
+
appropriate license and confirm the training-data terms before publishing this
|
| 92 |
+
package publicly. TinyStories is listed as CDLA-Sharing-1.0 on its dataset page.
|
__pycache__/model.cpython-311.pyc
ADDED
|
Binary file (10.6 kB). View file
|
|
|
config.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"CED"
|
| 4 |
+
],
|
| 5 |
+
"model_type": "tiny-ced",
|
| 6 |
+
"vocab_size": 8192,
|
| 7 |
+
"hidden_size": 384,
|
| 8 |
+
"num_attention_heads": 6,
|
| 9 |
+
"intermediate_size": 1024,
|
| 10 |
+
"num_hidden_layers": 4,
|
| 11 |
+
"local_window_size": 64,
|
| 12 |
+
"max_position_embeddings": 512,
|
| 13 |
+
"torch_dtype": "float32",
|
| 14 |
+
"pad_token_id": 0,
|
| 15 |
+
"bos_token_id": 1,
|
| 16 |
+
"eos_token_id": 2,
|
| 17 |
+
"unk_token_id": 3
|
| 18 |
+
}
|
generate.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import argparse
|
| 2 |
+
import json
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
|
| 5 |
+
import torch
|
| 6 |
+
from safetensors.torch import load_model
|
| 7 |
+
from tokenizers import Tokenizer
|
| 8 |
+
|
| 9 |
+
from model import CED
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
def main():
|
| 13 |
+
parser = argparse.ArgumentParser()
|
| 14 |
+
parser.add_argument(
|
| 15 |
+
"--model-dir",
|
| 16 |
+
type=Path,
|
| 17 |
+
default=Path(__file__).resolve().parent,
|
| 18 |
+
)
|
| 19 |
+
parser.add_argument("--text", default="Once upon a time, a little rabbit")
|
| 20 |
+
parser.add_argument("--device", choices=["cpu", "cuda"], default="cpu")
|
| 21 |
+
parser.add_argument("--temperature", type=float, default=0.8)
|
| 22 |
+
parser.add_argument("--max-new-tokens", type=int, default=150)
|
| 23 |
+
args = parser.parse_args()
|
| 24 |
+
|
| 25 |
+
config = json.loads((args.model_dir / "config.json").read_text())
|
| 26 |
+
model = CED(
|
| 27 |
+
vocab=config["vocab_size"],
|
| 28 |
+
dim=config["hidden_size"],
|
| 29 |
+
heads=config["num_attention_heads"],
|
| 30 |
+
ff=config["intermediate_size"],
|
| 31 |
+
layers=config["num_hidden_layers"],
|
| 32 |
+
window=config["local_window_size"],
|
| 33 |
+
)
|
| 34 |
+
load_model(model, args.model_dir / "model.safetensors")
|
| 35 |
+
model = model.to(args.device).eval()
|
| 36 |
+
tokenizer = Tokenizer.from_file(str(args.model_dir / "tokenizer.json"))
|
| 37 |
+
|
| 38 |
+
ids = [config["bos_token_id"]] + tokenizer.encode(args.text).ids
|
| 39 |
+
context_length = config["max_position_embeddings"]
|
| 40 |
+
if len(ids) > context_length:
|
| 41 |
+
raise ValueError(f"Prompt exceeds the {context_length}-token context length")
|
| 42 |
+
|
| 43 |
+
torch.manual_seed(123)
|
| 44 |
+
with torch.inference_mode():
|
| 45 |
+
for _ in range(min(args.max_new_tokens, context_length - len(ids))):
|
| 46 |
+
logits = model(torch.tensor([ids], device=args.device))[0, -1].float()
|
| 47 |
+
logits[
|
| 48 |
+
[config["pad_token_id"], config["bos_token_id"], config["unk_token_id"]]
|
| 49 |
+
] = -torch.inf
|
| 50 |
+
if args.temperature <= 0:
|
| 51 |
+
token = int(logits.argmax())
|
| 52 |
+
else:
|
| 53 |
+
token = int(torch.multinomial((logits / args.temperature).softmax(-1), 1))
|
| 54 |
+
ids.append(token)
|
| 55 |
+
if token == config["eos_token_id"]:
|
| 56 |
+
break
|
| 57 |
+
|
| 58 |
+
print(tokenizer.decode(ids))
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
if __name__ == "__main__":
|
| 62 |
+
main()
|
model.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Tiny CED model architecture required to load model.safetensors."""
|
| 2 |
+
|
| 3 |
+
import torch
|
| 4 |
+
import torch.nn.functional as F
|
| 5 |
+
from torch import nn
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
class RMSNorm(nn.Module):
|
| 9 |
+
def __init__(self, dim):
|
| 10 |
+
super().__init__()
|
| 11 |
+
self.weight = nn.Parameter(torch.ones(dim))
|
| 12 |
+
|
| 13 |
+
def forward(self, x):
|
| 14 |
+
z = x.float()
|
| 15 |
+
return (
|
| 16 |
+
z
|
| 17 |
+
* torch.rsqrt(z.square().mean(-1, keepdim=True) + 1e-5)
|
| 18 |
+
* self.weight
|
| 19 |
+
).to(x.dtype)
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def rope(x, pos):
|
| 23 |
+
d = x.shape[-1]
|
| 24 |
+
freq = 10000 ** (torch.arange(0, d, 2, device=x.device) / d)
|
| 25 |
+
angle = pos[:, None, :, None].float() * freq
|
| 26 |
+
a, b = x.float()[..., ::2], x.float()[..., 1::2]
|
| 27 |
+
c, s = angle.cos(), angle.sin()
|
| 28 |
+
return torch.stack((a * c - b * s, a * s + b * c), dim=-1).flatten(-2).to(x.dtype)
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
class Attention(nn.Module):
|
| 32 |
+
def __init__(self, dim, heads, window):
|
| 33 |
+
super().__init__()
|
| 34 |
+
self.heads = heads
|
| 35 |
+
self.window = window
|
| 36 |
+
for name in ["wq", "wkg", "wvg", "wkl", "wvl", "wo"]:
|
| 37 |
+
setattr(self, name, nn.Linear(dim, dim, bias=False))
|
| 38 |
+
|
| 39 |
+
def forward(self, u, source, pos, valid):
|
| 40 |
+
batch, length, dim = u.shape
|
| 41 |
+
|
| 42 |
+
def split(z):
|
| 43 |
+
return z.view(batch, length, self.heads, dim // self.heads).transpose(1, 2)
|
| 44 |
+
|
| 45 |
+
q = rope(split(self.wq(u)), pos)
|
| 46 |
+
kg, vg = rope(split(self.wkg(source)), pos), split(self.wvg(source))
|
| 47 |
+
kl, vl = rope(split(self.wkl(u)), pos), split(self.wvl(u))
|
| 48 |
+
distance = pos[:, :, None] - pos[:, None, :]
|
| 49 |
+
global_mask = (distance >= 0) & valid[:, :, None] & valid[:, None, :]
|
| 50 |
+
local_mask = global_mask & (distance < self.window)
|
| 51 |
+
mask = torch.cat((global_mask, local_mask), -1)[:, None]
|
| 52 |
+
k, v = torch.cat((kg, kl), 2), torch.cat((vg, vl), 2)
|
| 53 |
+
out = F.scaled_dot_product_attention(
|
| 54 |
+
q,
|
| 55 |
+
k,
|
| 56 |
+
v,
|
| 57 |
+
attn_mask=mask,
|
| 58 |
+
dropout_p=0,
|
| 59 |
+
is_causal=False,
|
| 60 |
+
)
|
| 61 |
+
return self.wo(out.transpose(1, 2).contiguous().view(batch, length, dim))
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
class Block(nn.Module):
|
| 65 |
+
def __init__(self, dim, heads, ff, window):
|
| 66 |
+
super().__init__()
|
| 67 |
+
self.attn_norm, self.ffn_norm = RMSNorm(dim), RMSNorm(dim)
|
| 68 |
+
self.attn = Attention(dim, heads, window)
|
| 69 |
+
self.gate, self.up = (nn.Linear(dim, ff, bias=False) for _ in range(2))
|
| 70 |
+
self.down = nn.Linear(ff, dim, bias=False)
|
| 71 |
+
|
| 72 |
+
def forward(self, h, e, pos, valid):
|
| 73 |
+
u = self.attn_norm(h)
|
| 74 |
+
h = h + self.attn(u, u if e is None else e, pos, valid)
|
| 75 |
+
z = self.ffn_norm(h)
|
| 76 |
+
return h + self.down(F.silu(self.gate(z)) * self.up(z))
|
| 77 |
+
|
| 78 |
+
|
| 79 |
+
class CED(nn.Module):
|
| 80 |
+
def __init__(self, vocab=8192, dim=384, heads=6, ff=1024, layers=4, window=64):
|
| 81 |
+
super().__init__()
|
| 82 |
+
assert dim % heads == 0 and (dim // heads) % 2 == 0
|
| 83 |
+
self.embedding = nn.Embedding(vocab, dim)
|
| 84 |
+
self.encoder = nn.ModuleList(
|
| 85 |
+
[Block(dim, heads, ff, window) for _ in range(layers)]
|
| 86 |
+
)
|
| 87 |
+
self.decoder = nn.ModuleList(
|
| 88 |
+
[Block(dim, heads, ff, window) for _ in range(layers)]
|
| 89 |
+
)
|
| 90 |
+
self.encoder_norm, self.final_norm = RMSNorm(dim), RMSNorm(dim)
|
| 91 |
+
self.head = nn.Linear(dim, vocab, bias=False)
|
| 92 |
+
self.head.weight = self.embedding.weight
|
| 93 |
+
for parameter in self.parameters():
|
| 94 |
+
if parameter.ndim > 1:
|
| 95 |
+
nn.init.normal_(parameter, std=0.02)
|
| 96 |
+
|
| 97 |
+
def forward(self, ids, positions=None, valid=None):
|
| 98 |
+
if positions is None:
|
| 99 |
+
positions = torch.arange(ids.shape[1], device=ids.device)[None].expand_as(ids)
|
| 100 |
+
if valid is None:
|
| 101 |
+
valid = torch.ones_like(ids, dtype=torch.bool)
|
| 102 |
+
h = self.embedding(ids)
|
| 103 |
+
for layer in self.encoder:
|
| 104 |
+
h = layer(h, None, positions, valid)
|
| 105 |
+
e = self.encoder_norm(h)
|
| 106 |
+
h = e
|
| 107 |
+
for layer in self.decoder:
|
| 108 |
+
h = layer(h, e, positions, valid)
|
| 109 |
+
return self.head(self.final_norm(h))
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4a60e58abb8fa784857588179cba839ccc76d4d56ef297887c376f3d8a5aacf5
|
| 3 |
+
size 78679560
|
requirements.txt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch>=2.0
|
| 2 |
+
safetensors>=0.4
|
| 3 |
+
tokenizers>=0.20
|
tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|