Ne30Charm commited on
Commit
d5b9248
·
verified ·
1 Parent(s): 75b4e9e

Upload folder using huggingface_hub

Browse files
README.md ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: pytorch
3
+ pipeline_tag: text-generation
4
+ language:
5
+ - en
6
+ datasets:
7
+ - roneneldan/TinyStories
8
+ tags:
9
+ - safetensors
10
+ - custom-code
11
+ ---
12
+
13
+ # Tiny CED
14
+
15
+ Tiny CED is a small custom PyTorch encoder-decoder language model trained from
16
+ scratch on the English [TinyStories](https://huggingface.co/datasets/roneneldan/TinyStories)
17
+ dataset. This package contains inference-only weights in Safetensors format.
18
+
19
+ This is not a Transformers `PreTrainedModel`. Load it with the included
20
+ `model.py` and `safetensors.torch.load_model` as shown below.
21
+
22
+ ## Model details
23
+
24
+ | Item | Value |
25
+ | --- | ---: |
26
+ | Independent parameters | 19,667,712 |
27
+ | Vocabulary size | 8,192 |
28
+ | Hidden size | 384 |
29
+ | Encoder layers | 4 |
30
+ | Decoder layers | 4 |
31
+ | Attention heads | 6 |
32
+ | Feed-forward size | 1,024 |
33
+ | Local attention window | 64 |
34
+ | Trained context length | 512 |
35
+ | Weight dtype | FP32 |
36
+
37
+ The token embedding and output head are tied. `model.safetensors` was exported
38
+ with `safetensors.torch.save_model` so the shared tensor is stored only once.
39
+
40
+ ## Usage
41
+
42
+ Install the runtime dependencies:
43
+
44
+ ```bash
45
+ pip install -r requirements.txt
46
+ ```
47
+
48
+ Run generation on CPU:
49
+
50
+ ```bash
51
+ python generate.py \
52
+ --device cpu \
53
+ --text "Once upon a time, a little rabbit" \
54
+ --max-new-tokens 100
55
+ ```
56
+
57
+ Use `--device cuda` when CUDA is available. Set `--temperature 0` for greedy
58
+ decoding.
59
+
60
+ ## Training and evaluation
61
+
62
+ The exported weights come from the existing `best.pt` checkpoint; no retraining
63
+ was performed during export.
64
+
65
+ - Optimizer: AdamW
66
+ - Peak learning rate: 3e-4
67
+ - Best-checkpoint training tokens: 40,004,782
68
+ - Best-checkpoint step: 5,856
69
+ - Validation loss: 1.896527
70
+ - Validation perplexity: 6.662715
71
+
72
+ Validation used the TinyStories validation split and the same 8,192-token BPE
73
+ tokenizer included in this repository.
74
+
75
+ ## Files
76
+
77
+ - `model.safetensors`: inference weights
78
+ - `model.py`: exact custom PyTorch architecture
79
+ - `config.json`: architecture and token IDs
80
+ - `tokenizer.json`: BPE tokenizer
81
+ - `generate.py`: minimal generation example
82
+
83
+ ## Limitations
84
+
85
+ This is a small research model trained only on synthetic English children's
86
+ stories. It may generate incorrect, repetitive, biased, unsafe, or otherwise
87
+ inappropriate text. It is not suitable for factual, safety-critical, or
88
+ production use without further evaluation.
89
+
90
+ The source project did not specify a license for its code or weights. Choose an
91
+ appropriate license and confirm the training-data terms before publishing this
92
+ package publicly. TinyStories is listed as CDLA-Sharing-1.0 on its dataset page.
__pycache__/model.cpython-311.pyc ADDED
Binary file (10.6 kB). View file
 
config.json ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "CED"
4
+ ],
5
+ "model_type": "tiny-ced",
6
+ "vocab_size": 8192,
7
+ "hidden_size": 384,
8
+ "num_attention_heads": 6,
9
+ "intermediate_size": 1024,
10
+ "num_hidden_layers": 4,
11
+ "local_window_size": 64,
12
+ "max_position_embeddings": 512,
13
+ "torch_dtype": "float32",
14
+ "pad_token_id": 0,
15
+ "bos_token_id": 1,
16
+ "eos_token_id": 2,
17
+ "unk_token_id": 3
18
+ }
generate.py ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import argparse
2
+ import json
3
+ from pathlib import Path
4
+
5
+ import torch
6
+ from safetensors.torch import load_model
7
+ from tokenizers import Tokenizer
8
+
9
+ from model import CED
10
+
11
+
12
+ def main():
13
+ parser = argparse.ArgumentParser()
14
+ parser.add_argument(
15
+ "--model-dir",
16
+ type=Path,
17
+ default=Path(__file__).resolve().parent,
18
+ )
19
+ parser.add_argument("--text", default="Once upon a time, a little rabbit")
20
+ parser.add_argument("--device", choices=["cpu", "cuda"], default="cpu")
21
+ parser.add_argument("--temperature", type=float, default=0.8)
22
+ parser.add_argument("--max-new-tokens", type=int, default=150)
23
+ args = parser.parse_args()
24
+
25
+ config = json.loads((args.model_dir / "config.json").read_text())
26
+ model = CED(
27
+ vocab=config["vocab_size"],
28
+ dim=config["hidden_size"],
29
+ heads=config["num_attention_heads"],
30
+ ff=config["intermediate_size"],
31
+ layers=config["num_hidden_layers"],
32
+ window=config["local_window_size"],
33
+ )
34
+ load_model(model, args.model_dir / "model.safetensors")
35
+ model = model.to(args.device).eval()
36
+ tokenizer = Tokenizer.from_file(str(args.model_dir / "tokenizer.json"))
37
+
38
+ ids = [config["bos_token_id"]] + tokenizer.encode(args.text).ids
39
+ context_length = config["max_position_embeddings"]
40
+ if len(ids) > context_length:
41
+ raise ValueError(f"Prompt exceeds the {context_length}-token context length")
42
+
43
+ torch.manual_seed(123)
44
+ with torch.inference_mode():
45
+ for _ in range(min(args.max_new_tokens, context_length - len(ids))):
46
+ logits = model(torch.tensor([ids], device=args.device))[0, -1].float()
47
+ logits[
48
+ [config["pad_token_id"], config["bos_token_id"], config["unk_token_id"]]
49
+ ] = -torch.inf
50
+ if args.temperature <= 0:
51
+ token = int(logits.argmax())
52
+ else:
53
+ token = int(torch.multinomial((logits / args.temperature).softmax(-1), 1))
54
+ ids.append(token)
55
+ if token == config["eos_token_id"]:
56
+ break
57
+
58
+ print(tokenizer.decode(ids))
59
+
60
+
61
+ if __name__ == "__main__":
62
+ main()
model.py ADDED
@@ -0,0 +1,109 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Tiny CED model architecture required to load model.safetensors."""
2
+
3
+ import torch
4
+ import torch.nn.functional as F
5
+ from torch import nn
6
+
7
+
8
+ class RMSNorm(nn.Module):
9
+ def __init__(self, dim):
10
+ super().__init__()
11
+ self.weight = nn.Parameter(torch.ones(dim))
12
+
13
+ def forward(self, x):
14
+ z = x.float()
15
+ return (
16
+ z
17
+ * torch.rsqrt(z.square().mean(-1, keepdim=True) + 1e-5)
18
+ * self.weight
19
+ ).to(x.dtype)
20
+
21
+
22
+ def rope(x, pos):
23
+ d = x.shape[-1]
24
+ freq = 10000 ** (torch.arange(0, d, 2, device=x.device) / d)
25
+ angle = pos[:, None, :, None].float() * freq
26
+ a, b = x.float()[..., ::2], x.float()[..., 1::2]
27
+ c, s = angle.cos(), angle.sin()
28
+ return torch.stack((a * c - b * s, a * s + b * c), dim=-1).flatten(-2).to(x.dtype)
29
+
30
+
31
+ class Attention(nn.Module):
32
+ def __init__(self, dim, heads, window):
33
+ super().__init__()
34
+ self.heads = heads
35
+ self.window = window
36
+ for name in ["wq", "wkg", "wvg", "wkl", "wvl", "wo"]:
37
+ setattr(self, name, nn.Linear(dim, dim, bias=False))
38
+
39
+ def forward(self, u, source, pos, valid):
40
+ batch, length, dim = u.shape
41
+
42
+ def split(z):
43
+ return z.view(batch, length, self.heads, dim // self.heads).transpose(1, 2)
44
+
45
+ q = rope(split(self.wq(u)), pos)
46
+ kg, vg = rope(split(self.wkg(source)), pos), split(self.wvg(source))
47
+ kl, vl = rope(split(self.wkl(u)), pos), split(self.wvl(u))
48
+ distance = pos[:, :, None] - pos[:, None, :]
49
+ global_mask = (distance >= 0) & valid[:, :, None] & valid[:, None, :]
50
+ local_mask = global_mask & (distance < self.window)
51
+ mask = torch.cat((global_mask, local_mask), -1)[:, None]
52
+ k, v = torch.cat((kg, kl), 2), torch.cat((vg, vl), 2)
53
+ out = F.scaled_dot_product_attention(
54
+ q,
55
+ k,
56
+ v,
57
+ attn_mask=mask,
58
+ dropout_p=0,
59
+ is_causal=False,
60
+ )
61
+ return self.wo(out.transpose(1, 2).contiguous().view(batch, length, dim))
62
+
63
+
64
+ class Block(nn.Module):
65
+ def __init__(self, dim, heads, ff, window):
66
+ super().__init__()
67
+ self.attn_norm, self.ffn_norm = RMSNorm(dim), RMSNorm(dim)
68
+ self.attn = Attention(dim, heads, window)
69
+ self.gate, self.up = (nn.Linear(dim, ff, bias=False) for _ in range(2))
70
+ self.down = nn.Linear(ff, dim, bias=False)
71
+
72
+ def forward(self, h, e, pos, valid):
73
+ u = self.attn_norm(h)
74
+ h = h + self.attn(u, u if e is None else e, pos, valid)
75
+ z = self.ffn_norm(h)
76
+ return h + self.down(F.silu(self.gate(z)) * self.up(z))
77
+
78
+
79
+ class CED(nn.Module):
80
+ def __init__(self, vocab=8192, dim=384, heads=6, ff=1024, layers=4, window=64):
81
+ super().__init__()
82
+ assert dim % heads == 0 and (dim // heads) % 2 == 0
83
+ self.embedding = nn.Embedding(vocab, dim)
84
+ self.encoder = nn.ModuleList(
85
+ [Block(dim, heads, ff, window) for _ in range(layers)]
86
+ )
87
+ self.decoder = nn.ModuleList(
88
+ [Block(dim, heads, ff, window) for _ in range(layers)]
89
+ )
90
+ self.encoder_norm, self.final_norm = RMSNorm(dim), RMSNorm(dim)
91
+ self.head = nn.Linear(dim, vocab, bias=False)
92
+ self.head.weight = self.embedding.weight
93
+ for parameter in self.parameters():
94
+ if parameter.ndim > 1:
95
+ nn.init.normal_(parameter, std=0.02)
96
+
97
+ def forward(self, ids, positions=None, valid=None):
98
+ if positions is None:
99
+ positions = torch.arange(ids.shape[1], device=ids.device)[None].expand_as(ids)
100
+ if valid is None:
101
+ valid = torch.ones_like(ids, dtype=torch.bool)
102
+ h = self.embedding(ids)
103
+ for layer in self.encoder:
104
+ h = layer(h, None, positions, valid)
105
+ e = self.encoder_norm(h)
106
+ h = e
107
+ for layer in self.decoder:
108
+ h = layer(h, e, positions, valid)
109
+ return self.head(self.final_norm(h))
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4a60e58abb8fa784857588179cba839ccc76d4d56ef297887c376f3d8a5aacf5
3
+ size 78679560
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ torch>=2.0
2
+ safetensors>=0.4
3
+ tokenizers>=0.20
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff