File size: 7,793 Bytes
d78b3c4
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
"""
model.py -- standalone architecture definition for GTM-3-base.

This is a plain PyTorch nanoGPT-style GPT model with RoPE (rotary position
embeddings), NOT a HuggingFace `transformers` AutoModel. To load the
released weights:

    pip install torch safetensors tiktoken

    import json, torch
    from safetensors.torch import load_file
    from model import GPT, GPTConfig

    with open("config.json") as f:
        config = GPTConfig(**json.load(f))
    model = GPT(config)
    state_dict = load_file("model.safetensors")
    model.load_state_dict(state_dict)
    model.eval()

    import tiktoken
    enc = tiktoken.get_encoding("gpt2")
    ids = enc.encode_ordinary("Once upon a time,")
    x = torch.tensor([ids], dtype=torch.long)
    out = model.generate(x, max_new_tokens=100, temperature=0.8, top_k=50,
                          eot_token=enc.eot_token, repetition_penalty=1.3)
    print(enc.decode(out[0].tolist()))
"""

import math
from dataclasses import dataclass

import torch
import torch.nn as nn
import torch.nn.functional as F


@dataclass
class GPTConfig:
    vocab_size: int = 50257
    block_size: int = 1024
    n_layer: int = 10
    n_head: int = 8
    n_embd: int = 608
    dropout: float = 0.0
    bias: bool = True
    rope_theta: float = 10000.0  # standard RoPE base frequency


def precompute_rope_freqs(head_dim, max_seq_len, theta=10000.0, device="cpu"):
    """Precompute the complex rotation frequencies used by RoPE, one pair
    per (position, frequency-band). Standard formula: theta_i = theta^(-2i/dim)."""
    assert head_dim % 2 == 0, "RoPE requires an even head_dim"
    freqs = 1.0 / (theta ** (torch.arange(0, head_dim, 2, device=device).float() / head_dim))
    t = torch.arange(max_seq_len, device=device).float()
    freqs = torch.outer(t, freqs)  # (max_seq_len, head_dim/2)
    return torch.polar(torch.ones_like(freqs), freqs)  # complex64, (max_seq_len, head_dim/2)


def apply_rope(x, freqs_cis):
    """Apply rotary position embeddings to a (B, n_head, T, head_dim) tensor."""
    B, n_head, T, head_dim = x.shape
    x_complex = torch.view_as_complex(x.float().reshape(B, n_head, T, head_dim // 2, 2))
    freqs_cis = freqs_cis[:T].view(1, 1, T, head_dim // 2)
    x_rotated = x_complex * freqs_cis
    x_out = torch.view_as_real(x_rotated).reshape(B, n_head, T, head_dim)
    return x_out.type_as(x)


class CausalSelfAttention(nn.Module):
    def __init__(self, config):
        super().__init__()
        assert config.n_embd % config.n_head == 0
        self.n_head = config.n_head
        self.n_embd = config.n_embd
        self.head_dim = config.n_embd // config.n_head
        self.c_attn = nn.Linear(config.n_embd, 3 * config.n_embd, bias=config.bias)
        self.c_proj = nn.Linear(config.n_embd, config.n_embd, bias=config.bias)
        self.attn_dropout = nn.Dropout(config.dropout)
        self.resid_dropout = nn.Dropout(config.dropout)
        self.dropout = config.dropout

    def forward(self, x, freqs_cis):
        B, T, C = x.shape
        q, k, v = self.c_attn(x).split(self.n_embd, dim=2)
        q = q.view(B, T, self.n_head, C // self.n_head).transpose(1, 2)
        k = k.view(B, T, self.n_head, C // self.n_head).transpose(1, 2)
        v = v.view(B, T, self.n_head, C // self.n_head).transpose(1, 2)
        # RoPE is applied to queries and keys only, not values -- this is what
        # makes attention scores depend on *relative* position between tokens
        q = apply_rope(q, freqs_cis)
        k = apply_rope(k, freqs_cis)
        y = F.scaled_dot_product_attention(
            q, k, v, is_causal=True,
            dropout_p=self.dropout if self.training else 0.0,
        )
        y = y.transpose(1, 2).contiguous().view(B, T, C)
        return self.resid_dropout(self.c_proj(y))


class MLP(nn.Module):
    def __init__(self, config):
        super().__init__()
        self.c_fc = nn.Linear(config.n_embd, 4 * config.n_embd, bias=config.bias)
        self.gelu = nn.GELU()
        self.c_proj = nn.Linear(4 * config.n_embd, config.n_embd, bias=config.bias)
        self.dropout = nn.Dropout(config.dropout)

    def forward(self, x):
        return self.dropout(self.c_proj(self.gelu(self.c_fc(x))))


class Block(nn.Module):
    def __init__(self, config):
        super().__init__()
        self.ln_1 = nn.LayerNorm(config.n_embd)
        self.attn = CausalSelfAttention(config)
        self.ln_2 = nn.LayerNorm(config.n_embd)
        self.mlp = MLP(config)

    def forward(self, x, freqs_cis):
        x = x + self.attn(self.ln_1(x), freqs_cis)
        x = x + self.mlp(self.ln_2(x))
        return x


class GPT(nn.Module):
    def __init__(self, config):
        super().__init__()
        self.config = config
        self.transformer = nn.ModuleDict(dict(
            wte=nn.Embedding(config.vocab_size, config.n_embd),
            # NOTE: no wpe (learned position embedding) -- RoPE replaces it entirely,
            # applied inside attention rather than added to the input embeddings
            drop=nn.Dropout(config.dropout),
            h=nn.ModuleList([Block(config) for _ in range(config.n_layer)]),
            ln_f=nn.LayerNorm(config.n_embd),
        ))
        self.lm_head = nn.Linear(config.n_embd, config.vocab_size, bias=False)
        self.transformer.wte.weight = self.lm_head.weight
        head_dim = config.n_embd // config.n_head
        freqs_cis = precompute_rope_freqs(head_dim, config.block_size, theta=config.rope_theta)
        self.register_buffer("freqs_cis", freqs_cis, persistent=False)
        self.apply(self._init_weights)
        for pn, p in self.named_parameters():
            if pn.endswith("c_proj.weight"):
                nn.init.normal_(p, mean=0.0, std=0.02 / math.sqrt(2 * config.n_layer))

    def _init_weights(self, module):
        if isinstance(module, nn.Linear):
            nn.init.normal_(module.weight, mean=0.0, std=0.02)
            if module.bias is not None:
                nn.init.zeros_(module.bias)
        elif isinstance(module, nn.Embedding):
            nn.init.normal_(module.weight, mean=0.0, std=0.02)

    def forward(self, idx, targets=None):
        B, T = idx.shape
        assert T <= self.config.block_size, "sequence longer than block_size"
        x = self.transformer.drop(self.transformer.wte(idx))
        freqs_cis = self.freqs_cis.to(x.device)
        for block in self.transformer.h:
            x = block(x, freqs_cis)
        x = self.transformer.ln_f(x)
        logits = self.lm_head(x)
        loss = None
        if targets is not None:
            loss = F.cross_entropy(logits.view(-1, logits.size(-1)), targets.view(-1), ignore_index=-1)
        return logits, loss

    @torch.no_grad()
    def generate(self, idx, max_new_tokens, temperature=1.0, top_k=None, eot_token=None,
                 repetition_penalty=1.0):
        for _ in range(max_new_tokens):
            idx_cond = idx if idx.size(1) <= self.config.block_size else idx[:, -self.config.block_size:]
            logits, _ = self(idx_cond)
            logits = logits[:, -1, :] / temperature
            if repetition_penalty != 1.0:
                for seen_id in set(idx[0].tolist()):
                    if logits[0, seen_id] > 0:
                        logits[0, seen_id] /= repetition_penalty
                    else:
                        logits[0, seen_id] *= repetition_penalty
            if top_k is not None:
                v, _ = torch.topk(logits, min(top_k, logits.size(-1)))
                logits[logits < v[:, [-1]]] = float("-inf")
            probs = F.softmax(logits, dim=-1)
            idx_next = torch.multinomial(probs, num_samples=1)
            idx = torch.cat((idx, idx_next), dim=1)
            if eot_token is not None and idx_next.item() == eot_token:
                break
        return idx