Download code/models/common/reference_rope.py from tt-hous/clef: direct link, hf CLI and curl.
- Browser
- Download file 3.23 kB
-
https://huggingface.co/tt-hous/clef/resolve/main/code/models/common/reference_rope.py
- Command line
-
hf download hf://tt-hous/clef/code/models/common/reference_rope.py
-
curl -L -o reference_rope.py https://huggingface.co/tt-hous/clef/resolve/main/code/models/common/reference_rope.py
3.23 kB
| # SPDX-FileCopyrightText: Copyright (c) Meta Platforms, Inc. and affiliates. | |
| # SPDX-License-Identifier: LicenseRef-LICENSE-FILE | |
| # All rights reserved. | |
| # | |
| # This source code is licensed under the terms described in LICENSE-Llama-3.1 | |
| # in this folder. | |
| # Copyright (c) Meta Platforms, Inc. and affiliates. | |
| # This software may be used and distributed in accordance with the terms of the Llama 3 Community License Agreement. | |
| """Frozen Llama-3.1 reference goldens for tests. | |
| Copied verbatim from the removed legacy Llama-3.1 reference model so RoPE and | |
| RMSNorm numerics stay bit-identical without depending on that submodule. | |
| """ | |
| import math | |
| from typing import Tuple | |
| import torch | |
| class ReferenceRMSNorm(torch.nn.Module): | |
| def __init__(self, dim: int, eps: float = 1e-6): | |
| super().__init__() | |
| self.eps = eps | |
| self.weight = torch.nn.Parameter(torch.ones(dim)) | |
| def _norm(self, x): | |
| return x * torch.rsqrt(x.pow(2).mean(-1, keepdim=True) + self.eps) | |
| def forward(self, x): | |
| output = self._norm(x.float()).type_as(x) | |
| return output * self.weight | |
| def apply_scaling(freqs: torch.Tensor, scale_factor: float = 8): | |
| low_freq_factor = 1 | |
| high_freq_factor = 4 | |
| old_context_len = 8192 # original llama3 length | |
| low_freq_wavelen = old_context_len / low_freq_factor | |
| high_freq_wavelen = old_context_len / high_freq_factor | |
| new_freqs = [] | |
| for freq in freqs: | |
| wavelen = 2 * math.pi / freq | |
| if wavelen < high_freq_wavelen: | |
| new_freqs.append(freq) | |
| elif wavelen > low_freq_wavelen: | |
| new_freqs.append(freq / scale_factor) | |
| else: | |
| assert low_freq_wavelen != high_freq_wavelen | |
| smooth = (old_context_len / wavelen - low_freq_factor) / (high_freq_factor - low_freq_factor) | |
| new_freqs.append((1 - smooth) * freq / scale_factor + smooth * freq) | |
| return torch.tensor(new_freqs, dtype=freqs.dtype, device=freqs.device) | |
| def precompute_freqs_cis(dim: int, end: int, theta: float = 10000.0, use_scaled: bool = False, scale_factor: float = 8): | |
| freqs = 1.0 / (theta ** (torch.arange(0, dim, 2)[: (dim // 2)].float() / dim)) | |
| t = torch.arange(end, device=freqs.device, dtype=torch.float32) | |
| if use_scaled: | |
| freqs = apply_scaling(freqs, scale_factor) | |
| freqs = torch.outer(t, freqs) | |
| freqs_cis = torch.polar(torch.ones_like(freqs), freqs) # complex64 | |
| return freqs_cis | |
| def reshape_for_broadcast(freqs_cis: torch.Tensor, x: torch.Tensor): | |
| ndim = x.ndim | |
| assert 0 <= 1 < ndim | |
| assert freqs_cis.shape == (x.shape[1], x.shape[-1]) | |
| shape = [d if i == 1 or i == ndim - 1 else 1 for i, d in enumerate(x.shape)] | |
| return freqs_cis.view(*shape) | |
| def apply_rotary_emb( | |
| xq: torch.Tensor, | |
| xk: torch.Tensor, | |
| freqs_cis: torch.Tensor, | |
| ) -> Tuple[torch.Tensor, torch.Tensor]: | |
| xq_ = torch.view_as_complex(xq.float().reshape(*xq.shape[:-1], -1, 2)) | |
| xk_ = torch.view_as_complex(xk.float().reshape(*xk.shape[:-1], -1, 2)) | |
| freqs_cis = reshape_for_broadcast(freqs_cis, xq_) | |
| xq_out = torch.view_as_real(xq_ * freqs_cis).flatten(3) | |
| xk_out = torch.view_as_real(xk_ * freqs_cis).flatten(3) | |
| return xq_out.type_as(xq), xk_out.type_as(xk) | |