Download src/layer0.cpp from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 12.1 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/src/layer0.cpp
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/src/layer0.cpp
-
curl -L -o layer0.cpp https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/src/layer0.cpp
12.1 kB
| namespace ling3 { | |
| namespace { | |
| using Clock = std::chrono::steady_clock; | |
| constexpr int kHidden = 1536; | |
| constexpr int kHeads = 16; | |
| constexpr int kHeadDimension = 128; | |
| constexpr int kAttention = kHeads * kHeadDimension; | |
| constexpr int kDense = 4608; | |
| constexpr float kEpsilon = 1.0e-6F; | |
| double Milliseconds(Clock::time_point begin, Clock::time_point end) { | |
| return std::chrono::duration<double, std::milli>(end - begin).count(); | |
| } | |
| template <typename T> | |
| std::span<const T> Typed(const TensorView & tensor, DataType type) { | |
| if (tensor.entry->dtype != static_cast<std::uint32_t>(type) || | |
| tensor.entry->data_bytes % sizeof(T) != 0) { | |
| throw std::runtime_error(std::string(tensor.name) + " has an incompatible dtype"); | |
| } | |
| return { | |
| reinterpret_cast<const T *>(tensor.data), | |
| static_cast<std::size_t>(tensor.entry->data_bytes / sizeof(T)), | |
| }; | |
| } | |
| std::span<const std::byte> Blob(const TensorView & tensor, TensorRole role) { | |
| if (tensor.entry->role != static_cast<std::uint32_t>(role)) { | |
| throw std::runtime_error(std::string(tensor.name) + " has an incompatible role"); | |
| } | |
| return {tensor.data, static_cast<std::size_t>(tensor.entry->data_bytes)}; | |
| } | |
| std::vector<float> DecodeBf16(const TensorView & tensor, std::size_t expected) { | |
| const auto input = Typed<std::uint16_t>(tensor, DataType::kBFloat16); | |
| if (input.size() != expected) { | |
| throw std::runtime_error(std::string(tensor.name) + " has an incompatible shape"); | |
| } | |
| std::vector<float> output(expected); | |
| for (std::size_t index = 0; index < expected; ++index) { | |
| output[index] = BFloat16ToFloat(input[index]); | |
| } | |
| return output; | |
| } | |
| std::unique_ptr<DynamicW4Linear> MakeLinear( | |
| const ModelPackage & package, | |
| const std::string & base) { | |
| const auto & weight = package.tensor(base + ".weight"); | |
| const auto & scales = package.tensor(base + ".scales"); | |
| const auto & correction = package.tensor(base + ".correction"); | |
| if (weight.entry->rank != 2 || | |
| weight.entry->dtype != static_cast<std::uint32_t>(DataType::kInt4Low) || | |
| weight.entry->layout != static_cast<std::uint32_t>(TensorLayout::kPackedInt4Low)) { | |
| throw std::runtime_error(base + " is not a packed W4 linear"); | |
| } | |
| const int k = static_cast<int>(weight.entry->dims[0]); | |
| const int n = static_cast<int>(weight.entry->dims[1]); | |
| return std::make_unique<DynamicW4Linear>( | |
| W4LinearConfig {k, n, static_cast<int>(weight.entry->flags), {0, 1, 2}}, | |
| Typed<std::byte>(weight, DataType::kInt4Low), | |
| Typed<float>(scales, DataType::kFloat32), | |
| Typed<std::int32_t>(correction, DataType::kInt32)); | |
| } | |
| void NormalizeHeads(std::span<float> values) { | |
| for (int head = 0; head < kHeads; ++head) { | |
| float sum = kEpsilon; | |
| const int begin = head * kHeadDimension; | |
| for (int index = 0; index < kHeadDimension; ++index) { | |
| const float value = values[begin + index]; | |
| sum += value * value; | |
| } | |
| const float inverse = 1.0F / std::sqrt(sum); | |
| for (int index = 0; index < kHeadDimension; ++index) { | |
| values[begin + index] *= inverse; | |
| } | |
| } | |
| } | |
| void CausalConvSilu( | |
| std::span<const float> input, | |
| std::span<const float> weight, | |
| std::span<float> state, | |
| std::span<float> output) { | |
| if (input.size() != kAttention || output.size() != kAttention || | |
| weight.size() != static_cast<std::size_t>(kAttention * 4) || | |
| state.size() != static_cast<std::size_t>(kAttention * 3)) { | |
| throw std::invalid_argument("causal convolution tensor sizes are incompatible"); | |
| } | |
| for (int channel = 0; channel < kAttention; ++channel) { | |
| auto * history = state.data() + static_cast<std::size_t>(channel) * 3; | |
| const auto * kernel = weight.data() + static_cast<std::size_t>(channel) * 4; | |
| const float value = history[0] * kernel[0] + history[1] * kernel[1] + | |
| history[2] * kernel[2] + input[channel] * kernel[3]; | |
| history[0] = history[1]; | |
| history[1] = history[2]; | |
| history[2] = input[channel]; | |
| output[channel] = value / (1.0F + std::exp(-value)); | |
| } | |
| } | |
| } // namespace | |
| struct Layer0::Impl { | |
| const ModelPackage & package; | |
| std::span<const std::uint16_t> embeddings; | |
| std::vector<float> input_norm_weight; | |
| std::vector<float> post_norm_weight; | |
| std::vector<float> q_conv_weight; | |
| std::vector<float> k_conv_weight; | |
| std::vector<float> v_conv_weight; | |
| std::vector<float> output_norm_weight; | |
| std::span<const float> a_log; | |
| std::span<const float> dt_bias; | |
| std::unique_ptr<DynamicW4Linear> qkvfgb; | |
| std::unique_ptr<DynamicW4Linear> attention_output; | |
| std::unique_ptr<DynamicW4Linear> gate_up; | |
| std::unique_ptr<DynamicW4Linear> down; | |
| GdnStep gdn; | |
| std::array<std::vector<float>, 3> convolution_state; | |
| std::vector<float> hidden; | |
| std::vector<float> normalized; | |
| std::vector<float> projected; | |
| std::vector<float> q; | |
| std::vector<float> k; | |
| std::vector<float> v; | |
| std::vector<float> recurrence_output; | |
| std::vector<float> attention_vector; | |
| std::vector<float> attention_result; | |
| std::vector<float> ffn_projected; | |
| std::vector<float> ffn_hidden; | |
| std::vector<float> ffn_result; | |
| explicit Impl(const ModelPackage & model) | |
| : package(model), | |
| embeddings(Typed<std::uint16_t>( | |
| package.tensor("model.word_embeddings.weight"), DataType::kBFloat16)), | |
| input_norm_weight(DecodeBf16( | |
| package.tensor("model.layers.0.input_layernorm.weight"), kHidden)), | |
| post_norm_weight(DecodeBf16( | |
| package.tensor("model.layers.0.post_attention_layernorm.weight"), kHidden)), | |
| q_conv_weight(DecodeBf16( | |
| package.tensor("model.layers.0.attention.q_conv1d.weight"), kAttention * 4)), | |
| k_conv_weight(DecodeBf16( | |
| package.tensor("model.layers.0.attention.k_conv1d.weight"), kAttention * 4)), | |
| v_conv_weight(DecodeBf16( | |
| package.tensor("model.layers.0.attention.v_conv1d.weight"), kAttention * 4)), | |
| output_norm_weight(DecodeBf16( | |
| package.tensor("model.layers.0.attention.o_norm.weight"), kHeadDimension)), | |
| a_log(Typed<float>(package.tensor("model.layers.0.attention.A_log"), DataType::kFloat32)), | |
| dt_bias(Typed<float>(package.tensor("model.layers.0.attention.dt_bias"), DataType::kFloat32)), | |
| qkvfgb(MakeLinear(package, "model.layers.0.attention.qkvfgb")), | |
| attention_output(MakeLinear(package, "model.layers.0.attention.o_proj")), | |
| gate_up(MakeLinear(package, "model.layers.0.mlp.gate_up")), | |
| down(MakeLinear(package, "model.layers.0.mlp.down_proj")), | |
| gdn( | |
| Blob(package.tensor("rknn.gdn.heads6"), TensorRole::kRknnIsland), | |
| Blob(package.tensor("rknn.gdn.heads5"), TensorRole::kRknnIsland)), | |
| hidden(kHidden), | |
| normalized(kHidden), | |
| projected(10304), | |
| q(kAttention), | |
| k(kAttention), | |
| v(kAttention), | |
| recurrence_output(kAttention), | |
| attention_vector(kAttention), | |
| attention_result(kHidden), | |
| ffn_projected(2 * kDense), | |
| ffn_hidden(kDense), | |
| ffn_result(kHidden) { | |
| if (embeddings.size() != static_cast<std::size_t>(157184 * kHidden) || | |
| a_log.size() != kHeads || dt_bias.size() != kAttention) { | |
| throw std::runtime_error("layer 0 raw tensor shapes are incompatible"); | |
| } | |
| for (auto & state : convolution_state) state.assign(kAttention * 3, 0.0F); | |
| } | |
| void Reset() { | |
| for (auto & state : convolution_state) std::fill(state.begin(), state.end(), 0.0F); | |
| gdn.Reset(); | |
| } | |
| Layer0Timings DecodeToken(std::uint32_t token, std::span<float> output) { | |
| if (token >= 157184 || output.size() != kHidden) { | |
| throw std::invalid_argument("layer 0 token or output is out of range"); | |
| } | |
| const auto begin = Clock::now(); | |
| const auto embedding = embeddings.subspan(static_cast<std::size_t>(token) * kHidden, kHidden); | |
| for (int index = 0; index < kHidden; ++index) hidden[index] = BFloat16ToFloat(embedding[index]); | |
| RmsNorm(hidden.data(), input_norm_weight.data(), normalized.data(), kHidden, kEpsilon); | |
| qkvfgb->Run(normalized, projected); | |
| CausalConvSilu( | |
| std::span<const float>(projected).subspan(0, kAttention), | |
| q_conv_weight, | |
| convolution_state[0], | |
| q); | |
| CausalConvSilu( | |
| std::span<const float>(projected).subspan(kAttention, kAttention), | |
| k_conv_weight, | |
| convolution_state[1], | |
| k); | |
| CausalConvSilu( | |
| std::span<const float>(projected).subspan(2 * kAttention, kAttention), | |
| v_conv_weight, | |
| convolution_state[2], | |
| v); | |
| NormalizeHeads(q); | |
| NormalizeHeads(k); | |
| const auto f = std::span<const float>(projected).subspan(3 * kAttention, kAttention); | |
| const auto gate = std::span<const float>(projected).subspan(4 * kAttention, kAttention); | |
| const auto beta_logits = std::span<const float>(projected).subspan(5 * kAttention, kHeads); | |
| std::vector<float> decay(kAttention); | |
| std::vector<float> beta(kHeads); | |
| for (int head = 0; head < kHeads; ++head) { | |
| const float a = std::exp(a_log[head]); | |
| beta[head] = 1.0F / (1.0F + std::exp(-beta_logits[head])); | |
| for (int index = 0; index < kHeadDimension; ++index) { | |
| const int offset = head * kHeadDimension + index; | |
| const float argument = a * (f[offset] + dt_bias[offset]); | |
| decay[offset] = -5.0F / (1.0F + std::exp(-argument)); | |
| } | |
| } | |
| gdn.Run(q, k, v, decay, beta, recurrence_output); | |
| for (int head = 0; head < kHeads; ++head) { | |
| const int base = head * kHeadDimension; | |
| float sum = 0.0F; | |
| for (int index = 0; index < kHeadDimension; ++index) { | |
| const float value = recurrence_output[base + index]; | |
| sum += value * value; | |
| } | |
| const float inverse = 1.0F / | |
| std::sqrt(sum / static_cast<float>(kHeadDimension) + kEpsilon); | |
| for (int index = 0; index < kHeadDimension; ++index) { | |
| const int offset = base + index; | |
| const float sigmoid_gate = 1.0F / (1.0F + std::exp(-gate[offset])); | |
| attention_vector[offset] = recurrence_output[offset] * inverse * | |
| output_norm_weight[index] * sigmoid_gate; | |
| } | |
| } | |
| attention_output->Run(attention_vector, attention_result); | |
| for (int index = 0; index < kHidden; ++index) hidden[index] += attention_result[index]; | |
| const auto attention_end = Clock::now(); | |
| RmsNorm(hidden.data(), post_norm_weight.data(), normalized.data(), kHidden, kEpsilon); | |
| gate_up->Run(normalized, ffn_projected); | |
| SiluMultiply(ffn_projected.data(), ffn_projected.data() + kDense, ffn_hidden.data(), kDense); | |
| down->Run(ffn_hidden, ffn_result); | |
| for (int index = 0; index < kHidden; ++index) output[index] = hidden[index] + ffn_result[index]; | |
| const auto end = Clock::now(); | |
| return { | |
| Milliseconds(begin, attention_end), | |
| Milliseconds(attention_end, end), | |
| Milliseconds(begin, end), | |
| }; | |
| } | |
| }; | |
| Layer0::Layer0(const ModelPackage & package) : impl_(std::make_unique<Impl>(package)) {} | |
| Layer0::~Layer0() = default; | |
| void Layer0::Reset() { impl_->Reset(); } | |
| Layer0Timings Layer0::DecodeToken(std::uint32_t token, std::span<float> output) { | |
| return impl_->DecodeToken(token, output); | |
| } | |
| } // namespace ling3 | |