#include "ling3/layer0.h" #include "ling3/cpu_kernels.h" #include #include #include #include #include #include #include #include namespace ling3 { namespace { using Clock = std::chrono::steady_clock; constexpr int kHidden = 1536; constexpr int kHeads = 16; constexpr int kHeadDimension = 128; constexpr int kAttention = kHeads * kHeadDimension; constexpr int kDense = 4608; constexpr float kEpsilon = 1.0e-6F; double Milliseconds(Clock::time_point begin, Clock::time_point end) { return std::chrono::duration(end - begin).count(); } template std::span Typed(const TensorView & tensor, DataType type) { if (tensor.entry->dtype != static_cast(type) || tensor.entry->data_bytes % sizeof(T) != 0) { throw std::runtime_error(std::string(tensor.name) + " has an incompatible dtype"); } return { reinterpret_cast(tensor.data), static_cast(tensor.entry->data_bytes / sizeof(T)), }; } std::span Blob(const TensorView & tensor, TensorRole role) { if (tensor.entry->role != static_cast(role)) { throw std::runtime_error(std::string(tensor.name) + " has an incompatible role"); } return {tensor.data, static_cast(tensor.entry->data_bytes)}; } std::vector DecodeBf16(const TensorView & tensor, std::size_t expected) { const auto input = Typed(tensor, DataType::kBFloat16); if (input.size() != expected) { throw std::runtime_error(std::string(tensor.name) + " has an incompatible shape"); } std::vector output(expected); for (std::size_t index = 0; index < expected; ++index) { output[index] = BFloat16ToFloat(input[index]); } return output; } std::unique_ptr MakeLinear( const ModelPackage & package, const std::string & base) { const auto & weight = package.tensor(base + ".weight"); const auto & scales = package.tensor(base + ".scales"); const auto & correction = package.tensor(base + ".correction"); if (weight.entry->rank != 2 || weight.entry->dtype != static_cast(DataType::kInt4Low) || weight.entry->layout != static_cast(TensorLayout::kPackedInt4Low)) { throw std::runtime_error(base + " is not a packed W4 linear"); } const int k = static_cast(weight.entry->dims[0]); const int n = static_cast(weight.entry->dims[1]); return std::make_unique( W4LinearConfig {k, n, static_cast(weight.entry->flags), {0, 1, 2}}, Typed(weight, DataType::kInt4Low), Typed(scales, DataType::kFloat32), Typed(correction, DataType::kInt32)); } void NormalizeHeads(std::span values) { for (int head = 0; head < kHeads; ++head) { float sum = kEpsilon; const int begin = head * kHeadDimension; for (int index = 0; index < kHeadDimension; ++index) { const float value = values[begin + index]; sum += value * value; } const float inverse = 1.0F / std::sqrt(sum); for (int index = 0; index < kHeadDimension; ++index) { values[begin + index] *= inverse; } } } void CausalConvSilu( std::span input, std::span weight, std::span state, std::span output) { if (input.size() != kAttention || output.size() != kAttention || weight.size() != static_cast(kAttention * 4) || state.size() != static_cast(kAttention * 3)) { throw std::invalid_argument("causal convolution tensor sizes are incompatible"); } for (int channel = 0; channel < kAttention; ++channel) { auto * history = state.data() + static_cast(channel) * 3; const auto * kernel = weight.data() + static_cast(channel) * 4; const float value = history[0] * kernel[0] + history[1] * kernel[1] + history[2] * kernel[2] + input[channel] * kernel[3]; history[0] = history[1]; history[1] = history[2]; history[2] = input[channel]; output[channel] = value / (1.0F + std::exp(-value)); } } } // namespace struct Layer0::Impl { const ModelPackage & package; std::span embeddings; std::vector input_norm_weight; std::vector post_norm_weight; std::vector q_conv_weight; std::vector k_conv_weight; std::vector v_conv_weight; std::vector output_norm_weight; std::span a_log; std::span dt_bias; std::unique_ptr qkvfgb; std::unique_ptr attention_output; std::unique_ptr gate_up; std::unique_ptr down; GdnStep gdn; std::array, 3> convolution_state; std::vector hidden; std::vector normalized; std::vector projected; std::vector q; std::vector k; std::vector v; std::vector recurrence_output; std::vector attention_vector; std::vector attention_result; std::vector ffn_projected; std::vector ffn_hidden; std::vector ffn_result; explicit Impl(const ModelPackage & model) : package(model), embeddings(Typed( package.tensor("model.word_embeddings.weight"), DataType::kBFloat16)), input_norm_weight(DecodeBf16( package.tensor("model.layers.0.input_layernorm.weight"), kHidden)), post_norm_weight(DecodeBf16( package.tensor("model.layers.0.post_attention_layernorm.weight"), kHidden)), q_conv_weight(DecodeBf16( package.tensor("model.layers.0.attention.q_conv1d.weight"), kAttention * 4)), k_conv_weight(DecodeBf16( package.tensor("model.layers.0.attention.k_conv1d.weight"), kAttention * 4)), v_conv_weight(DecodeBf16( package.tensor("model.layers.0.attention.v_conv1d.weight"), kAttention * 4)), output_norm_weight(DecodeBf16( package.tensor("model.layers.0.attention.o_norm.weight"), kHeadDimension)), a_log(Typed(package.tensor("model.layers.0.attention.A_log"), DataType::kFloat32)), dt_bias(Typed(package.tensor("model.layers.0.attention.dt_bias"), DataType::kFloat32)), qkvfgb(MakeLinear(package, "model.layers.0.attention.qkvfgb")), attention_output(MakeLinear(package, "model.layers.0.attention.o_proj")), gate_up(MakeLinear(package, "model.layers.0.mlp.gate_up")), down(MakeLinear(package, "model.layers.0.mlp.down_proj")), gdn( Blob(package.tensor("rknn.gdn.heads6"), TensorRole::kRknnIsland), Blob(package.tensor("rknn.gdn.heads5"), TensorRole::kRknnIsland)), hidden(kHidden), normalized(kHidden), projected(10304), q(kAttention), k(kAttention), v(kAttention), recurrence_output(kAttention), attention_vector(kAttention), attention_result(kHidden), ffn_projected(2 * kDense), ffn_hidden(kDense), ffn_result(kHidden) { if (embeddings.size() != static_cast(157184 * kHidden) || a_log.size() != kHeads || dt_bias.size() != kAttention) { throw std::runtime_error("layer 0 raw tensor shapes are incompatible"); } for (auto & state : convolution_state) state.assign(kAttention * 3, 0.0F); } void Reset() { for (auto & state : convolution_state) std::fill(state.begin(), state.end(), 0.0F); gdn.Reset(); } Layer0Timings DecodeToken(std::uint32_t token, std::span output) { if (token >= 157184 || output.size() != kHidden) { throw std::invalid_argument("layer 0 token or output is out of range"); } const auto begin = Clock::now(); const auto embedding = embeddings.subspan(static_cast(token) * kHidden, kHidden); for (int index = 0; index < kHidden; ++index) hidden[index] = BFloat16ToFloat(embedding[index]); RmsNorm(hidden.data(), input_norm_weight.data(), normalized.data(), kHidden, kEpsilon); qkvfgb->Run(normalized, projected); CausalConvSilu( std::span(projected).subspan(0, kAttention), q_conv_weight, convolution_state[0], q); CausalConvSilu( std::span(projected).subspan(kAttention, kAttention), k_conv_weight, convolution_state[1], k); CausalConvSilu( std::span(projected).subspan(2 * kAttention, kAttention), v_conv_weight, convolution_state[2], v); NormalizeHeads(q); NormalizeHeads(k); const auto f = std::span(projected).subspan(3 * kAttention, kAttention); const auto gate = std::span(projected).subspan(4 * kAttention, kAttention); const auto beta_logits = std::span(projected).subspan(5 * kAttention, kHeads); std::vector decay(kAttention); std::vector beta(kHeads); for (int head = 0; head < kHeads; ++head) { const float a = std::exp(a_log[head]); beta[head] = 1.0F / (1.0F + std::exp(-beta_logits[head])); for (int index = 0; index < kHeadDimension; ++index) { const int offset = head * kHeadDimension + index; const float argument = a * (f[offset] + dt_bias[offset]); decay[offset] = -5.0F / (1.0F + std::exp(-argument)); } } gdn.Run(q, k, v, decay, beta, recurrence_output); for (int head = 0; head < kHeads; ++head) { const int base = head * kHeadDimension; float sum = 0.0F; for (int index = 0; index < kHeadDimension; ++index) { const float value = recurrence_output[base + index]; sum += value * value; } const float inverse = 1.0F / std::sqrt(sum / static_cast(kHeadDimension) + kEpsilon); for (int index = 0; index < kHeadDimension; ++index) { const int offset = base + index; const float sigmoid_gate = 1.0F / (1.0F + std::exp(-gate[offset])); attention_vector[offset] = recurrence_output[offset] * inverse * output_norm_weight[index] * sigmoid_gate; } } attention_output->Run(attention_vector, attention_result); for (int index = 0; index < kHidden; ++index) hidden[index] += attention_result[index]; const auto attention_end = Clock::now(); RmsNorm(hidden.data(), post_norm_weight.data(), normalized.data(), kHidden, kEpsilon); gate_up->Run(normalized, ffn_projected); SiluMultiply(ffn_projected.data(), ffn_projected.data() + kDense, ffn_hidden.data(), kDense); down->Run(ffn_hidden, ffn_result); for (int index = 0; index < kHidden; ++index) output[index] = hidden[index] + ffn_result[index]; const auto end = Clock::now(); return { Milliseconds(begin, attention_end), Milliseconds(attention_end, end), Milliseconds(begin, end), }; } }; Layer0::Layer0(const ModelPackage & package) : impl_(std::make_unique(package)) {} Layer0::~Layer0() = default; void Layer0::Reset() { impl_->Reset(); } Layer0Timings Layer0::DecodeToken(std::uint32_t token, std::span output) { return impl_->DecodeToken(token, output); } } // namespace ling3