Ling-3.0-tiny-RKNN / src /layer0.cpp
Sariel00's picture
Publish Ling-3.0-tiny RKNN engine and model
3fd1a35 verified
Raw History Blame Contribute Delete
12.1 kB
#include "ling3/layer0.h"
#include "ling3/cpu_kernels.h"
#include <algorithm>
#include <chrono>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <stdexcept>
#include <string>
#include <vector>
namespace ling3 {
namespace {
using Clock = std::chrono::steady_clock;
constexpr int kHidden = 1536;
constexpr int kHeads = 16;
constexpr int kHeadDimension = 128;
constexpr int kAttention = kHeads * kHeadDimension;
constexpr int kDense = 4608;
constexpr float kEpsilon = 1.0e-6F;
double Milliseconds(Clock::time_point begin, Clock::time_point end) {
return std::chrono::duration<double, std::milli>(end - begin).count();
}
template <typename T>
std::span<const T> Typed(const TensorView & tensor, DataType type) {
if (tensor.entry->dtype != static_cast<std::uint32_t>(type) ||
tensor.entry->data_bytes % sizeof(T) != 0) {
throw std::runtime_error(std::string(tensor.name) + " has an incompatible dtype");
}
return {
reinterpret_cast<const T *>(tensor.data),
static_cast<std::size_t>(tensor.entry->data_bytes / sizeof(T)),
};
}
std::span<const std::byte> Blob(const TensorView & tensor, TensorRole role) {
if (tensor.entry->role != static_cast<std::uint32_t>(role)) {
throw std::runtime_error(std::string(tensor.name) + " has an incompatible role");
}
return {tensor.data, static_cast<std::size_t>(tensor.entry->data_bytes)};
}
std::vector<float> DecodeBf16(const TensorView & tensor, std::size_t expected) {
const auto input = Typed<std::uint16_t>(tensor, DataType::kBFloat16);
if (input.size() != expected) {
throw std::runtime_error(std::string(tensor.name) + " has an incompatible shape");
}
std::vector<float> output(expected);
for (std::size_t index = 0; index < expected; ++index) {
output[index] = BFloat16ToFloat(input[index]);
}
return output;
}
std::unique_ptr<DynamicW4Linear> MakeLinear(
const ModelPackage & package,
const std::string & base) {
const auto & weight = package.tensor(base + ".weight");
const auto & scales = package.tensor(base + ".scales");
const auto & correction = package.tensor(base + ".correction");
if (weight.entry->rank != 2 ||
weight.entry->dtype != static_cast<std::uint32_t>(DataType::kInt4Low) ||
weight.entry->layout != static_cast<std::uint32_t>(TensorLayout::kPackedInt4Low)) {
throw std::runtime_error(base + " is not a packed W4 linear");
}
const int k = static_cast<int>(weight.entry->dims[0]);
const int n = static_cast<int>(weight.entry->dims[1]);
return std::make_unique<DynamicW4Linear>(
W4LinearConfig {k, n, static_cast<int>(weight.entry->flags), {0, 1, 2}},
Typed<std::byte>(weight, DataType::kInt4Low),
Typed<float>(scales, DataType::kFloat32),
Typed<std::int32_t>(correction, DataType::kInt32));
}
void NormalizeHeads(std::span<float> values) {
for (int head = 0; head < kHeads; ++head) {
float sum = kEpsilon;
const int begin = head * kHeadDimension;
for (int index = 0; index < kHeadDimension; ++index) {
const float value = values[begin + index];
sum += value * value;
}
const float inverse = 1.0F / std::sqrt(sum);
for (int index = 0; index < kHeadDimension; ++index) {
values[begin + index] *= inverse;
}
}
}
void CausalConvSilu(
std::span<const float> input,
std::span<const float> weight,
std::span<float> state,
std::span<float> output) {
if (input.size() != kAttention || output.size() != kAttention ||
weight.size() != static_cast<std::size_t>(kAttention * 4) ||
state.size() != static_cast<std::size_t>(kAttention * 3)) {
throw std::invalid_argument("causal convolution tensor sizes are incompatible");
}
for (int channel = 0; channel < kAttention; ++channel) {
auto * history = state.data() + static_cast<std::size_t>(channel) * 3;
const auto * kernel = weight.data() + static_cast<std::size_t>(channel) * 4;
const float value = history[0] * kernel[0] + history[1] * kernel[1] +
history[2] * kernel[2] + input[channel] * kernel[3];
history[0] = history[1];
history[1] = history[2];
history[2] = input[channel];
output[channel] = value / (1.0F + std::exp(-value));
}
}
} // namespace
struct Layer0::Impl {
const ModelPackage & package;
std::span<const std::uint16_t> embeddings;
std::vector<float> input_norm_weight;
std::vector<float> post_norm_weight;
std::vector<float> q_conv_weight;
std::vector<float> k_conv_weight;
std::vector<float> v_conv_weight;
std::vector<float> output_norm_weight;
std::span<const float> a_log;
std::span<const float> dt_bias;
std::unique_ptr<DynamicW4Linear> qkvfgb;
std::unique_ptr<DynamicW4Linear> attention_output;
std::unique_ptr<DynamicW4Linear> gate_up;
std::unique_ptr<DynamicW4Linear> down;
GdnStep gdn;
std::array<std::vector<float>, 3> convolution_state;
std::vector<float> hidden;
std::vector<float> normalized;
std::vector<float> projected;
std::vector<float> q;
std::vector<float> k;
std::vector<float> v;
std::vector<float> recurrence_output;
std::vector<float> attention_vector;
std::vector<float> attention_result;
std::vector<float> ffn_projected;
std::vector<float> ffn_hidden;
std::vector<float> ffn_result;
explicit Impl(const ModelPackage & model)
: package(model),
embeddings(Typed<std::uint16_t>(
package.tensor("model.word_embeddings.weight"), DataType::kBFloat16)),
input_norm_weight(DecodeBf16(
package.tensor("model.layers.0.input_layernorm.weight"), kHidden)),
post_norm_weight(DecodeBf16(
package.tensor("model.layers.0.post_attention_layernorm.weight"), kHidden)),
q_conv_weight(DecodeBf16(
package.tensor("model.layers.0.attention.q_conv1d.weight"), kAttention * 4)),
k_conv_weight(DecodeBf16(
package.tensor("model.layers.0.attention.k_conv1d.weight"), kAttention * 4)),
v_conv_weight(DecodeBf16(
package.tensor("model.layers.0.attention.v_conv1d.weight"), kAttention * 4)),
output_norm_weight(DecodeBf16(
package.tensor("model.layers.0.attention.o_norm.weight"), kHeadDimension)),
a_log(Typed<float>(package.tensor("model.layers.0.attention.A_log"), DataType::kFloat32)),
dt_bias(Typed<float>(package.tensor("model.layers.0.attention.dt_bias"), DataType::kFloat32)),
qkvfgb(MakeLinear(package, "model.layers.0.attention.qkvfgb")),
attention_output(MakeLinear(package, "model.layers.0.attention.o_proj")),
gate_up(MakeLinear(package, "model.layers.0.mlp.gate_up")),
down(MakeLinear(package, "model.layers.0.mlp.down_proj")),
gdn(
Blob(package.tensor("rknn.gdn.heads6"), TensorRole::kRknnIsland),
Blob(package.tensor("rknn.gdn.heads5"), TensorRole::kRknnIsland)),
hidden(kHidden),
normalized(kHidden),
projected(10304),
q(kAttention),
k(kAttention),
v(kAttention),
recurrence_output(kAttention),
attention_vector(kAttention),
attention_result(kHidden),
ffn_projected(2 * kDense),
ffn_hidden(kDense),
ffn_result(kHidden) {
if (embeddings.size() != static_cast<std::size_t>(157184 * kHidden) ||
a_log.size() != kHeads || dt_bias.size() != kAttention) {
throw std::runtime_error("layer 0 raw tensor shapes are incompatible");
}
for (auto & state : convolution_state) state.assign(kAttention * 3, 0.0F);
}
void Reset() {
for (auto & state : convolution_state) std::fill(state.begin(), state.end(), 0.0F);
gdn.Reset();
}
Layer0Timings DecodeToken(std::uint32_t token, std::span<float> output) {
if (token >= 157184 || output.size() != kHidden) {
throw std::invalid_argument("layer 0 token or output is out of range");
}
const auto begin = Clock::now();
const auto embedding = embeddings.subspan(static_cast<std::size_t>(token) * kHidden, kHidden);
for (int index = 0; index < kHidden; ++index) hidden[index] = BFloat16ToFloat(embedding[index]);
RmsNorm(hidden.data(), input_norm_weight.data(), normalized.data(), kHidden, kEpsilon);
qkvfgb->Run(normalized, projected);
CausalConvSilu(
std::span<const float>(projected).subspan(0, kAttention),
q_conv_weight,
convolution_state[0],
q);
CausalConvSilu(
std::span<const float>(projected).subspan(kAttention, kAttention),
k_conv_weight,
convolution_state[1],
k);
CausalConvSilu(
std::span<const float>(projected).subspan(2 * kAttention, kAttention),
v_conv_weight,
convolution_state[2],
v);
NormalizeHeads(q);
NormalizeHeads(k);
const auto f = std::span<const float>(projected).subspan(3 * kAttention, kAttention);
const auto gate = std::span<const float>(projected).subspan(4 * kAttention, kAttention);
const auto beta_logits = std::span<const float>(projected).subspan(5 * kAttention, kHeads);
std::vector<float> decay(kAttention);
std::vector<float> beta(kHeads);
for (int head = 0; head < kHeads; ++head) {
const float a = std::exp(a_log[head]);
beta[head] = 1.0F / (1.0F + std::exp(-beta_logits[head]));
for (int index = 0; index < kHeadDimension; ++index) {
const int offset = head * kHeadDimension + index;
const float argument = a * (f[offset] + dt_bias[offset]);
decay[offset] = -5.0F / (1.0F + std::exp(-argument));
}
}
gdn.Run(q, k, v, decay, beta, recurrence_output);
for (int head = 0; head < kHeads; ++head) {
const int base = head * kHeadDimension;
float sum = 0.0F;
for (int index = 0; index < kHeadDimension; ++index) {
const float value = recurrence_output[base + index];
sum += value * value;
}
const float inverse = 1.0F /
std::sqrt(sum / static_cast<float>(kHeadDimension) + kEpsilon);
for (int index = 0; index < kHeadDimension; ++index) {
const int offset = base + index;
const float sigmoid_gate = 1.0F / (1.0F + std::exp(-gate[offset]));
attention_vector[offset] = recurrence_output[offset] * inverse *
output_norm_weight[index] * sigmoid_gate;
}
}
attention_output->Run(attention_vector, attention_result);
for (int index = 0; index < kHidden; ++index) hidden[index] += attention_result[index];
const auto attention_end = Clock::now();
RmsNorm(hidden.data(), post_norm_weight.data(), normalized.data(), kHidden, kEpsilon);
gate_up->Run(normalized, ffn_projected);
SiluMultiply(ffn_projected.data(), ffn_projected.data() + kDense, ffn_hidden.data(), kDense);
down->Run(ffn_hidden, ffn_result);
for (int index = 0; index < kHidden; ++index) output[index] = hidden[index] + ffn_result[index];
const auto end = Clock::now();
return {
Milliseconds(begin, attention_end),
Milliseconds(attention_end, end),
Milliseconds(begin, end),
};
}
};
Layer0::Layer0(const ModelPackage & package) : impl_(std::make_unique<Impl>(package)) {}
Layer0::~Layer0() = default;
void Layer0::Reset() { impl_->Reset(); }
Layer0Timings Layer0::DecodeToken(std::uint32_t token, std::span<float> output) {
return impl_->DecodeToken(token, output);
}
} // namespace ling3