Ling-3.0-tiny-RKNN / src /execution_plan.cpp
Sariel00's picture
Publish Ling-3.0-tiny RKNN engine and model
3fd1a35 verified
Raw History Blame Contribute Delete
1.42 kB
#include "ling3/execution_plan.h"
namespace ling3 {
ExecutionPlan BuildExecutionPlan(std::size_t max_context) {
ExecutionPlan plan;
std::size_t kda_layers = 0;
std::size_t mla_layers = 0;
for (std::size_t layer = 0; layer < plan.layers.size(); ++layer) {
const bool mla = (layer + 1) % 4 == 0;
plan.layers[layer] = {
static_cast<std::uint8_t>(layer),
mla ? AttentionKind::kMla : AttentionKind::kKda,
layer == 0 ? FfnKind::kDense : FfnKind::kMoe,
};
kda_layers += !mla;
mla_layers += mla;
}
constexpr std::size_t routed_expert_contexts = 128 * 2;
constexpr std::size_t shared_expert_contexts = 2;
constexpr std::size_t kda_projection_contexts = 7;
constexpr std::size_t mla_projection_contexts = 6;
plan.decode_matmul_contexts =
23 * (routed_expert_contexts + shared_expert_contexts) +
kda_layers * kda_projection_contexts + mla_layers * mla_projection_contexts + 3;
plan.rknn_island_contexts = kda_layers * 3 + 23 * 3 + 24;
plan.kda_state_bytes_fp16 = kda_layers * 16 * 128 * 128 * sizeof(std::uint16_t);
plan.mla_cache_bytes_fp16 = mla_layers * max_context * (512 + 64) * sizeof(std::uint16_t);
plan.activation_arena_bytes =
2 * 1536 + 3 * (1024 * sizeof(std::int32_t) + 1536 * sizeof(std::int32_t)) + 16 * 128 * 4;
return plan;
}
} // namespace ling3