Sariel00's picture
Publish Ling-3.0-tiny RKNN engine and model
3fd1a35 verified
Raw History Blame Contribute Delete
2.47 kB
#pragma once
#include <cstddef>
#include <cstdint>
#include <memory>
#include <span>
#include <vector>
namespace ling3 {
struct W4LinearConfig {
int k = 0;
int n = 0;
int k_splits = 1;
std::vector<int> cores {0, 1, 2};
int iommu_domain_id = 0;
};
struct W4RunTimings {
double quantize_pack_ms = 0.0;
double input_sync_ms = 0.0;
double npu_ms = 0.0;
double gather_ms = 0.0;
double total_ms = 0.0;
float input_scale = 1.0F;
};
class DynamicW4Linear {
public:
DynamicW4Linear(
W4LinearConfig config,
std::span<const std::byte> packed_weights,
std::span<const float> weight_scales,
std::span<const std::int32_t> correction);
~DynamicW4Linear();
DynamicW4Linear(const DynamicW4Linear &) = delete;
DynamicW4Linear & operator=(const DynamicW4Linear &) = delete;
DynamicW4Linear(DynamicW4Linear &&) noexcept;
DynamicW4Linear & operator=(DynamicW4Linear &&) noexcept;
W4RunTimings Run(std::span<const float> input, std::span<float> output);
W4RunTimings RunBatch(
std::span<const float> input,
std::size_t rows,
std::span<float> output);
W4RunTimings RunBatchWithWeights(
std::span<const float> input,
std::size_t rows,
const DynamicW4Linear & weights,
std::span<float> output);
// Select rows from a shared, per-token INT8 activation table. Quantization
// scales belong to source rows; expert weights retain their own scales.
W4RunTimings RunBatchQuantizedRows(
std::span<const std::int8_t> input,
std::span<const float> input_scales,
std::span<const std::size_t> row_indices,
const DynamicW4Linear & weights,
std::span<float> output);
W4RunTimings RunBatch32(std::span<const float> input, std::span<float> output);
// Indexed inputs already have a shared INT8 table, so no private quantizer
// scratch is needed. A later ordinary RunBatch can allocate it on demand.
void PrepareBatch(std::size_t rows, bool indexed_input = false);
float PrepareInput(std::span<const float> input);
W4RunTimings RunPrepared(float input_scale, std::span<float> output);
void ShareInputFrom(DynamicW4Linear & owner);
void SetSingleCore(int core);
const W4LinearConfig & config() const noexcept;
std::size_t resident_weight_bytes() const noexcept;
private:
struct Impl;
std::unique_ptr<Impl> impl_;
};
} // namespace ling3