Download include/ling3/w4_linear.h from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 2.47 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/include/ling3/w4_linear.h
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/include/ling3/w4_linear.h
-
curl -L -o w4_linear.h https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/include/ling3/w4_linear.h
2.47 kB
| namespace ling3 { | |
| struct W4LinearConfig { | |
| int k = 0; | |
| int n = 0; | |
| int k_splits = 1; | |
| std::vector<int> cores {0, 1, 2}; | |
| int iommu_domain_id = 0; | |
| }; | |
| struct W4RunTimings { | |
| double quantize_pack_ms = 0.0; | |
| double input_sync_ms = 0.0; | |
| double npu_ms = 0.0; | |
| double gather_ms = 0.0; | |
| double total_ms = 0.0; | |
| float input_scale = 1.0F; | |
| }; | |
| class DynamicW4Linear { | |
| public: | |
| DynamicW4Linear( | |
| W4LinearConfig config, | |
| std::span<const std::byte> packed_weights, | |
| std::span<const float> weight_scales, | |
| std::span<const std::int32_t> correction); | |
| ~DynamicW4Linear(); | |
| DynamicW4Linear(const DynamicW4Linear &) = delete; | |
| DynamicW4Linear & operator=(const DynamicW4Linear &) = delete; | |
| DynamicW4Linear(DynamicW4Linear &&) noexcept; | |
| DynamicW4Linear & operator=(DynamicW4Linear &&) noexcept; | |
| W4RunTimings Run(std::span<const float> input, std::span<float> output); | |
| W4RunTimings RunBatch( | |
| std::span<const float> input, | |
| std::size_t rows, | |
| std::span<float> output); | |
| W4RunTimings RunBatchWithWeights( | |
| std::span<const float> input, | |
| std::size_t rows, | |
| const DynamicW4Linear & weights, | |
| std::span<float> output); | |
| // Select rows from a shared, per-token INT8 activation table. Quantization | |
| // scales belong to source rows; expert weights retain their own scales. | |
| W4RunTimings RunBatchQuantizedRows( | |
| std::span<const std::int8_t> input, | |
| std::span<const float> input_scales, | |
| std::span<const std::size_t> row_indices, | |
| const DynamicW4Linear & weights, | |
| std::span<float> output); | |
| W4RunTimings RunBatch32(std::span<const float> input, std::span<float> output); | |
| // Indexed inputs already have a shared INT8 table, so no private quantizer | |
| // scratch is needed. A later ordinary RunBatch can allocate it on demand. | |
| void PrepareBatch(std::size_t rows, bool indexed_input = false); | |
| float PrepareInput(std::span<const float> input); | |
| W4RunTimings RunPrepared(float input_scale, std::span<float> output); | |
| void ShareInputFrom(DynamicW4Linear & owner); | |
| void SetSingleCore(int core); | |
| const W4LinearConfig & config() const noexcept; | |
| std::size_t resident_weight_bytes() const noexcept; | |
| private: | |
| struct Impl; | |
| std::unique_ptr<Impl> impl_; | |
| }; | |
| } // namespace ling3 | |