#pragma once #include #include #include #include #include namespace ling3 { struct W4LinearConfig { int k = 0; int n = 0; int k_splits = 1; std::vector cores {0, 1, 2}; int iommu_domain_id = 0; }; struct W4RunTimings { double quantize_pack_ms = 0.0; double input_sync_ms = 0.0; double npu_ms = 0.0; double gather_ms = 0.0; double total_ms = 0.0; float input_scale = 1.0F; }; class DynamicW4Linear { public: DynamicW4Linear( W4LinearConfig config, std::span packed_weights, std::span weight_scales, std::span correction); ~DynamicW4Linear(); DynamicW4Linear(const DynamicW4Linear &) = delete; DynamicW4Linear & operator=(const DynamicW4Linear &) = delete; DynamicW4Linear(DynamicW4Linear &&) noexcept; DynamicW4Linear & operator=(DynamicW4Linear &&) noexcept; W4RunTimings Run(std::span input, std::span output); W4RunTimings RunBatch( std::span input, std::size_t rows, std::span output); W4RunTimings RunBatchWithWeights( std::span input, std::size_t rows, const DynamicW4Linear & weights, std::span output); // Select rows from a shared, per-token INT8 activation table. Quantization // scales belong to source rows; expert weights retain their own scales. W4RunTimings RunBatchQuantizedRows( std::span input, std::span input_scales, std::span row_indices, const DynamicW4Linear & weights, std::span output); W4RunTimings RunBatch32(std::span input, std::span output); // Indexed inputs already have a shared INT8 table, so no private quantizer // scratch is needed. A later ordinary RunBatch can allocate it on demand. void PrepareBatch(std::size_t rows, bool indexed_input = false); float PrepareInput(std::span input); W4RunTimings RunPrepared(float input_scale, std::span output); void ShareInputFrom(DynamicW4Linear & owner); void SetSingleCore(int core); const W4LinearConfig & config() const noexcept; std::size_t resident_weight_bytes() const noexcept; private: struct Impl; std::unique_ptr impl_; }; } // namespace ling3