#pragma once #include "ling3/mla_stats.h" #include #include #include #include #include namespace ling3 { // One workspace per decoder, shared by all six MLA layers. No persistent KV: // checkpoint/rewind semantics continue to belong to the decoder's BF16 cache. class MlaNpu { public: MlaNpu(); ~MlaNpu(); void Prepare(std::size_t rows); bool Run(std::span q, std::span k, std::span v, std::size_t rows, std::size_t history, std::span output); void CpuCall(); MlaBackendStats Stats() const; private: struct Impl; std::unique_ptr impl_; }; }