#include "ling3/decoder.h" #include "ling3/model_package.h" #include "ling3/tokenizer.h" #include #include #include #include #include #include #include #include #include #include // Diagnostic entry only: exact official token IDs, one warmed decoder for all cases. int main(int argc, char **argv) { using Json = nlohmann::json; using Clock = std::chrono::steady_clock; try { if (argc != 4) throw std::invalid_argument("usage: quant-loss-probe MODEL.l3r SUITE.json OUTPUT_DIR"); for (auto key : {"LING3_PREFILL_W4A4", "LING3_EXPERT_ALL_CORES", "LING3_GDN_PREFILL_DIR"}) unsetenv(key); for (auto key : {"LING3_PREWARM_EXPERTS", "LING3_EXPERT_BALANCED", "LING3_EXPERT_ZERO_COPY", "LING3_GDN_CPU_PREFILL", "LING3_GDN_CPU_FP32_STATE", "LING3_GDN_CPU_DECODE", "LING3_GDN_FULL_FP32", "LING3_MLA_SIMD", "LING3_VECTOR_MATH"}) setenv(key,"1",1); setenv("LING3_MLA_BACKEND", "auto", 1); cpu_set_t affinity; CPU_ZERO(&affinity); for (int core=4; core<8; ++core) CPU_SET(core,&affinity); if (sched_setaffinity(0,sizeof(affinity),&affinity)) throw std::runtime_error("affinity failed"); rlimit limit{}; if (getrlimit(RLIMIT_NOFILE,&limit)) throw std::runtime_error("getrlimit failed"); limit.rlim_cur=std::min(262144,limit.rlim_max); if (setrlimit(RLIMIT_NOFILE,&limit)) throw std::runtime_error("setrlimit failed"); std::ifstream suite_file(argv[2]); Json suite; suite_file >> suite; const ling3::ModelPackage model(argv[1]); const auto &t=model.tensor("tokenizer"); ling3::Tokenizer tokenizer({t.data,static_cast(t.entry->data_bytes)}); std::size_t capacity=4096; for(const auto &c:suite.at("cases")) capacity=std::max(capacity,c.at("prompt_ids").size()+c.at("teacher_ids").size()); ling3::Decoder decoder(model,capacity); for(std::size_t rows=1;rows<=128;rows*=2)decoder.PrepareBatch(rows); std::vector logits(model.header().vocab_size); decoder.Eval(16,logits); decoder.Reset(); const std::filesystem::path output(argv[3]); std::filesystem::create_directories(output); Json results=Json::array(); for(const auto &c:suite.at("cases")) { const auto name=c.at("id").get(); const auto prompt=c.at("prompt_ids").get>(); auto teacher=c.at("teacher_ids").get>(); if(prompt.empty()||teacher.empty())throw std::runtime_error("empty sequence"); if(tokenizer.Encode(c.at("prompt_text").get())!=prompt || tokenizer.Encode(c.at("target_text").get())!=teacher) throw std::runtime_error("official/C++ tokenizer mismatch: "+name); const auto full_teacher_tokens=teacher.size(); if(const char* limit=std::getenv("LING3_LOSS_STEPS")){ const auto count=std::stoul(limit); if(count<1)throw std::invalid_argument("LING3_LOSS_STEPS must be positive"); teacher.resize(std::min(teacher.size(),count)); } decoder.Reset(); const auto before=decoder.AttentionStats(); const auto start=Clock::now(); for(std::size_t offset=0;offset(128,prompt.size()-offset); const auto block=std::span(prompt).subspan(offset,rows); if(rows==1)decoder.Eval(block[0],logits); else if(offset+rows==prompt.size())decoder.EvalBatch(block,logits); else decoder.EvalBatchState(block); offset+=rows; } const auto prefill=Clock::now(); std::ofstream values(output/(name+".rknn.f32"),std::ios::binary); if(!values)throw std::runtime_error("cannot open logit output"); for(std::size_t step=0;step(prefill-start).count()}, {"total_ms",std::chrono::duration(end-start).count()}}; results.push_back(row);std::cout<