#include "ling3/decoder.h" #include "ling3/execution_plan.h" #include "ling3/gdn_step.h" #include "ling3/layer0.h" #include "ling3/model_package.h" #include "ling3/quantization.h" #include "ling3/rknn_backend.h" #include "ling3/service.h" #include "ling3/tokenizer.h" #include "ling3/w4_linear.h" #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #include #if defined(__linux__) #include #include #include #include #endif namespace { std::size_t ParseSize(std::string_view text, const char * label) { std::size_t value = 0; const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value); if (error != std::errc {} || end != text.data() + text.size()) { throw std::runtime_error(std::string("invalid ") + label + ": " + std::string(text)); } return value; } void PrintUsage(const char * program) { std::cerr << "Usage:\n" << " " << program << " inspect MODEL.l3r\n" << " " << program << " plan [MAX_CONTEXT]\n" << " " << program << " rknn-smoke [ITERATIONS]\n" << " " << program << " tokenizer-smoke MODEL.l3r\n" << " " << program << " gdn-smoke MODEL.l3r [ITERATIONS]\n" << " " << program << " gdn-batch16-smoke MODEL.l3r [ITERATIONS]\n" << " " << program << " gdn-cpu-state-check MODEL.l3r\n" << " " << program << " gdn-reference-check MODEL.l3r\n" << " " << program << " layer0-smoke MODEL.l3r TOKEN REF.f32 [ITERATIONS]\n" << " " << program << " linear-smoke MODEL.l3r TENSOR_BASE [ITERATIONS]\n" << " " << program << " linear-batch-smoke MODEL.l3r TENSOR_BASE ROWS [ITERATIONS] [WEIGHT_BASE]\n" << " " << program << " prefill32-smoke MODEL.l3r\n" << " " << program << " logits-smoke MODEL.l3r TOKEN [REF.f32]\n" << " " << program << " generate MODEL.l3r USER_TEXT [MAX_NEW_TOKENS]\n" << " " << program << " benchmark MODEL.l3r SEQLEN NEW_TOKENS [ITERATIONS]\n" #if LING3_EXPERIMENTAL_MTP << " " << program << " mtp-benchmark MODEL.l3r SEQLEN NEW_TOKENS ITERATIONS CONTEXT [CASES.txt]\n" #endif << " " << program << " sequence-check MODEL.l3r SEQLEN NEW_TOKENS OUT_PREFIX [REFERENCE_PREFIX [CORPUS.txt]]\n" << " " << program << " serve MODEL.l3r [PORT]\n"; } template std::span TensorSpan(const ling3::TensorView & tensor); void PinToA76Cores() { #if defined(__linux__) && defined(__aarch64__) cpu_set_t cores; CPU_ZERO(&cores); for (int core = 4; core <= 7; ++core) CPU_SET(core, &cores); if (sched_setaffinity(0, sizeof(cores), &cores) != 0) { std::cerr << "warning: cannot pin inference threads to CPU cores 4-7: " << std::strerror(errno) << '\n'; } #endif } void RaiseFileDescriptorLimit() { #if defined(__linux__) rlimit limit {}; if (getrlimit(RLIMIT_NOFILE, &limit) != 0) { throw std::runtime_error( std::string("getrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno)); } constexpr rlim_t requested = 262144; const rlim_t target = std::min(requested, limit.rlim_max); if (limit.rlim_cur >= target) return; limit.rlim_cur = target; if (setrlimit(RLIMIT_NOFILE, &limit) != 0) { throw std::runtime_error( std::string("setrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno)); } #endif } void ConfigureInferenceProcess() { RaiseFileDescriptorLimit(); PinToA76Cores(); } ling3::Tokenizer PackageTokenizer(const ling3::ModelPackage & package) { const auto & asset = package.tensor("tokenizer"); if (asset.entry->role != static_cast(ling3::TensorRole::kTokenizer)) { throw std::runtime_error("tokenizer tensor has the wrong role"); } return ling3::Tokenizer(TensorSpan(asset)); } std::string Utf8(std::u8string_view text) { return {reinterpret_cast(text.data()), text.size()}; } int RunTokenizerSmoke(const std::string & package_path) { const ling3::ModelPackage package(package_path); const auto & asset = package.tensor("tokenizer"); if (asset.entry->role != static_cast(ling3::TensorRole::kTokenizer)) { throw std::runtime_error("tokenizer tensor has the wrong role"); } const ling3::Tokenizer tokenizer(TensorSpan(asset)); struct Case { std::string text; std::vector expected; }; const std::vector cases = { {Utf8(u8"你好,世界!"), {34355, 44291, 859}}, {"Hello world!", {14455, 1931, 0}}, {"I'm testing 123.", {40, 3180, 7750, 220, 16, 17, 18, 13}}, {Utf8(u8"ABC e\u0301\n第二行"), {83140, 270, 95, 93086, 18878, 198, 2378, 685}}, {Utf8(u8"想一下"), {156903, 122843, 156904}}, { Utf8(u8"<|startoftext|>user\n你好<|role_end|>\n<|startoftext|>assistant\n"), {156891, 3840, 198, 34355, 156895, 198, 156891, 598, 10450, 198}, }, }; for (std::size_t index = 0; index < cases.size(); ++index) { const auto actual = tokenizer.Encode(cases[index].text); if (actual != cases[index].expected) { std::cerr << "tokenizer mismatch in case " << index << "\nexpected:"; for (auto token : cases[index].expected) std::cerr << ' ' << token; std::cerr << "\nactual:"; for (auto token : actual) std::cerr << ' ' << token; std::cerr << '\n'; return 8; } if (tokenizer.Decode(actual) != cases[index].text && index != 3) { std::cerr << "tokenizer round-trip mismatch in case " << index << '\n'; return 9; } } if (!tokenizer.uses_official_unicode_rules()) { std::cerr << "token IDs matched, but this build lacks official ICU Unicode rules\n"; return 10; } std::cout << "tokenizer smoke passed (" << cases.size() << " official vectors, ICU rules enabled)\n"; return 0; } int RunGdnSmoke(const std::string & package_path, std::size_t iterations) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("gdn-smoke requires a build with RKNN support"); } if (iterations == 0) iterations = 1; const ling3::ModelPackage package(package_path); const auto & heads6 = package.tensor("rknn.gdn.heads6"); const auto & heads5 = package.tensor("rknn.gdn.heads5"); if (heads6.entry->role != static_cast(ling3::TensorRole::kRknnIsland) || heads5.entry->role != static_cast(ling3::TensorRole::kRknnIsland)) { throw std::runtime_error("GDN tensors do not contain RKNN islands"); } constexpr int heads = 16; constexpr int dimension = 128; constexpr int elements = heads * dimension; std::vector query(elements), key(elements), value(elements), decay(elements); std::vector beta(heads), output(elements), reference(elements); for (int head = 0; head < heads; ++head) { float q_norm = 1.0e-6F; float k_norm = 1.0e-6F; for (int index = 0; index < dimension; ++index) { const int offset = head * dimension + index; query[offset] = std::sin(static_cast(offset + 1) * 0.013F); key[offset] = std::cos(static_cast(offset + 3) * 0.017F); value[offset] = 0.2F * std::sin(static_cast(offset + 5) * 0.021F); decay[offset] = -2.0F - 0.5F * std::cos(static_cast(offset) * 0.009F); q_norm += query[offset] * query[offset]; k_norm += key[offset] * key[offset]; } q_norm = std::sqrt(q_norm); k_norm = std::sqrt(k_norm); for (int index = 0; index < dimension; ++index) { query[head * dimension + index] /= q_norm; key[head * dimension + index] /= k_norm; } beta[head] = 1.0F / (1.0F + std::exp(-0.2F * std::sin(static_cast(head)))); } const auto initialize_begin = std::chrono::steady_clock::now(); ling3::GdnStep gdn(TensorSpan(heads6), TensorSpan(heads5)); const auto initialize_end = std::chrono::steady_clock::now(); const double initialize_ms = std::chrono::duration( initialize_end - initialize_begin).count(); gdn.Run(query, key, value, decay, beta, output); const float query_scale = 1.0F / std::sqrt(static_cast(dimension)); for (int head = 0; head < heads; ++head) { float key_dot_query = 0.0F; for (int index = 0; index < dimension; ++index) { key_dot_query += key[head * dimension + index] * query[head * dimension + index] * query_scale; } for (int index = 0; index < dimension; ++index) { reference[head * dimension + index] = beta[head] * value[head * dimension + index] * key_dot_query; } } double dot = 0.0; double norm_output = 0.0; double norm_reference = 0.0; double max_error = 0.0; double mean_error = 0.0; for (int index = 0; index < elements; ++index) { const double error = std::abs(static_cast(output[index]) - reference[index]); max_error = std::max(max_error, error); mean_error += error; dot += static_cast(output[index]) * reference[index]; norm_output += static_cast(output[index]) * output[index]; norm_reference += static_cast(reference[index]) * reference[index]; } mean_error /= elements; const double cosine = dot / std::sqrt(norm_output * norm_reference); gdn.Reset(); for (int index = 0; index < 5; ++index) { gdn.Run(query, key, value, decay, beta, output); } ling3::GdnRunTimings mean; for (std::size_t index = 0; index < iterations; ++index) { const auto sample = gdn.Run(query, key, value, decay, beta, output); mean.stage_ms += sample.stage_ms; mean.npu_ms += sample.npu_ms; mean.collect_ms += sample.collect_ms; mean.total_ms += sample.total_ms; } const double divisor = static_cast(iterations); mean.stage_ms /= divisor; mean.npu_ms /= divisor; mean.collect_ms /= divisor; mean.total_ms /= divisor; std::cout << std::fixed << std::setprecision(6) << "heads=6+5+5\n" << "cores=0,1,2\n" << "initialization_ms=" << initialize_ms << '\n' << "state_bytes=" << gdn.state_bytes() << '\n' << "first_step_cosine=" << cosine << '\n' << "first_step_mean_abs_error=" << mean_error << '\n' << "first_step_max_abs_error=" << max_error << '\n' << "iterations=" << iterations << '\n' << "stage_mean_ms=" << mean.stage_ms << '\n' << "npu_mean_ms=" << mean.npu_ms << '\n' << "collect_mean_ms=" << mean.collect_ms << '\n' << "total_mean_ms=" << mean.total_ms << '\n'; return cosine >= 0.999 && max_error <= 0.005 ? 0 : 11; } void FillGdnToken( int token, std::span query, std::span key, std::span value, std::span decay, std::span beta) { constexpr int heads = 16; constexpr int dimension = 128; for (int head = 0; head < heads; ++head) { float q_norm = 1.0e-6F; float k_norm = 1.0e-6F; for (int index = 0; index < dimension; ++index) { const int offset = head * dimension + index; query[offset] = std::sin(static_cast(offset + 13 * token + 1) * 0.013F); key[offset] = std::cos(static_cast(offset + 7 * token + 3) * 0.017F); value[offset] = 0.2F * std::sin(static_cast(offset + 5 * token + 5) * 0.021F); decay[offset] = -2.0F - 0.5F * std::cos(static_cast(offset + 3 * token) * 0.009F); q_norm += query[offset] * query[offset]; k_norm += key[offset] * key[offset]; } q_norm = std::sqrt(q_norm); k_norm = std::sqrt(k_norm); for (int index = 0; index < dimension; ++index) { query[head * dimension + index] /= q_norm; key[head * dimension + index] /= k_norm; } beta[head] = 1.0F / (1.0F + std::exp(-0.2F * std::sin(static_cast(head + token)))); } } int RunGdnBatch16Smoke(const std::string & package_path, std::size_t iterations) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("gdn-batch16-smoke requires a build with RKNN support"); } if (iterations == 0) iterations = 1; const ling3::ModelPackage package(package_path); const auto & heads6 = package.tensor("rknn.gdn.heads6"); const auto & heads5 = package.tensor("rknn.gdn.heads5"); constexpr std::size_t width = 16 * 128; constexpr std::size_t rows = 16; ling3::GdnStep sequential(TensorSpan(heads6), TensorSpan(heads5)); ling3::GdnStep batch(TensorSpan(heads6), TensorSpan(heads5)); if (!batch.has_batch16()) { throw std::runtime_error("set LING3_GDN_PREFILL_DIR to the 16-token RKNN models"); } std::vector query(rows * width), key(rows * width), value(rows * width); std::vector decay(rows * width), beta(rows * 16); std::vector sequential_output(rows * width), batch_output(rows * width); std::vector scratch(width), scratch_key(width), scratch_value(width), scratch_decay(width); std::vector scratch_beta(16), sequential_tail(width), batch_tail(width); for (int token = 0; token < 3; ++token) { FillGdnToken(token, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta); sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail); batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail); } for (std::size_t row = 0; row < rows; ++row) { FillGdnToken( static_cast(row + 3), std::span(query).subspan(row * width, width), std::span(key).subspan(row * width, width), std::span(value).subspan(row * width, width), std::span(decay).subspan(row * width, width), std::span(beta).subspan(row * 16, 16)); sequential.Run( std::span(query).subspan(row * width, width), std::span(key).subspan(row * width, width), std::span(value).subspan(row * width, width), std::span(decay).subspan(row * width, width), std::span(beta).subspan(row * 16, 16), std::span(sequential_output).subspan(row * width, width)); } const auto first_timing = batch.RunBatch16( query, key, value, decay, beta, batch_output); FillGdnToken(19, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta); sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail); batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail); auto compare = [](std::span actual, std::span expected) { std::array result {}; double actual_norm = 0.0; double expected_norm = 0.0; for (std::size_t index = 0; index < actual.size(); ++index) { const double a = actual[index]; const double b = expected[index]; result[0] += a * b; actual_norm += a * a; expected_norm += b * b; const double error = std::abs(a - b); result[1] += error; result[2] = std::max(result[2], error); } result[0] /= std::sqrt(actual_norm * expected_norm); result[1] /= static_cast(actual.size()); return result; }; const auto output_error = compare(batch_output, sequential_output); const auto state_error = compare(batch_tail, sequential_tail); ling3::GdnRunTimings mean = first_timing; for (std::size_t iteration = 1; iteration < iterations; ++iteration) { const auto sample = batch.RunBatch16(query, key, value, decay, beta, batch_output); mean.stage_ms += sample.stage_ms; mean.npu_ms += sample.npu_ms; mean.collect_ms += sample.collect_ms; mean.total_ms += sample.total_ms; } const double divisor = static_cast(iterations); mean.stage_ms /= divisor; mean.npu_ms /= divisor; mean.collect_ms /= divisor; mean.total_ms /= divisor; std::cout << std::fixed << std::setprecision(8) << "rows=16\n" << "history_tokens=3\n" << "output_cosine=" << output_error[0] << '\n' << "output_mean_abs_error=" << output_error[1] << '\n' << "output_max_abs_error=" << output_error[2] << '\n' << "tail_state_cosine=" << state_error[0] << '\n' << "tail_mean_abs_error=" << state_error[1] << '\n' << "tail_max_abs_error=" << state_error[2] << '\n' << "stage_mean_ms=" << mean.stage_ms << '\n' << "npu_mean_ms=" << mean.npu_ms << '\n' << "collect_mean_ms=" << mean.collect_ms << '\n' << "total_mean_ms=" << mean.total_ms << '\n'; return output_error[0] >= 0.99999 && state_error[0] >= 0.99999 ? 0 : 14; } std::vector ReadFloatFile(const std::string & path, std::size_t count) { std::ifstream stream(path, std::ios::binary | std::ios::ate); if (!stream) throw std::runtime_error("cannot open " + path); const auto bytes = static_cast(stream.tellg()); if (bytes != count * sizeof(float)) { throw std::runtime_error(path + " has an incompatible byte count"); } stream.seekg(0); std::vector result(count); stream.read(reinterpret_cast(result.data()), static_cast(bytes)); if (!stream) throw std::runtime_error("cannot read " + path); return result; } int RunLayer0Smoke( const std::string & package_path, std::uint32_t token, const std::string & reference_path, std::size_t iterations) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("layer0-smoke requires a build with RKNN support"); } if (iterations == 0) iterations = 1; const ling3::ModelPackage package(package_path); const auto initialization_begin = std::chrono::steady_clock::now(); ling3::Layer0 layer(package); const auto initialization_end = std::chrono::steady_clock::now(); const double initialization_ms = std::chrono::duration( initialization_end - initialization_begin).count(); std::vector output(1536); layer.Reset(); layer.DecodeToken(token, output); const auto reference = ReadFloatFile(reference_path, output.size()); double dot = 0.0, output_norm = 0.0, reference_norm = 0.0; double mean_error = 0.0, maximum_error = 0.0; for (std::size_t index = 0; index < output.size(); ++index) { const double error = std::abs(static_cast(output[index]) - reference[index]); mean_error += error; maximum_error = std::max(maximum_error, error); dot += static_cast(output[index]) * reference[index]; output_norm += static_cast(output[index]) * output[index]; reference_norm += static_cast(reference[index]) * reference[index]; } mean_error /= output.size(); const double cosine = dot / std::sqrt(output_norm * reference_norm); layer.Reset(); for (int index = 0; index < 3; ++index) layer.DecodeToken(token, output); ling3::Layer0Timings mean; for (std::size_t index = 0; index < iterations; ++index) { const auto sample = layer.DecodeToken(token, output); mean.attention_ms += sample.attention_ms; mean.dense_ffn_ms += sample.dense_ffn_ms; mean.total_ms += sample.total_ms; } const double divisor = static_cast(iterations); mean.attention_ms /= divisor; mean.dense_ffn_ms /= divisor; mean.total_ms /= divisor; std::cout << std::fixed << std::setprecision(6) << "token=" << token << '\n' << "initialization_ms=" << initialization_ms << '\n' << "cosine_vs_official_bf16=" << cosine << '\n' << "mean_abs_error=" << mean_error << '\n' << "max_abs_error=" << maximum_error << '\n' << "iterations=" << iterations << '\n' << "attention_mean_ms=" << mean.attention_ms << '\n' << "dense_ffn_mean_ms=" << mean.dense_ffn_ms << '\n' << "total_mean_ms=" << mean.total_ms << '\n'; return cosine >= 0.98 ? 0 : 12; } template std::span TensorSpan(const ling3::TensorView & tensor) { if (tensor.entry->data_bytes % sizeof(T) != 0) { throw std::runtime_error(std::string(tensor.name) + " has an invalid byte count"); } return { reinterpret_cast(tensor.data), static_cast(tensor.entry->data_bytes / sizeof(T)), }; } int RunLinearSmoke( const std::string & package_path, const std::string & base, std::size_t iterations) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("linear-smoke requires a build with RKNN support"); } if (iterations == 0) iterations = 1; const ling3::ModelPackage package(package_path); ling3::ValidateLing3Tiny(package.header()); const auto & weight = package.tensor(base + ".weight"); const auto & scale = package.tensor(base + ".scales"); const auto & correction = package.tensor(base + ".correction"); if (weight.entry->dtype != static_cast(ling3::DataType::kInt4Low) || weight.entry->layout != static_cast(ling3::TensorLayout::kPackedInt4Low) || weight.entry->rank != 2 || scale.entry->dtype != static_cast(ling3::DataType::kFloat32) || correction.entry->dtype != static_cast(ling3::DataType::kInt32)) { throw std::runtime_error("linear-smoke tensors have incompatible metadata"); } const int k = static_cast(weight.entry->dims[0]); const int n = static_cast(weight.entry->dims[1]); const int k_splits = static_cast(weight.entry->flags); const auto weights = TensorSpan(weight); const auto scales = TensorSpan(scale); const auto corrections = TensorSpan(correction); std::vector input(k); for (int index = 0; index < k; ++index) { input[index] = 0.7F * std::sin(static_cast(index) * 0.03125F) + 0.2F * std::cos(static_cast(index) * 0.0078125F); } std::vector output(n); const auto initialize_begin = std::chrono::steady_clock::now(); ling3::DynamicW4Linear linear( {k, n, k_splits, {0, 1, 2}}, weights, scales, corrections); const auto initialize_end = std::chrono::steady_clock::now(); const auto initialize_ms = std::chrono::duration( initialize_end - initialize_begin).count(); linear.Run(input, output); std::vector samples; samples.reserve(iterations); for (std::size_t iteration = 0; iteration < iterations; ++iteration) { samples.push_back(linear.Run(input, output)); } std::vector input_codes(k); const auto quantization = ling3::QuantizeSymmetricInt8(input, input_codes); std::vector reference_accumulator(n); ling3::ReferenceW4Linear(input_codes, weights, n, reference_accumulator); std::vector reference(n); ling3::DequantizePerChannel(reference_accumulator, quantization.scale, scales, reference); double maximum_error = 0.0; double mean_error = 0.0; for (int index = 0; index < n; ++index) { const double error = std::abs(static_cast(output[index]) - reference[index]); maximum_error = std::max(maximum_error, error); mean_error += error; } mean_error /= n; ling3::W4RunTimings mean; for (const auto & sample : samples) { mean.quantize_pack_ms += sample.quantize_pack_ms; mean.input_sync_ms += sample.input_sync_ms; mean.npu_ms += sample.npu_ms; mean.gather_ms += sample.gather_ms; mean.total_ms += sample.total_ms; } const double divisor = static_cast(samples.size()); mean.quantize_pack_ms /= divisor; mean.input_sync_ms /= divisor; mean.npu_ms /= divisor; mean.gather_ms /= divisor; mean.total_ms /= divisor; std::cout << std::fixed << std::setprecision(6) << "tensor=" << base << '\n' << "k=" << k << '\n' << "n=" << n << '\n' << "k_splits=" << k_splits << '\n' << "cores=0,1,2\n" << "initialization_ms=" << initialize_ms << '\n' << "resident_weight_bytes=" << linear.resident_weight_bytes() << '\n' << "iterations=" << iterations << '\n' << "quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n' << "input_sync_mean_ms=" << mean.input_sync_ms << '\n' << "npu_mean_ms=" << mean.npu_ms << '\n' << "gather_mean_ms=" << mean.gather_ms << '\n' << "total_mean_ms=" << mean.total_ms << '\n' << "reference_mean_abs_error=" << mean_error << '\n' << "reference_max_abs_error=" << maximum_error << '\n'; return maximum_error <= 1.0e-5 ? 0 : 7; } int RunLinearBatchSmoke( const std::string & package_path, const std::string & base, std::size_t rows, std::size_t iterations, const std::string & weight_base) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("linear-batch-smoke requires a build with RKNN support"); } if (rows < 1 || rows > 128) throw std::runtime_error("rows must be in [1, 128]"); if (iterations == 0) iterations = 1; const ling3::ModelPackage package(package_path); ling3::ValidateLing3Tiny(package.header()); const auto & weight = package.tensor(base + ".weight"); const auto & scale = package.tensor(base + ".scales"); const auto & correction = package.tensor(base + ".correction"); const int k = static_cast(weight.entry->dims[0]); const int n = static_cast(weight.entry->dims[1]); const int k_splits = static_cast(weight.entry->flags); const std::vector cores = std::getenv("LING3_LINEAR_SINGLE_CORE") == nullptr ? std::vector {0, 1, 2} : std::vector {0}; ling3::DynamicW4Linear linear( {k, n, k_splits, cores}, TensorSpan(weight), TensorSpan(scale), TensorSpan(correction)); std::unique_ptr alternate_weights; if (!weight_base.empty() && weight_base != base) { const auto & alternate_weight = package.tensor(weight_base + ".weight"); const auto & alternate_scale = package.tensor(weight_base + ".scales"); const auto & alternate_correction = package.tensor(weight_base + ".correction"); if (static_cast(alternate_weight.entry->dims[0]) != k || static_cast(alternate_weight.entry->dims[1]) != n || static_cast(alternate_weight.entry->flags) != k_splits) { throw std::runtime_error("alternate weight tensor has an incompatible shape"); } alternate_weights = std::make_unique( ling3::W4LinearConfig {k, n, k_splits, cores}, TensorSpan(alternate_weight), TensorSpan(alternate_scale), TensorSpan(alternate_correction)); } auto & reference_linear = alternate_weights ? *alternate_weights : linear; std::vector input(rows * static_cast(k)); for (std::size_t row = 0; row < rows; ++row) { for (int column = 0; column < k; ++column) { input[row * k + column] = 0.7F * std::sin(static_cast(column + 7 * row) * 0.03125F) + 0.2F * std::cos(static_cast(3 * column + row) * 0.0078125F); } } std::vector batch_output(rows * static_cast(n)); std::vector sequential_output(rows * static_cast(n)); linear.RunBatchWithWeights(input, rows, reference_linear, batch_output); for (std::size_t row = 0; row < rows; ++row) { reference_linear.Run( std::span(input).subspan(row * k, k), std::span(sequential_output).subspan(row * n, n)); } double maximum_error = 0.0; double mean_error = 0.0; for (std::size_t index = 0; index < batch_output.size(); ++index) { const double error = std::abs( static_cast(batch_output[index]) - sequential_output[index]); maximum_error = std::max(maximum_error, error); mean_error += error; } mean_error /= static_cast(batch_output.size()); // Check shared quantized inputs with reordered/repeated rows, a smaller // tail, and a return to the original capacity in the same context. double indexed_maximum_error = 0.0; if (std::getenv("LING3_PREFILL_W4A4") == nullptr) { std::vector quantized(input.size()); std::vector input_scales(rows); std::vector indices(rows); for (std::size_t row = 0; row < rows; ++row) { input_scales[row] = ling3::QuantizeSymmetricInt8( std::span(input).subspan(row * k, k), std::span(quantized).subspan(row * k, k)).scale; indices[row] = rows - 1 - row / 2; } for (const std::size_t count : {rows, std::max(1, rows / 2), rows}) { linear.RunBatchQuantizedRows( quantized, input_scales, std::span(indices).first(count), reference_linear, std::span(batch_output).first(count * n)); for (std::size_t row = 0; row < count; ++row) { for (int column = 0; column < n; ++column) { const double actual = batch_output[row * n + column]; if (!std::isfinite(actual)) { throw std::runtime_error("indexed W4 batch produced a non-finite output"); } indexed_maximum_error = std::max(indexed_maximum_error, std::abs( actual - sequential_output[indices[row] * n + column])); } } } } ling3::W4RunTimings mean; const auto sequential_begin = std::chrono::steady_clock::now(); for (std::size_t iteration = 0; iteration < iterations; ++iteration) { for (std::size_t row = 0; row < rows; ++row) { reference_linear.Run( std::span(input).subspan(row * k, k), std::span(sequential_output).subspan( row * n, n)); } } const double sequential_ms = std::chrono::duration( std::chrono::steady_clock::now() - sequential_begin).count() / static_cast(iterations); for (std::size_t iteration = 0; iteration < iterations; ++iteration) { const auto sample = linear.RunBatchWithWeights(input, rows, reference_linear, batch_output); mean.quantize_pack_ms += sample.quantize_pack_ms; mean.input_sync_ms += sample.input_sync_ms; mean.npu_ms += sample.npu_ms; mean.gather_ms += sample.gather_ms; mean.total_ms += sample.total_ms; } const double divisor = static_cast(iterations); mean.quantize_pack_ms /= divisor; mean.input_sync_ms /= divisor; mean.npu_ms /= divisor; mean.gather_ms /= divisor; mean.total_ms /= divisor; std::cout << std::fixed << std::setprecision(6) << "tensor=" << base << '\n' << "weight_tensor=" << (weight_base.empty() ? base : weight_base) << '\n' << "rows=" << rows << '\n' << "cores=" << (cores.size() == 1 ? "0" : "0,1,2") << '\n' << "sequential_mean_ms=" << sequential_ms << '\n' << "batch_quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n' << "batch_input_sync_mean_ms=" << mean.input_sync_ms << '\n' << "batch_npu_mean_ms=" << mean.npu_ms << '\n' << "batch_gather_mean_ms=" << mean.gather_ms << '\n' << "batch_total_mean_ms=" << mean.total_ms << '\n' << "speedup=" << sequential_ms / mean.total_ms << '\n' << "sequential_mean_abs_error=" << mean_error << '\n' << "sequential_max_abs_error=" << maximum_error << '\n' << "indexed_max_abs_error=" << indexed_maximum_error << '\n'; return maximum_error <= 1.0e-5 && indexed_maximum_error <= 1.0e-5 ? 0 : 13; } std::array CompareVectors( std::span actual, std::span expected) { if (actual.size() != expected.size() || actual.empty()) { throw std::invalid_argument("comparison vectors have incompatible sizes"); } double dot = 0.0; double actual_norm = 0.0; double expected_norm = 0.0; double mean_error = 0.0; double maximum_error = 0.0; for (std::size_t index = 0; index < actual.size(); ++index) { const double a = actual[index]; const double b = expected[index]; const double error = std::abs(a - b); dot += a * b; actual_norm += a * a; expected_norm += b * b; mean_error += error; maximum_error = std::max(maximum_error, error); } return { dot / std::sqrt(actual_norm * expected_norm), mean_error / static_cast(actual.size()), maximum_error, }; } int RunGdnReferenceCheck(const std::string & path) { if (std::getenv("LING3_GDN_CPU_DECODE")) throw std::invalid_argument("unset LING3_GDN_CPU_DECODE for the NPU comparator"); ConfigureInferenceProcess(); const ling3::ModelPackage package(path); const auto h6 = TensorSpan(package.tensor("rknn.gdn.heads6")); const auto h5 = TensorSpan(package.tensor("rknn.gdn.heads5")); ling3::GdnStep npu(h6, h5), cpu(h6, h5); constexpr int heads = 16, dimension = 128, width = heads * dimension, steps = 256; std::vector q(width), k(width), v(width), d(width), b(heads), a(width), c(width); std::vector state(heads * dimension * dimension, 0.0); double npu_square_error = 0, cpu_square_error = 0, reference_square = 0; double npu_max = 0, cpu_max = 0; for (int t = 0; t < steps; ++t) { FillGdnToken(t, q, k, v, d, b); // Include near-unit retention as well as fast forgetting; a zero-state // first-token test alone cannot exercise accumulated recurrence error. for (int i = 0; i < width; ++i) d[i] = i % 3 == 0 ? -0.0001F : (i % 3 == 1 ? -0.05F : d[i]); npu.Run(q, k, v, d, b, a); cpu.RunBatchCpu(q, k, v, d, b, c); for (int head = 0; head < heads; ++head) { std::array factor; for (int j = 0; j < dimension; ++j) factor[j] = std::exp(double(d[head * dimension + j])); for (int col = 0; col < dimension; ++col) { auto * row = state.data() + (head * dimension + col) * dimension; double predicted = 0; for (int j = 0; j < dimension; ++j) { row[j] *= factor[j]; predicted += row[j] * double(k[head * dimension + j]); } const double delta = double(b[head]) * (double(v[head * dimension + col]) - predicted); double result = 0; for (int j = 0; j < dimension; ++j) { row[j] += delta * double(k[head * dimension + j]); result += row[j] * double(q[head * dimension + j]) / std::sqrt(128.0); } const int index = head * dimension + col; if (!std::isfinite(a[index]) || !std::isfinite(c[index])) throw std::runtime_error("non-finite recurrence"); const double ea = a[index] - result, ec = c[index] - result; npu_square_error += ea * ea; cpu_square_error += ec * ec; reference_square += result * result; npu_max = std::max(npu_max, std::abs(ea)); cpu_max = std::max(cpu_max, std::abs(ec)); } } } std::cout << std::setprecision(12) << "steps=" << steps << " npu_relative_rms=" << std::sqrt(npu_square_error / reference_square) << " cpu_relative_rms=" << std::sqrt(cpu_square_error / reference_square) << " npu_max=" << npu_max << " cpu_max=" << cpu_max << '\n'; return cpu_square_error < npu_square_error && cpu_max < 1e-6 ? 0 : 2; } int RunGdnCpuStateCheck(const std::string & path) { if (std::getenv("LING3_GDN_CPU_DECODE")) throw std::invalid_argument("unset LING3_GDN_CPU_DECODE to test transitions to NPU"); ConfigureInferenceProcess(); const ling3::ModelPackage package(path); const auto h6 = TensorSpan(package.tensor("rknn.gdn.heads6")); const auto h5 = TensorSpan(package.tensor("rknn.gdn.heads5")); ling3::GdnStep batch(h6, h5), chunks(h6, h5); constexpr std::size_t rows = 97, width = 16 * 128; std::vector q(rows * width), k(q.size()), v(q.size()), d(q.size()), b(rows * 16); std::vector one(q.size()), many(q.size()), replay(q.size()); for (std::size_t t = 0; t < rows; ++t) { FillGdnToken(t, std::span(q).subspan(t * width, width), std::span(k).subspan(t * width, width), std::span(v).subspan(t * width, width), std::span(d).subspan(t * width, width), std::span(b).subspan(t * 16, 16)); } batch.RunBatchCpu(q, k, v, d, b, one); std::size_t offset = 0; for (const std::size_t count : {1, 3, 16, 31, 46}) { chunks.RunBatchCpu(std::span(q).subspan(offset * width, count * width), std::span(k).subspan(offset * width, count * width), std::span(v).subspan(offset * width, count * width), std::span(d).subspan(offset * width, count * width), std::span(b).subspan(offset * 16, count * 16), std::span(many).subspan(offset * width, count * width)); offset += count; } const auto error = CompareVectors(many, one); std::vector tail1(width), tail2(width); batch.Run(std::span(q).last(width), std::span(k).last(width), std::span(v).last(width), std::span(d).last(width), std::span(b).last(16), tail1); chunks.Run(std::span(q).last(width), std::span(k).last(width), std::span(v).last(width), std::span(d).last(width), std::span(b).last(16), tail2); const auto transition = CompareVectors(tail1, tail2); // After a device step, return to CPU and verify the captured shadow state. batch.RunBatchCpu(q, k, v, d, b, replay); chunks.RunBatchCpu(q, k, v, d, b, many); const auto back = CompareVectors(replay, many); batch.Reset(); batch.RunBatchCpu(q, k, v, d, b, replay); const auto reset = CompareVectors(replay, one); std::cout << "chunk_max_error=" << error[2] << " cpu_to_npu_max_error=" << transition[2] << " npu_to_cpu_max_error=" << back[2] << " reset_max_error=" << reset[2] << '\n'; return error[2] == 0 && transition[2] == 0 && back[2] == 0 && reset[2] == 0 ? 0 : 2; } int RunPrefill32Smoke(const std::string & package_path) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("prefill32-smoke requires a build with RKNN support"); } ConfigureInferenceProcess(); const ling3::ModelPackage package(package_path); const auto tokenizer = PackageTokenizer(package); auto encoded = tokenizer.Encode(Utf8( u8"SYSTEMdetailed thinking off<|role_end|>" u8"HUMAN请用中文简要说明今天的工作安排,并列出三个重点事项。" u8"<|role_end|>ASSISTANT\n")); if (encoded.size() < 32) throw std::runtime_error("prefill32 test prompt is too short"); encoded.resize(32); ling3::Decoder decoder(package); if (!decoder.has_dynamic_batch()) { throw std::runtime_error("set LING3_GDN_PREFILL_DIR for prefill32-smoke"); } std::vector sequential_logits(package.header().vocab_size); std::vector sequential_tail(package.header().vocab_size); std::vector batch_logits(package.header().vocab_size); std::vector batch_tail(package.header().vocab_size); double sequential_ms = 0.0; for (const auto token : encoded) { sequential_ms += decoder.Eval(token, sequential_logits).total_ms; } const auto next_token = static_cast( std::max_element(sequential_logits.begin(), sequential_logits.end()) - sequential_logits.begin()); decoder.Eval(next_token, sequential_tail); decoder.Reset(); const auto cold_batch = decoder.EvalBatch32(encoded, batch_logits); decoder.Eval(next_token, batch_tail); const auto logits_error = CompareVectors(batch_logits, sequential_logits); const auto tail_error = CompareVectors(batch_tail, sequential_tail); decoder.Reset(); const auto warm_batch = decoder.EvalBatch32(encoded, batch_logits); std::cout << std::fixed << std::setprecision(8) << "tokens=32\n" << "sequential_ms=" << sequential_ms << '\n' << "cold_batch_ms=" << cold_batch.total_ms << '\n' << "warm_batch_ms=" << warm_batch.total_ms << '\n' << "speedup=" << sequential_ms / warm_batch.total_ms << '\n' << "logits_cosine=" << logits_error[0] << '\n' << "logits_mean_abs_error=" << logits_error[1] << '\n' << "logits_max_abs_error=" << logits_error[2] << '\n' << "tail_logits_cosine=" << tail_error[0] << '\n' << "tail_logits_mean_abs_error=" << tail_error[1] << '\n' << "tail_logits_max_abs_error=" << tail_error[2] << '\n' << "next_token=" << next_token << '\n'; return logits_error[0] >= 0.999 && tail_error[0] >= 0.999 ? 0 : 15; } std::vector> TopLogits( std::span logits, std::size_t count) { count = std::min(count, logits.size()); std::vector indices(logits.size()); std::iota(indices.begin(), indices.end(), 0U); std::partial_sort( indices.begin(), indices.begin() + static_cast(count), indices.end(), [&logits](std::uint32_t left, std::uint32_t right) { return logits[left] > logits[right]; }); std::vector> result; result.reserve(count); for (std::size_t index = 0; index < count; ++index) { result.emplace_back(indices[index], logits[indices[index]]); } return result; } int RunLogitsSmoke( const std::string & package_path, std::uint32_t token, const std::string * reference_path) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("logits-smoke requires a build with RKNN support"); } ConfigureInferenceProcess(); const ling3::ModelPackage package(package_path); const auto tokenizer = PackageTokenizer(package); const auto initialize_begin = std::chrono::steady_clock::now(); ling3::Decoder decoder(package); const auto initialize_end = std::chrono::steady_clock::now(); std::vector logits(package.header().vocab_size); const auto timings = decoder.Eval(token, logits); const auto top = TopLogits(logits, 10); std::cout << std::fixed << std::setprecision(6) << "token=" << token << '\n' << "initialization_ms=" << std::chrono::duration(initialize_end - initialize_begin).count() << '\n' << "layers_ms=" << timings.layers_ms << '\n' << "output_head_ms=" << timings.output_head_ms << '\n' << "total_ms=" << timings.total_ms << '\n'; for (std::size_t index = 0; index < top.size(); ++index) { const std::array id {top[index].first}; std::cout << "top" << index << "_id=" << top[index].first << " logit=" << top[index].second << " piece=" << std::quoted(tokenizer.Decode(id)) << '\n'; } if (reference_path == nullptr) return 0; const auto reference = ReadFloatFile(*reference_path, logits.size()); double dot = 0.0, norm = 0.0, reference_norm = 0.0; double mean_error = 0.0, maximum_error = 0.0; for (std::size_t index = 0; index < logits.size(); ++index) { const double error = std::abs(static_cast(logits[index]) - reference[index]); mean_error += error; maximum_error = std::max(maximum_error, error); dot += static_cast(logits[index]) * reference[index]; norm += static_cast(logits[index]) * logits[index]; reference_norm += static_cast(reference[index]) * reference[index]; } mean_error /= static_cast(logits.size()); const double cosine = dot / std::sqrt(norm * reference_norm); std::cout << "cosine_vs_reference=" << cosine << '\n' << "mean_abs_error=" << mean_error << '\n' << "max_abs_error=" << maximum_error << '\n'; return cosine >= 0.90 ? 0 : 13; } int RunGenerate( const std::string & package_path, std::string_view user_text, std::size_t maximum_new_tokens) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("generate requires a build with RKNN support"); } if (maximum_new_tokens == 0) throw std::invalid_argument("MAX_NEW_TOKENS must be positive"); ConfigureInferenceProcess(); const ling3::ModelPackage package(package_path); const auto tokenizer = PackageTokenizer(package); const std::string prompt = "SYSTEMdetailed thinking off<|role_end|>" "HUMAN" + std::string(user_text) + "<|role_end|>ASSISTANT\n"; const auto tokens = tokenizer.Encode(prompt); if (tokens.empty() || tokens.size() + maximum_new_tokens > package.header().max_context) { throw std::runtime_error("prompt and output exceed the package context capacity"); } const auto initialize_begin = std::chrono::steady_clock::now(); ling3::Decoder decoder(package); const auto initialize_end = std::chrono::steady_clock::now(); std::vector logits(package.header().vocab_size); const auto prefill_begin = std::chrono::steady_clock::now(); for (const auto token : tokens) decoder.Eval(token, logits); const auto prefill_end = std::chrono::steady_clock::now(); std::vector decode_ms; std::vector generated; decode_ms.reserve(maximum_new_tokens); generated.reserve(maximum_new_tokens); std::cout << "response=" << std::flush; for (std::size_t index = 0; index < maximum_new_tokens; ++index) { const auto found = std::max_element(logits.begin(), logits.end()); const auto token = static_cast(found - logits.begin()); if (token == package.header().eos_token) break; generated.push_back(token); std::cout << tokenizer.Piece(token) << std::flush; const auto timing = decoder.Eval(token, logits); decode_ms.push_back(timing.total_ms); std::cerr << "\ntoken=" << token << " position=" << decoder.position() << " layers_ms=" << timing.layers_ms << " head_ms=" << timing.output_head_ms << " total_ms=" << timing.total_ms << std::flush; } std::cout << '\n'; const double initialize_ms = std::chrono::duration( initialize_end - initialize_begin).count(); const double prefill_ms = std::chrono::duration( prefill_end - prefill_begin).count(); const double decode_total = std::accumulate(decode_ms.begin(), decode_ms.end(), 0.0); std::cerr << '\n' << std::fixed << std::setprecision(3) << "prompt_tokens=" << tokens.size() << '\n' << "generated_tokens=" << generated.size() << '\n' << "initialization_ms=" << initialize_ms << '\n' << "prefill_ms=" << prefill_ms << '\n' << "prefill_tokens_per_second=" << (prefill_ms == 0.0 ? 0.0 : 1000.0 * tokens.size() / prefill_ms) << '\n' << "decode_ms=" << decode_total << '\n' << "decode_tokens_per_second=" << (decode_total == 0.0 ? 0.0 : 1000.0 * decode_ms.size() / decode_total) << '\n'; return 0; } struct ProcessMemory { std::size_t rss_kb = 0; std::size_t hwm_kb = 0; }; ProcessMemory ReadProcessMemory() { #if defined(__linux__) std::ifstream status("/proc/self/status"); if (!status) throw std::runtime_error("cannot read /proc/self/status"); ProcessMemory memory; std::string line; while (std::getline(status, line)) { std::istringstream fields(line); std::string key; std::size_t value = 0; std::string unit; if (!(fields >> key >> value >> unit)) continue; if (key == "VmRSS:") memory.rss_kb = value; if (key == "VmHWM:") memory.hwm_kb = value; } return memory; #else return {}; #endif } std::vector BuildBenchmarkPrompt( const ling3::Tokenizer & tokenizer, std::size_t sequence_length) { const auto prefix = tokenizer.Encode( "SYSTEMdetailed thinking off<|role_end|>" "HUMAN"); const auto body = tokenizer.Encode(Utf8( u8"请用中文详细说明如何安排一天的工作,依次讨论目标、执行步骤、风险和复盘方法。" u8"回答需要完整、连贯并包含具体例子。")); const auto suffix = tokenizer.Encode( "<|role_end|>ASSISTANT\n"); if (body.empty() || prefix.size() + suffix.size() > sequence_length) { throw std::invalid_argument("SEQLEN is too short for the deterministic benchmark prompt"); } std::vector prompt; prompt.reserve(sequence_length); prompt.insert(prompt.end(), prefix.begin(), prefix.end()); std::size_t body_index = 0; while (prompt.size() + suffix.size() < sequence_length) { prompt.push_back(body[body_index++ % body.size()]); } prompt.insert(prompt.end(), suffix.begin(), suffix.end()); return prompt; } std::uint32_t GreedyToken(std::span logits) { return static_cast( std::max_element(logits.begin(), logits.end()) - logits.begin()); } std::uint64_t TokenHash(std::span tokens) { std::uint64_t hash = 1469598103934665603ULL; for (const auto token : tokens) { for (unsigned shift = 0; shift < 32; shift += 8) { hash ^= (token >> shift) & 0xffU; hash *= 1099511628211ULL; } } return hash; } // Reference tokens are teacher-forced so small numerical differences cannot // change the input sequence and invalidate comparisons of recurrent state. int RunSequenceCheck(const std::string & path, std::size_t length, std::size_t steps, const std::string & out, const std::string & reference, const std::string & corpus = "") { ConfigureInferenceProcess(); const ling3::ModelPackage package(path); if (!steps || length + steps > package.header().max_context) throw std::invalid_argument("invalid sequence-check lengths"); const auto tokenizer = PackageTokenizer(package); std::vector corpus_tokens; auto prompt = BuildBenchmarkPrompt(tokenizer, length); if (!corpus.empty()) { std::ifstream file(corpus); if (!file) throw std::runtime_error("cannot open evaluation corpus"); const std::string body{std::istreambuf_iterator(file), std::istreambuf_iterator()}; corpus_tokens = tokenizer.Encode(body); if (corpus_tokens.size() < length + steps) throw std::runtime_error("evaluation corpus is too short: " + std::to_string(corpus_tokens.size())); prompt.assign(corpus_tokens.begin(), corpus_tokens.begin() + length); std::ofstream teacher(out + ".teacher.ids"); for (std::size_t i = 0; i < steps; ++i) teacher << corpus_tokens[length + i] << '\n'; if (!teacher) throw std::runtime_error("cannot save evaluation teacher IDs"); } std::ofstream prompt_ids(out + ".prompt.ids"); for (const auto token : prompt) prompt_ids << token << '\n'; if (!prompt_ids) throw std::runtime_error("cannot save sequence-check prompt IDs"); ling3::Decoder decoder(package); std::vector logits(package.header().vocab_size), expected(logits.size()); std::ofstream values(out + ".f32", std::ios::binary), ids(out + ".ids"); std::ifstream ref_values, ref_ids; if (!reference.empty()) { ref_values.open(reference + ".f32", std::ios::binary); ref_ids.open(reference + ".ids"); if (!ref_values || !ref_ids) throw std::runtime_error("cannot open reference"); } if (!values || !ids) throw std::runtime_error("cannot open sequence-check output"); std::size_t offset = 0; while (offset < length) { const auto rows = std::min(128, length - offset); if (rows == 1) decoder.Eval(prompt[offset], logits); else { decoder.PrepareBatch(rows); decoder.EvalBatch(std::span(prompt).subspan(offset, rows), logits); } offset += rows; } double min_cosine = 1.0, max_error = 0.0; std::size_t agreement = 0; for (std::size_t step = 0; step < steps; ++step) { for (const auto x : logits) if (!std::isfinite(x)) throw std::runtime_error("non-finite logits"); const auto predicted = GreedyToken(logits); auto token = predicted; values.write(reinterpret_cast(logits.data()), logits.size() * sizeof(float)); ids << predicted << '\n'; if (!reference.empty()) { ref_values.read(reinterpret_cast(expected.data()), expected.size() * sizeof(float)); if (!ref_values || !(ref_ids >> token) || token >= logits.size()) throw std::runtime_error("invalid/truncated reference"); for (const auto x : expected) if (!std::isfinite(x)) throw std::runtime_error("non-finite reference logits"); const auto error = CompareVectors(logits, expected); min_cosine = std::min(min_cosine, error[0]); max_error = std::max(max_error, error[2]); agreement += token == predicted; std::cout << "step=" << step << " cosine=" << std::setprecision(10) << error[0] << " mae=" << error[1] << " max=" << error[2] << " top1=" << (token == predicted) << '\n'; } if (!corpus_tokens.empty()) token = corpus_tokens[length + step]; if (step + 1 < steps) decoder.Eval(token, logits); } if (!values || !ids) throw std::runtime_error("failed writing reference"); std::cout << "steps=" << steps << " min_cosine=" << min_cosine << " max_error=" << max_error << " top1_agreement=" << agreement << '\n'; const auto mla=decoder.AttentionStats(); std::cout << "mla_npu_calls=" << mla.npu_calls << " mla_cpu_calls=" << mla.cpu_calls << " mla_fallbacks=" << mla.fallbacks << " mla_npu_ms=" << mla.npu_ms << '\n'; return reference.empty() || min_cosine >= 0.999 ? 0 : 2; } struct BenchmarkSample { double ttft_ms = 0.0; double prefill_ms = 0.0; double decode_ms = 0.0; double tokens_per_second = 0.0; std::size_t eos_tokens = 0; std::uint64_t token_hash = 0; ProcessMemory memory; }; #if LING3_EXPERIMENTAL_MTP // Same resident decoder for baseline and MTP. Populate every MTP prefix slot // sequentially, and teacher-force baseline tokens so the work and histories // stay comparable. This measures forward overhead, not speculative speedup. int RunMtpBenchmark(const std::string & path, std::size_t length, std::size_t steps, std::size_t repeats, std::size_t capacity, const std::string & cases_path) { ConfigureInferenceProcess(); const ling3::ModelPackage package(path); if (!length || steps < 2 || !repeats || length + steps > capacity || capacity > 262144) throw std::invalid_argument("invalid MTP benchmark lengths"); const auto tokenizer = PackageTokenizer(package); std::vector> prompts{BuildBenchmarkPrompt(tokenizer, length)}; if (!cases_path.empty()) { std::ifstream file(cases_path); if (!file) throw std::runtime_error("cannot open MTP cases file"); std::string question; while (std::getline(file, question)) { if (question.empty()) continue; auto ids = tokenizer.Encode("SYSTEMdetailed thinking off<|role_end|>" "HUMAN" + question + "<|role_end|>ASSISTANT\n"); if (ids.empty() || ids.size() + steps > capacity) throw std::invalid_argument("case does not fit the requested context"); prompts.push_back(std::move(ids)); } } const auto started = std::chrono::steady_clock::now(); ling3::Decoder decoder(package, capacity); decoder.EnableMtp(); const std::vector modes = decoder.HasMtp() ? std::vector{false, true} : std::vector{false}; std::vector logits(package.header().vocab_size), draft(logits.size()); const auto milliseconds = [](auto a, auto b) { return std::chrono::duration(b - a).count(); }; std::cout << std::fixed << std::setprecision(4) << "benchmark=resident_mtp_forward_not_speculative\n" << "context_capacity=" << capacity << " cases=" << prompts.size() << " mtp_enabled=" << decoder.HasMtp() << " new_tokens=" << steps << " iterations=" << repeats << '\n' << "initialization_ms=" << milliseconds(started, std::chrono::steady_clock::now()) << std::endl; const auto initial_memory = ReadProcessMemory(); std::cout << "initial_rss_mib=" << initial_memory.rss_kb / 1024.0 << " initial_hwm_mib=" << initial_memory.hwm_kb / 1024.0 << std::endl; for (std::size_t case_index = 0; case_index < prompts.size(); ++case_index) { const auto & prompt = prompts[case_index]; length = prompt.size(); for (std::size_t round = 0; round <= repeats; ++round) { std::vector teacher; teacher.reserve(steps); for (bool with_mtp : modes) { decoder.Reset(); const auto attention_before = decoder.AttentionStats(); const auto begin = std::chrono::steady_clock::now(); double trunk = 0, mtp = 0, mtp_layers = 0, mtp_head = 0; for (std::size_t i = 0; i < length; ++i) { decoder.Eval(prompt[i], logits); if (with_mtp && i + 1 < length) decoder.EvalMtp(prompt[i + 1], draft); } std::uint32_t token = GreedyToken(logits); std::size_t target_agreement = 0, draft_hits = 0; if (!with_mtp) teacher.push_back(token); else { target_agreement += token == teacher[0]; token = teacher[0]; } const auto first = std::chrono::steady_clock::now(); std::vector trunk_samples, mtp_samples; for (std::size_t i = 1; i < steps; ++i) { std::uint32_t proposed = 0; if (with_mtp) { const auto t = decoder.EvalMtp(token, draft); mtp += t.total_ms; mtp_layers += t.layers_ms; mtp_head += t.output_head_ms; mtp_samples.push_back(t.total_ms); proposed = GreedyToken(draft); } const auto t = decoder.Eval(token, logits); trunk += t.total_ms; trunk_samples.push_back(t.total_ms); token = GreedyToken(logits); if (!with_mtp) teacher.push_back(token); else { target_agreement += token == teacher[i]; draft_hits += proposed == teacher[i]; token = teacher[i]; } } const auto end = std::chrono::steady_clock::now(); const auto percentile = [](std::vector values, double q) { std::sort(values.begin(), values.end()); return values[static_cast(q * (values.size() - 1))]; }; const auto memory = ReadProcessMemory(); const auto attention = decoder.AttentionStats(); std::cout << "phase=" << (round == 0 ? "warmup" : "measured") << " case=" << case_index << " seqlen=" << length << " round=" << round << " mode=" << (with_mtp ? "target_plus_mtp" : "target_only") << " sequential_prefill_ms=" << milliseconds(begin, first) << " decode_wall_ms=" << milliseconds(first, end) << " tokens_per_second=" << 1000.0 * (steps - 1) / milliseconds(first, end) << " target_mean_ms=" << trunk / (steps - 1) << " target_p50_ms=" << percentile(trunk_samples, 0.5) << " target_p95_ms=" << percentile(trunk_samples, 0.95); if (with_mtp) std::cout << " mtp_mean_ms=" << mtp / (steps - 1) << " mtp_p50_ms=" << percentile(mtp_samples, 0.5) << " mtp_p95_ms=" << percentile(mtp_samples, 0.95) << " mtp_layers_mean_ms=" << mtp_layers / (steps - 1) << " mtp_head_mean_ms=" << mtp_head / (steps - 1) << " target_agreement=" << target_agreement << '/' << steps << " draft_top1_hits=" << draft_hits << '/' << (steps - 1); std::cout << " mla_npu_calls=" << attention.npu_calls - attention_before.npu_calls << " mla_cpu_calls=" << attention.cpu_calls - attention_before.cpu_calls << " mla_fallbacks=" << attention.fallbacks - attention_before.fallbacks << " rss_mib=" << memory.rss_kb / 1024.0 << " hwm_mib=" << memory.hwm_kb / 1024.0 << " teacher_hash=" << std::hex << TokenHash(teacher) << std::dec << std::endl; if (with_mtp && target_agreement != steps) throw std::runtime_error("MTP probe changed target predictions for the same teacher-forced tokens"); } } } return 0; } #endif int RunBenchmark( const std::string & package_path, std::size_t sequence_length, std::size_t new_tokens, std::size_t iterations) { if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("benchmark requires a build with RKNN support"); } if (sequence_length == 0 || new_tokens < 2 || iterations == 0) { throw std::invalid_argument( "SEQLEN and ITERATIONS must be positive; NEW_TOKENS must be at least 2"); } ConfigureInferenceProcess(); const ling3::ModelPackage package(package_path); const auto tokenizer = PackageTokenizer(package); if (sequence_length + new_tokens > package.header().max_context) { throw std::invalid_argument("SEQLEN + NEW_TOKENS exceeds the package context capacity"); } const auto prompt = BuildBenchmarkPrompt(tokenizer, sequence_length); const auto initialization_begin = std::chrono::steady_clock::now(); ling3::Decoder decoder(package); const std::size_t batch_granularity = decoder.batch_granularity(); std::size_t prefill_batch = std::min(sequence_length, 128); if (batch_granularity != 0) prefill_batch -= prefill_batch % batch_granularity; if (prefill_batch >= 2) decoder.PrepareBatch(prefill_batch); const auto initialization_end = std::chrono::steady_clock::now(); const double initialization_ms = std::chrono::duration( initialization_end - initialization_begin).count(); std::vector logits(package.header().vocab_size); const auto run_once = [&]() { BenchmarkSample sample; decoder.Reset(); const auto request_begin = std::chrono::steady_clock::now(); std::size_t offset = 0; while (batch_granularity != 0 && sequence_length - offset >= 2) { std::size_t rows = std::min(sequence_length - offset, 128); rows -= rows % batch_granularity; if (rows < 2) break; const auto block = std::span(prompt).subspan(offset, rows); const bool final_block = sequence_length - offset == rows; if (final_block) { decoder.EvalBatch(block, logits); } else { decoder.EvalBatchState(block); } offset += rows; } for (; offset < sequence_length; ++offset) decoder.Eval(prompt[offset], logits); const auto first_token = GreedyToken(logits); const auto first_token_at = std::chrono::steady_clock::now(); sample.ttft_ms = std::chrono::duration( first_token_at - request_begin).count(); sample.prefill_ms = sample.ttft_ms; std::vector generated; generated.reserve(new_tokens); generated.push_back(first_token); if (first_token == package.header().eos_token) ++sample.eos_tokens; for (std::size_t index = 1; index < new_tokens; ++index) { decoder.Eval(generated.back(), logits); const auto token = GreedyToken(logits); generated.push_back(token); if (token == package.header().eos_token) ++sample.eos_tokens; } const auto final_token_at = std::chrono::steady_clock::now(); sample.decode_ms = std::chrono::duration( final_token_at - first_token_at).count(); sample.tokens_per_second = 1000.0 * static_cast(new_tokens - 1) / sample.decode_ms; sample.token_hash = TokenHash(generated); sample.memory = ReadProcessMemory(); return sample; }; const auto before_warmup = ReadProcessMemory(); const auto warmup = run_once(); std::vector samples; samples.reserve(iterations); for (std::size_t iteration = 0; iteration < iterations; ++iteration) { samples.push_back(run_once()); } const auto statistic = [&samples](auto member, bool minimum) { double result = samples.front().*member; for (const auto & sample : samples) { result = minimum ? std::min(result, sample.*member) : std::max(result, sample.*member); } return result; }; const auto mean = [&samples](auto member) { double total = 0.0; for (const auto & sample : samples) total += sample.*member; return total / static_cast(samples.size()); }; std::size_t maximum_rss_kb = 0; std::size_t maximum_hwm_kb = 0; for (const auto & sample : samples) { maximum_rss_kb = std::max(maximum_rss_kb, sample.memory.rss_kb); maximum_hwm_kb = std::max(maximum_hwm_kb, sample.memory.hwm_kb); } std::cout << std::fixed << std::setprecision(3) << "benchmark=official_style_autoregressive\n" << "model=Ling-3.0-tiny\n" << "model_total_parameters_b=7.9\n" << "model_active_parameters_b=1.3\n" << "dtype=W4A8+FP16/FP32_mixed\n" << "seqlen=" << sequence_length << '\n' << "new_tokens=" << new_tokens << '\n' << "iterations=" << iterations << '\n' << "cpu_cores=4,5,6,7\n" << "npu_cores=0,1,2\n" << "initialization_ms=" << initialization_ms << '\n' << "model_file_mb=" << package.mapped_bytes() / (1024.0 * 1024.0) << '\n' << "rss_before_warmup_mb=" << before_warmup.rss_kb / 1024.0 << '\n' << "warmup_ttft_ms=" << warmup.ttft_ms << '\n' << "warmup_tokens_per_second=" << warmup.tokens_per_second << '\n'; for (std::size_t index = 0; index < samples.size(); ++index) { const auto & sample = samples[index]; std::cout << "sample_" << index + 1 << "_ttft_ms=" << sample.ttft_ms << '\n' << "sample_" << index + 1 << "_tokens_per_second=" << sample.tokens_per_second << '\n' << "sample_" << index + 1 << "_rss_mb=" << sample.memory.rss_kb / 1024.0 << '\n' << "sample_" << index + 1 << "_hwm_mb=" << sample.memory.hwm_kb / 1024.0 << '\n' << "sample_" << index + 1 << "_eos_tokens=" << sample.eos_tokens << '\n' << "sample_" << index + 1 << "_token_hash=" << std::hex << sample.token_hash << std::dec << '\n'; } std::cout << "ttft_mean_ms=" << mean(&BenchmarkSample::ttft_ms) << '\n' << "ttft_min_ms=" << statistic(&BenchmarkSample::ttft_ms, true) << '\n' << "ttft_max_ms=" << statistic(&BenchmarkSample::ttft_ms, false) << '\n' << "tokens_per_second_mean=" << mean(&BenchmarkSample::tokens_per_second) << '\n' << "tokens_per_second_min=" << statistic(&BenchmarkSample::tokens_per_second, true) << '\n' << "tokens_per_second_max=" << statistic(&BenchmarkSample::tokens_per_second, false) << '\n' << "memory_rss_mb=" << maximum_rss_kb / 1024.0 << '\n' << "memory_hwm_mb=" << maximum_hwm_kb / 1024.0 << '\n'; return 0; } } // namespace int main(int argc, char ** argv) { try { if (argc < 2) { PrintUsage(argv[0]); return 1; } const std::string_view command(argv[1]); if (command == "inspect" && argc == 3) { const ling3::ModelPackage package(argv[2]); ling3::ValidateLing3Tiny(package.header()); std::cout << "valid Ling-3.0-tiny package\n" << "bytes=" << package.mapped_bytes() << '\n' << "tensors=" << package.tensors().size() << '\n' << "max_context=" << package.header().max_context << '\n' << "weight_bits=" << package.header().default_weight_bits << '\n' << "activation_bits=" << package.header().default_activation_bits << '\n'; return 0; } if (command == "plan" && argc <= 3) { const std::size_t max_context = argc == 3 ? ParseSize(argv[2], "context") : 4096; const ling3::ExecutionPlan plan = ling3::BuildExecutionPlan(max_context); std::cout << "layers=" << plan.layers.size() << '\n' << "decode_matmul_contexts=" << plan.decode_matmul_contexts << '\n' << "rknn_island_contexts=" << plan.rknn_island_contexts << '\n' << "kda_state_bytes_fp16=" << plan.kda_state_bytes_fp16 << '\n' << "mla_cache_bytes_fp16=" << plan.mla_cache_bytes_fp16 << '\n' << "activation_arena_bytes=" << plan.activation_arena_bytes << '\n'; return 0; } if (command == "rknn-smoke" && argc <= 3) { const std::size_t iterations = argc == 3 ? ParseSize(argv[2], "iterations") : 10; return ling3::RunRknnMatmulSmoke(iterations); } if (command == "tokenizer-smoke" && argc == 3) { return RunTokenizerSmoke(argv[2]); } if (command == "gdn-smoke" && (argc == 3 || argc == 4)) { const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 50; return RunGdnSmoke(argv[2], iterations); } if (command == "gdn-batch16-smoke" && (argc == 3 || argc == 4)) { const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 10; return RunGdnBatch16Smoke(argv[2], iterations); } if (command == "layer0-smoke" && (argc == 5 || argc == 6)) { const auto token = ParseSize(argv[3], "token"); if (token > std::numeric_limits::max()) { throw std::runtime_error("token is out of range"); } const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 20; return RunLayer0Smoke( argv[2], static_cast(token), argv[4], iterations); } if (command == "linear-smoke" && (argc == 4 || argc == 5)) { const std::size_t iterations = argc == 5 ? ParseSize(argv[4], "iterations") : 20; return RunLinearSmoke(argv[2], argv[3], iterations); } if (command == "linear-batch-smoke" && argc >= 5 && argc <= 7) { const std::size_t rows = ParseSize(argv[4], "rows"); const std::size_t iterations = argc >= 6 ? ParseSize(argv[5], "iterations") : 20; const std::string weight_base = argc == 7 ? argv[6] : argv[3]; return RunLinearBatchSmoke(argv[2], argv[3], rows, iterations, weight_base); } if (command == "prefill32-smoke" && argc == 3) { return RunPrefill32Smoke(argv[2]); } if (command == "gdn-cpu-state-check" && argc == 3) { return RunGdnCpuStateCheck(argv[2]); } if (command == "gdn-reference-check" && argc == 3) { return RunGdnReferenceCheck(argv[2]); } if (command == "logits-smoke" && (argc == 4 || argc == 5)) { const auto token = ParseSize(argv[3], "token"); if (token > std::numeric_limits::max()) { throw std::runtime_error("token is out of range"); } const std::string reference = argc == 5 ? argv[4] : ""; return RunLogitsSmoke( argv[2], static_cast(token), argc == 5 ? &reference : nullptr); } if (command == "generate" && (argc == 4 || argc == 5)) { const std::size_t maximum_new_tokens = argc == 5 ? ParseSize(argv[4], "maximum new tokens") : 64; return RunGenerate(argv[2], argv[3], maximum_new_tokens); } if (command == "benchmark" && (argc == 5 || argc == 6)) { const std::size_t sequence_length = ParseSize(argv[3], "sequence length"); const std::size_t new_tokens = ParseSize(argv[4], "new tokens"); const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 3; return RunBenchmark(argv[2], sequence_length, new_tokens, iterations); } #if LING3_EXPERIMENTAL_MTP if (command == "mtp-benchmark" && (argc == 7 || argc == 8)) { return RunMtpBenchmark(argv[2], ParseSize(argv[3], "sequence length"), ParseSize(argv[4], "new tokens"), ParseSize(argv[5], "iterations"), ParseSize(argv[6], "context capacity"), argc == 8 ? argv[7] : ""); } #endif if (command == "sequence-check" && argc >= 6 && argc <= 8) { return RunSequenceCheck(argv[2], ParseSize(argv[3], "sequence length"), ParseSize(argv[4], "new tokens"), argv[5], argc >= 7 ? argv[6] : "", argc == 8 ? argv[7] : ""); } if (command == "serve" && (argc == 3 || argc == 4)) { const std::size_t port = argc == 4 ? ParseSize(argv[3], "port") : 9091; if (port == 0 || port > std::numeric_limits::max()) { throw std::runtime_error("port must be in [1, 65535]"); } if (!ling3::RknnBackendAvailable()) { throw std::runtime_error("serve requires a build with RKNN support"); } ConfigureInferenceProcess(); return ling3::RunHttpService(argv[2], static_cast(port)); } PrintUsage(argv[0]); return 1; } catch (const std::exception & error) { std::cerr << "error: " << error.what() << '\n'; return 1; } }