Ling-3.0-tiny-RKNN / src /main.cpp
Sariel00's picture
Keep MTP opt-in: default-off build, isolated experimental package and measurements
3706f1c verified
Raw History Blame Contribute Delete
78.6 kB
#include "ling3/decoder.h"
#include "ling3/execution_plan.h"
#include "ling3/gdn_step.h"
#include "ling3/layer0.h"
#include "ling3/model_package.h"
#include "ling3/quantization.h"
#include "ling3/rknn_backend.h"
#include "ling3/service.h"
#include "ling3/tokenizer.h"
#include "ling3/w4_linear.h"
#include <algorithm>
#include <charconv>
#include <chrono>
#include <cmath>
#include <cstddef>
#include <cstdint>
#include <cstdlib>
#include <exception>
#include <iomanip>
#include <fstream>
#include <iostream>
#include <limits>
#include <numeric>
#include <span>
#include <string>
#include <string_view>
#include <utility>
#include <vector>
#if defined(__linux__)
#include <cerrno>
#include <cstring>
#include <sched.h>
#include <sys/resource.h>
#endif
namespace {
std::size_t ParseSize(std::string_view text, const char * label) {
std::size_t value = 0;
const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value);
if (error != std::errc {} || end != text.data() + text.size()) {
throw std::runtime_error(std::string("invalid ") + label + ": " + std::string(text));
}
return value;
}
void PrintUsage(const char * program) {
std::cerr << "Usage:\n"
<< " " << program << " inspect MODEL.l3r\n"
<< " " << program << " plan [MAX_CONTEXT]\n"
<< " " << program << " rknn-smoke [ITERATIONS]\n"
<< " " << program << " tokenizer-smoke MODEL.l3r\n"
<< " " << program << " gdn-smoke MODEL.l3r [ITERATIONS]\n"
<< " " << program << " gdn-batch16-smoke MODEL.l3r [ITERATIONS]\n"
<< " " << program << " gdn-cpu-state-check MODEL.l3r\n"
<< " " << program << " gdn-reference-check MODEL.l3r\n"
<< " " << program << " layer0-smoke MODEL.l3r TOKEN REF.f32 [ITERATIONS]\n"
<< " " << program << " linear-smoke MODEL.l3r TENSOR_BASE [ITERATIONS]\n"
<< " " << program
<< " linear-batch-smoke MODEL.l3r TENSOR_BASE ROWS [ITERATIONS] [WEIGHT_BASE]\n"
<< " " << program << " prefill32-smoke MODEL.l3r\n"
<< " " << program << " logits-smoke MODEL.l3r TOKEN [REF.f32]\n"
<< " " << program << " generate MODEL.l3r USER_TEXT [MAX_NEW_TOKENS]\n"
<< " " << program << " benchmark MODEL.l3r SEQLEN NEW_TOKENS [ITERATIONS]\n"
#if LING3_EXPERIMENTAL_MTP
<< " " << program << " mtp-benchmark MODEL.l3r SEQLEN NEW_TOKENS ITERATIONS CONTEXT [CASES.txt]\n"
#endif
<< " " << program << " sequence-check MODEL.l3r SEQLEN NEW_TOKENS OUT_PREFIX [REFERENCE_PREFIX [CORPUS.txt]]\n"
<< " " << program << " serve MODEL.l3r [PORT]\n";
}
template <typename T>
std::span<const T> TensorSpan(const ling3::TensorView & tensor);
void PinToA76Cores() {
#if defined(__linux__) && defined(__aarch64__)
cpu_set_t cores;
CPU_ZERO(&cores);
for (int core = 4; core <= 7; ++core) CPU_SET(core, &cores);
if (sched_setaffinity(0, sizeof(cores), &cores) != 0) {
std::cerr << "warning: cannot pin inference threads to CPU cores 4-7: "
<< std::strerror(errno) << '\n';
}
#endif
}
void RaiseFileDescriptorLimit() {
#if defined(__linux__)
rlimit limit {};
if (getrlimit(RLIMIT_NOFILE, &limit) != 0) {
throw std::runtime_error(
std::string("getrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno));
}
constexpr rlim_t requested = 262144;
const rlim_t target = std::min(requested, limit.rlim_max);
if (limit.rlim_cur >= target) return;
limit.rlim_cur = target;
if (setrlimit(RLIMIT_NOFILE, &limit) != 0) {
throw std::runtime_error(
std::string("setrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno));
}
#endif
}
void ConfigureInferenceProcess() {
RaiseFileDescriptorLimit();
PinToA76Cores();
}
ling3::Tokenizer PackageTokenizer(const ling3::ModelPackage & package) {
const auto & asset = package.tensor("tokenizer");
if (asset.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kTokenizer)) {
throw std::runtime_error("tokenizer tensor has the wrong role");
}
return ling3::Tokenizer(TensorSpan<std::byte>(asset));
}
std::string Utf8(std::u8string_view text) {
return {reinterpret_cast<const char *>(text.data()), text.size()};
}
int RunTokenizerSmoke(const std::string & package_path) {
const ling3::ModelPackage package(package_path);
const auto & asset = package.tensor("tokenizer");
if (asset.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kTokenizer)) {
throw std::runtime_error("tokenizer tensor has the wrong role");
}
const ling3::Tokenizer tokenizer(TensorSpan<std::byte>(asset));
struct Case {
std::string text;
std::vector<std::uint32_t> expected;
};
const std::vector<Case> cases = {
{Utf8(u8"你好,世界!"), {34355, 44291, 859}},
{"Hello world!", {14455, 1931, 0}},
{"I'm testing 123.", {40, 3180, 7750, 220, 16, 17, 18, 13}},
{Utf8(u8"ABC e\u0301\n第二行"), {83140, 270, 95, 93086, 18878, 198, 2378, 685}},
{Utf8(u8"<think>想一下</think>"), {156903, 122843, 156904}},
{
Utf8(u8"<|startoftext|>user\n你好<|role_end|>\n<|startoftext|>assistant\n"),
{156891, 3840, 198, 34355, 156895, 198, 156891, 598, 10450, 198},
},
};
for (std::size_t index = 0; index < cases.size(); ++index) {
const auto actual = tokenizer.Encode(cases[index].text);
if (actual != cases[index].expected) {
std::cerr << "tokenizer mismatch in case " << index << "\nexpected:";
for (auto token : cases[index].expected) std::cerr << ' ' << token;
std::cerr << "\nactual:";
for (auto token : actual) std::cerr << ' ' << token;
std::cerr << '\n';
return 8;
}
if (tokenizer.Decode(actual) != cases[index].text && index != 3) {
std::cerr << "tokenizer round-trip mismatch in case " << index << '\n';
return 9;
}
}
if (!tokenizer.uses_official_unicode_rules()) {
std::cerr << "token IDs matched, but this build lacks official ICU Unicode rules\n";
return 10;
}
std::cout << "tokenizer smoke passed (" << cases.size()
<< " official vectors, ICU rules enabled)\n";
return 0;
}
int RunGdnSmoke(const std::string & package_path, std::size_t iterations) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("gdn-smoke requires a build with RKNN support");
}
if (iterations == 0) iterations = 1;
const ling3::ModelPackage package(package_path);
const auto & heads6 = package.tensor("rknn.gdn.heads6");
const auto & heads5 = package.tensor("rknn.gdn.heads5");
if (heads6.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kRknnIsland) ||
heads5.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kRknnIsland)) {
throw std::runtime_error("GDN tensors do not contain RKNN islands");
}
constexpr int heads = 16;
constexpr int dimension = 128;
constexpr int elements = heads * dimension;
std::vector<float> query(elements), key(elements), value(elements), decay(elements);
std::vector<float> beta(heads), output(elements), reference(elements);
for (int head = 0; head < heads; ++head) {
float q_norm = 1.0e-6F;
float k_norm = 1.0e-6F;
for (int index = 0; index < dimension; ++index) {
const int offset = head * dimension + index;
query[offset] = std::sin(static_cast<float>(offset + 1) * 0.013F);
key[offset] = std::cos(static_cast<float>(offset + 3) * 0.017F);
value[offset] = 0.2F * std::sin(static_cast<float>(offset + 5) * 0.021F);
decay[offset] = -2.0F - 0.5F * std::cos(static_cast<float>(offset) * 0.009F);
q_norm += query[offset] * query[offset];
k_norm += key[offset] * key[offset];
}
q_norm = std::sqrt(q_norm);
k_norm = std::sqrt(k_norm);
for (int index = 0; index < dimension; ++index) {
query[head * dimension + index] /= q_norm;
key[head * dimension + index] /= k_norm;
}
beta[head] = 1.0F / (1.0F + std::exp(-0.2F * std::sin(static_cast<float>(head))));
}
const auto initialize_begin = std::chrono::steady_clock::now();
ling3::GdnStep gdn(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5));
const auto initialize_end = std::chrono::steady_clock::now();
const double initialize_ms = std::chrono::duration<double, std::milli>(
initialize_end - initialize_begin).count();
gdn.Run(query, key, value, decay, beta, output);
const float query_scale = 1.0F / std::sqrt(static_cast<float>(dimension));
for (int head = 0; head < heads; ++head) {
float key_dot_query = 0.0F;
for (int index = 0; index < dimension; ++index) {
key_dot_query += key[head * dimension + index] *
query[head * dimension + index] * query_scale;
}
for (int index = 0; index < dimension; ++index) {
reference[head * dimension + index] =
beta[head] * value[head * dimension + index] * key_dot_query;
}
}
double dot = 0.0;
double norm_output = 0.0;
double norm_reference = 0.0;
double max_error = 0.0;
double mean_error = 0.0;
for (int index = 0; index < elements; ++index) {
const double error = std::abs(static_cast<double>(output[index]) - reference[index]);
max_error = std::max(max_error, error);
mean_error += error;
dot += static_cast<double>(output[index]) * reference[index];
norm_output += static_cast<double>(output[index]) * output[index];
norm_reference += static_cast<double>(reference[index]) * reference[index];
}
mean_error /= elements;
const double cosine = dot / std::sqrt(norm_output * norm_reference);
gdn.Reset();
for (int index = 0; index < 5; ++index) {
gdn.Run(query, key, value, decay, beta, output);
}
ling3::GdnRunTimings mean;
for (std::size_t index = 0; index < iterations; ++index) {
const auto sample = gdn.Run(query, key, value, decay, beta, output);
mean.stage_ms += sample.stage_ms;
mean.npu_ms += sample.npu_ms;
mean.collect_ms += sample.collect_ms;
mean.total_ms += sample.total_ms;
}
const double divisor = static_cast<double>(iterations);
mean.stage_ms /= divisor;
mean.npu_ms /= divisor;
mean.collect_ms /= divisor;
mean.total_ms /= divisor;
std::cout << std::fixed << std::setprecision(6)
<< "heads=6+5+5\n"
<< "cores=0,1,2\n"
<< "initialization_ms=" << initialize_ms << '\n'
<< "state_bytes=" << gdn.state_bytes() << '\n'
<< "first_step_cosine=" << cosine << '\n'
<< "first_step_mean_abs_error=" << mean_error << '\n'
<< "first_step_max_abs_error=" << max_error << '\n'
<< "iterations=" << iterations << '\n'
<< "stage_mean_ms=" << mean.stage_ms << '\n'
<< "npu_mean_ms=" << mean.npu_ms << '\n'
<< "collect_mean_ms=" << mean.collect_ms << '\n'
<< "total_mean_ms=" << mean.total_ms << '\n';
return cosine >= 0.999 && max_error <= 0.005 ? 0 : 11;
}
void FillGdnToken(
int token,
std::span<float> query,
std::span<float> key,
std::span<float> value,
std::span<float> decay,
std::span<float> beta) {
constexpr int heads = 16;
constexpr int dimension = 128;
for (int head = 0; head < heads; ++head) {
float q_norm = 1.0e-6F;
float k_norm = 1.0e-6F;
for (int index = 0; index < dimension; ++index) {
const int offset = head * dimension + index;
query[offset] = std::sin(static_cast<float>(offset + 13 * token + 1) * 0.013F);
key[offset] = std::cos(static_cast<float>(offset + 7 * token + 3) * 0.017F);
value[offset] = 0.2F *
std::sin(static_cast<float>(offset + 5 * token + 5) * 0.021F);
decay[offset] = -2.0F - 0.5F *
std::cos(static_cast<float>(offset + 3 * token) * 0.009F);
q_norm += query[offset] * query[offset];
k_norm += key[offset] * key[offset];
}
q_norm = std::sqrt(q_norm);
k_norm = std::sqrt(k_norm);
for (int index = 0; index < dimension; ++index) {
query[head * dimension + index] /= q_norm;
key[head * dimension + index] /= k_norm;
}
beta[head] = 1.0F /
(1.0F + std::exp(-0.2F * std::sin(static_cast<float>(head + token))));
}
}
int RunGdnBatch16Smoke(const std::string & package_path, std::size_t iterations) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("gdn-batch16-smoke requires a build with RKNN support");
}
if (iterations == 0) iterations = 1;
const ling3::ModelPackage package(package_path);
const auto & heads6 = package.tensor("rknn.gdn.heads6");
const auto & heads5 = package.tensor("rknn.gdn.heads5");
constexpr std::size_t width = 16 * 128;
constexpr std::size_t rows = 16;
ling3::GdnStep sequential(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5));
ling3::GdnStep batch(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5));
if (!batch.has_batch16()) {
throw std::runtime_error("set LING3_GDN_PREFILL_DIR to the 16-token RKNN models");
}
std::vector<float> query(rows * width), key(rows * width), value(rows * width);
std::vector<float> decay(rows * width), beta(rows * 16);
std::vector<float> sequential_output(rows * width), batch_output(rows * width);
std::vector<float> scratch(width), scratch_key(width), scratch_value(width), scratch_decay(width);
std::vector<float> scratch_beta(16), sequential_tail(width), batch_tail(width);
for (int token = 0; token < 3; ++token) {
FillGdnToken(token, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta);
sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail);
batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail);
}
for (std::size_t row = 0; row < rows; ++row) {
FillGdnToken(
static_cast<int>(row + 3),
std::span<float>(query).subspan(row * width, width),
std::span<float>(key).subspan(row * width, width),
std::span<float>(value).subspan(row * width, width),
std::span<float>(decay).subspan(row * width, width),
std::span<float>(beta).subspan(row * 16, 16));
sequential.Run(
std::span<const float>(query).subspan(row * width, width),
std::span<const float>(key).subspan(row * width, width),
std::span<const float>(value).subspan(row * width, width),
std::span<const float>(decay).subspan(row * width, width),
std::span<const float>(beta).subspan(row * 16, 16),
std::span<float>(sequential_output).subspan(row * width, width));
}
const auto first_timing = batch.RunBatch16(
query, key, value, decay, beta, batch_output);
FillGdnToken(19, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta);
sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail);
batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail);
auto compare = [](std::span<const float> actual, std::span<const float> expected) {
std::array<double, 3> result {};
double actual_norm = 0.0;
double expected_norm = 0.0;
for (std::size_t index = 0; index < actual.size(); ++index) {
const double a = actual[index];
const double b = expected[index];
result[0] += a * b;
actual_norm += a * a;
expected_norm += b * b;
const double error = std::abs(a - b);
result[1] += error;
result[2] = std::max(result[2], error);
}
result[0] /= std::sqrt(actual_norm * expected_norm);
result[1] /= static_cast<double>(actual.size());
return result;
};
const auto output_error = compare(batch_output, sequential_output);
const auto state_error = compare(batch_tail, sequential_tail);
ling3::GdnRunTimings mean = first_timing;
for (std::size_t iteration = 1; iteration < iterations; ++iteration) {
const auto sample = batch.RunBatch16(query, key, value, decay, beta, batch_output);
mean.stage_ms += sample.stage_ms;
mean.npu_ms += sample.npu_ms;
mean.collect_ms += sample.collect_ms;
mean.total_ms += sample.total_ms;
}
const double divisor = static_cast<double>(iterations);
mean.stage_ms /= divisor;
mean.npu_ms /= divisor;
mean.collect_ms /= divisor;
mean.total_ms /= divisor;
std::cout << std::fixed << std::setprecision(8)
<< "rows=16\n"
<< "history_tokens=3\n"
<< "output_cosine=" << output_error[0] << '\n'
<< "output_mean_abs_error=" << output_error[1] << '\n'
<< "output_max_abs_error=" << output_error[2] << '\n'
<< "tail_state_cosine=" << state_error[0] << '\n'
<< "tail_mean_abs_error=" << state_error[1] << '\n'
<< "tail_max_abs_error=" << state_error[2] << '\n'
<< "stage_mean_ms=" << mean.stage_ms << '\n'
<< "npu_mean_ms=" << mean.npu_ms << '\n'
<< "collect_mean_ms=" << mean.collect_ms << '\n'
<< "total_mean_ms=" << mean.total_ms << '\n';
return output_error[0] >= 0.99999 && state_error[0] >= 0.99999 ? 0 : 14;
}
std::vector<float> ReadFloatFile(const std::string & path, std::size_t count) {
std::ifstream stream(path, std::ios::binary | std::ios::ate);
if (!stream) throw std::runtime_error("cannot open " + path);
const auto bytes = static_cast<std::size_t>(stream.tellg());
if (bytes != count * sizeof(float)) {
throw std::runtime_error(path + " has an incompatible byte count");
}
stream.seekg(0);
std::vector<float> result(count);
stream.read(reinterpret_cast<char *>(result.data()), static_cast<std::streamsize>(bytes));
if (!stream) throw std::runtime_error("cannot read " + path);
return result;
}
int RunLayer0Smoke(
const std::string & package_path,
std::uint32_t token,
const std::string & reference_path,
std::size_t iterations) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("layer0-smoke requires a build with RKNN support");
}
if (iterations == 0) iterations = 1;
const ling3::ModelPackage package(package_path);
const auto initialization_begin = std::chrono::steady_clock::now();
ling3::Layer0 layer(package);
const auto initialization_end = std::chrono::steady_clock::now();
const double initialization_ms = std::chrono::duration<double, std::milli>(
initialization_end - initialization_begin).count();
std::vector<float> output(1536);
layer.Reset();
layer.DecodeToken(token, output);
const auto reference = ReadFloatFile(reference_path, output.size());
double dot = 0.0, output_norm = 0.0, reference_norm = 0.0;
double mean_error = 0.0, maximum_error = 0.0;
for (std::size_t index = 0; index < output.size(); ++index) {
const double error = std::abs(static_cast<double>(output[index]) - reference[index]);
mean_error += error;
maximum_error = std::max(maximum_error, error);
dot += static_cast<double>(output[index]) * reference[index];
output_norm += static_cast<double>(output[index]) * output[index];
reference_norm += static_cast<double>(reference[index]) * reference[index];
}
mean_error /= output.size();
const double cosine = dot / std::sqrt(output_norm * reference_norm);
layer.Reset();
for (int index = 0; index < 3; ++index) layer.DecodeToken(token, output);
ling3::Layer0Timings mean;
for (std::size_t index = 0; index < iterations; ++index) {
const auto sample = layer.DecodeToken(token, output);
mean.attention_ms += sample.attention_ms;
mean.dense_ffn_ms += sample.dense_ffn_ms;
mean.total_ms += sample.total_ms;
}
const double divisor = static_cast<double>(iterations);
mean.attention_ms /= divisor;
mean.dense_ffn_ms /= divisor;
mean.total_ms /= divisor;
std::cout << std::fixed << std::setprecision(6)
<< "token=" << token << '\n'
<< "initialization_ms=" << initialization_ms << '\n'
<< "cosine_vs_official_bf16=" << cosine << '\n'
<< "mean_abs_error=" << mean_error << '\n'
<< "max_abs_error=" << maximum_error << '\n'
<< "iterations=" << iterations << '\n'
<< "attention_mean_ms=" << mean.attention_ms << '\n'
<< "dense_ffn_mean_ms=" << mean.dense_ffn_ms << '\n'
<< "total_mean_ms=" << mean.total_ms << '\n';
return cosine >= 0.98 ? 0 : 12;
}
template <typename T>
std::span<const T> TensorSpan(const ling3::TensorView & tensor) {
if (tensor.entry->data_bytes % sizeof(T) != 0) {
throw std::runtime_error(std::string(tensor.name) + " has an invalid byte count");
}
return {
reinterpret_cast<const T *>(tensor.data),
static_cast<std::size_t>(tensor.entry->data_bytes / sizeof(T)),
};
}
int RunLinearSmoke(
const std::string & package_path,
const std::string & base,
std::size_t iterations) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("linear-smoke requires a build with RKNN support");
}
if (iterations == 0) iterations = 1;
const ling3::ModelPackage package(package_path);
ling3::ValidateLing3Tiny(package.header());
const auto & weight = package.tensor(base + ".weight");
const auto & scale = package.tensor(base + ".scales");
const auto & correction = package.tensor(base + ".correction");
if (weight.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kInt4Low) ||
weight.entry->layout != static_cast<std::uint32_t>(ling3::TensorLayout::kPackedInt4Low) ||
weight.entry->rank != 2 || scale.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kFloat32) ||
correction.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kInt32)) {
throw std::runtime_error("linear-smoke tensors have incompatible metadata");
}
const int k = static_cast<int>(weight.entry->dims[0]);
const int n = static_cast<int>(weight.entry->dims[1]);
const int k_splits = static_cast<int>(weight.entry->flags);
const auto weights = TensorSpan<std::byte>(weight);
const auto scales = TensorSpan<float>(scale);
const auto corrections = TensorSpan<std::int32_t>(correction);
std::vector<float> input(k);
for (int index = 0; index < k; ++index) {
input[index] = 0.7F * std::sin(static_cast<float>(index) * 0.03125F) +
0.2F * std::cos(static_cast<float>(index) * 0.0078125F);
}
std::vector<float> output(n);
const auto initialize_begin = std::chrono::steady_clock::now();
ling3::DynamicW4Linear linear(
{k, n, k_splits, {0, 1, 2}}, weights, scales, corrections);
const auto initialize_end = std::chrono::steady_clock::now();
const auto initialize_ms = std::chrono::duration<double, std::milli>(
initialize_end - initialize_begin).count();
linear.Run(input, output);
std::vector<ling3::W4RunTimings> samples;
samples.reserve(iterations);
for (std::size_t iteration = 0; iteration < iterations; ++iteration) {
samples.push_back(linear.Run(input, output));
}
std::vector<std::int8_t> input_codes(k);
const auto quantization = ling3::QuantizeSymmetricInt8(input, input_codes);
std::vector<std::int32_t> reference_accumulator(n);
ling3::ReferenceW4Linear(input_codes, weights, n, reference_accumulator);
std::vector<float> reference(n);
ling3::DequantizePerChannel(reference_accumulator, quantization.scale, scales, reference);
double maximum_error = 0.0;
double mean_error = 0.0;
for (int index = 0; index < n; ++index) {
const double error = std::abs(static_cast<double>(output[index]) - reference[index]);
maximum_error = std::max(maximum_error, error);
mean_error += error;
}
mean_error /= n;
ling3::W4RunTimings mean;
for (const auto & sample : samples) {
mean.quantize_pack_ms += sample.quantize_pack_ms;
mean.input_sync_ms += sample.input_sync_ms;
mean.npu_ms += sample.npu_ms;
mean.gather_ms += sample.gather_ms;
mean.total_ms += sample.total_ms;
}
const double divisor = static_cast<double>(samples.size());
mean.quantize_pack_ms /= divisor;
mean.input_sync_ms /= divisor;
mean.npu_ms /= divisor;
mean.gather_ms /= divisor;
mean.total_ms /= divisor;
std::cout << std::fixed << std::setprecision(6)
<< "tensor=" << base << '\n'
<< "k=" << k << '\n'
<< "n=" << n << '\n'
<< "k_splits=" << k_splits << '\n'
<< "cores=0,1,2\n"
<< "initialization_ms=" << initialize_ms << '\n'
<< "resident_weight_bytes=" << linear.resident_weight_bytes() << '\n'
<< "iterations=" << iterations << '\n'
<< "quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n'
<< "input_sync_mean_ms=" << mean.input_sync_ms << '\n'
<< "npu_mean_ms=" << mean.npu_ms << '\n'
<< "gather_mean_ms=" << mean.gather_ms << '\n'
<< "total_mean_ms=" << mean.total_ms << '\n'
<< "reference_mean_abs_error=" << mean_error << '\n'
<< "reference_max_abs_error=" << maximum_error << '\n';
return maximum_error <= 1.0e-5 ? 0 : 7;
}
int RunLinearBatchSmoke(
const std::string & package_path,
const std::string & base,
std::size_t rows,
std::size_t iterations,
const std::string & weight_base) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("linear-batch-smoke requires a build with RKNN support");
}
if (rows < 1 || rows > 128) throw std::runtime_error("rows must be in [1, 128]");
if (iterations == 0) iterations = 1;
const ling3::ModelPackage package(package_path);
ling3::ValidateLing3Tiny(package.header());
const auto & weight = package.tensor(base + ".weight");
const auto & scale = package.tensor(base + ".scales");
const auto & correction = package.tensor(base + ".correction");
const int k = static_cast<int>(weight.entry->dims[0]);
const int n = static_cast<int>(weight.entry->dims[1]);
const int k_splits = static_cast<int>(weight.entry->flags);
const std::vector<int> cores = std::getenv("LING3_LINEAR_SINGLE_CORE") == nullptr
? std::vector<int> {0, 1, 2}
: std::vector<int> {0};
ling3::DynamicW4Linear linear(
{k, n, k_splits, cores}, TensorSpan<std::byte>(weight),
TensorSpan<float>(scale), TensorSpan<std::int32_t>(correction));
std::unique_ptr<ling3::DynamicW4Linear> alternate_weights;
if (!weight_base.empty() && weight_base != base) {
const auto & alternate_weight = package.tensor(weight_base + ".weight");
const auto & alternate_scale = package.tensor(weight_base + ".scales");
const auto & alternate_correction = package.tensor(weight_base + ".correction");
if (static_cast<int>(alternate_weight.entry->dims[0]) != k ||
static_cast<int>(alternate_weight.entry->dims[1]) != n ||
static_cast<int>(alternate_weight.entry->flags) != k_splits) {
throw std::runtime_error("alternate weight tensor has an incompatible shape");
}
alternate_weights = std::make_unique<ling3::DynamicW4Linear>(
ling3::W4LinearConfig {k, n, k_splits, cores},
TensorSpan<std::byte>(alternate_weight), TensorSpan<float>(alternate_scale),
TensorSpan<std::int32_t>(alternate_correction));
}
auto & reference_linear = alternate_weights ? *alternate_weights : linear;
std::vector<float> input(rows * static_cast<std::size_t>(k));
for (std::size_t row = 0; row < rows; ++row) {
for (int column = 0; column < k; ++column) {
input[row * k + column] =
0.7F * std::sin(static_cast<float>(column + 7 * row) * 0.03125F) +
0.2F * std::cos(static_cast<float>(3 * column + row) * 0.0078125F);
}
}
std::vector<float> batch_output(rows * static_cast<std::size_t>(n));
std::vector<float> sequential_output(rows * static_cast<std::size_t>(n));
linear.RunBatchWithWeights(input, rows, reference_linear, batch_output);
for (std::size_t row = 0; row < rows; ++row) {
reference_linear.Run(
std::span<const float>(input).subspan(row * k, k),
std::span<float>(sequential_output).subspan(row * n, n));
}
double maximum_error = 0.0;
double mean_error = 0.0;
for (std::size_t index = 0; index < batch_output.size(); ++index) {
const double error = std::abs(
static_cast<double>(batch_output[index]) - sequential_output[index]);
maximum_error = std::max(maximum_error, error);
mean_error += error;
}
mean_error /= static_cast<double>(batch_output.size());
// Check shared quantized inputs with reordered/repeated rows, a smaller
// tail, and a return to the original capacity in the same context.
double indexed_maximum_error = 0.0;
if (std::getenv("LING3_PREFILL_W4A4") == nullptr) {
std::vector<std::int8_t> quantized(input.size());
std::vector<float> input_scales(rows);
std::vector<std::size_t> indices(rows);
for (std::size_t row = 0; row < rows; ++row) {
input_scales[row] = ling3::QuantizeSymmetricInt8(
std::span<const float>(input).subspan(row * k, k),
std::span<std::int8_t>(quantized).subspan(row * k, k)).scale;
indices[row] = rows - 1 - row / 2;
}
for (const std::size_t count : {rows, std::max<std::size_t>(1, rows / 2), rows}) {
linear.RunBatchQuantizedRows(
quantized, input_scales,
std::span<const std::size_t>(indices).first(count), reference_linear,
std::span<float>(batch_output).first(count * n));
for (std::size_t row = 0; row < count; ++row) {
for (int column = 0; column < n; ++column) {
const double actual = batch_output[row * n + column];
if (!std::isfinite(actual)) {
throw std::runtime_error("indexed W4 batch produced a non-finite output");
}
indexed_maximum_error = std::max(indexed_maximum_error, std::abs(
actual - sequential_output[indices[row] * n + column]));
}
}
}
}
ling3::W4RunTimings mean;
const auto sequential_begin = std::chrono::steady_clock::now();
for (std::size_t iteration = 0; iteration < iterations; ++iteration) {
for (std::size_t row = 0; row < rows; ++row) {
reference_linear.Run(
std::span<const float>(input).subspan(row * k, k),
std::span<float>(sequential_output).subspan(
row * n, n));
}
}
const double sequential_ms = std::chrono::duration<double, std::milli>(
std::chrono::steady_clock::now() - sequential_begin).count() /
static_cast<double>(iterations);
for (std::size_t iteration = 0; iteration < iterations; ++iteration) {
const auto sample = linear.RunBatchWithWeights(input, rows, reference_linear, batch_output);
mean.quantize_pack_ms += sample.quantize_pack_ms;
mean.input_sync_ms += sample.input_sync_ms;
mean.npu_ms += sample.npu_ms;
mean.gather_ms += sample.gather_ms;
mean.total_ms += sample.total_ms;
}
const double divisor = static_cast<double>(iterations);
mean.quantize_pack_ms /= divisor;
mean.input_sync_ms /= divisor;
mean.npu_ms /= divisor;
mean.gather_ms /= divisor;
mean.total_ms /= divisor;
std::cout << std::fixed << std::setprecision(6)
<< "tensor=" << base << '\n'
<< "weight_tensor=" << (weight_base.empty() ? base : weight_base) << '\n'
<< "rows=" << rows << '\n'
<< "cores=" << (cores.size() == 1 ? "0" : "0,1,2") << '\n'
<< "sequential_mean_ms=" << sequential_ms << '\n'
<< "batch_quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n'
<< "batch_input_sync_mean_ms=" << mean.input_sync_ms << '\n'
<< "batch_npu_mean_ms=" << mean.npu_ms << '\n'
<< "batch_gather_mean_ms=" << mean.gather_ms << '\n'
<< "batch_total_mean_ms=" << mean.total_ms << '\n'
<< "speedup=" << sequential_ms / mean.total_ms << '\n'
<< "sequential_mean_abs_error=" << mean_error << '\n'
<< "sequential_max_abs_error=" << maximum_error << '\n'
<< "indexed_max_abs_error=" << indexed_maximum_error << '\n';
return maximum_error <= 1.0e-5 && indexed_maximum_error <= 1.0e-5 ? 0 : 13;
}
std::array<double, 3> CompareVectors(
std::span<const float> actual,
std::span<const float> expected) {
if (actual.size() != expected.size() || actual.empty()) {
throw std::invalid_argument("comparison vectors have incompatible sizes");
}
double dot = 0.0;
double actual_norm = 0.0;
double expected_norm = 0.0;
double mean_error = 0.0;
double maximum_error = 0.0;
for (std::size_t index = 0; index < actual.size(); ++index) {
const double a = actual[index];
const double b = expected[index];
const double error = std::abs(a - b);
dot += a * b;
actual_norm += a * a;
expected_norm += b * b;
mean_error += error;
maximum_error = std::max(maximum_error, error);
}
return {
dot / std::sqrt(actual_norm * expected_norm),
mean_error / static_cast<double>(actual.size()),
maximum_error,
};
}
int RunGdnReferenceCheck(const std::string & path) {
if (std::getenv("LING3_GDN_CPU_DECODE"))
throw std::invalid_argument("unset LING3_GDN_CPU_DECODE for the NPU comparator");
ConfigureInferenceProcess();
const ling3::ModelPackage package(path);
const auto h6 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads6"));
const auto h5 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads5"));
ling3::GdnStep npu(h6, h5), cpu(h6, h5);
constexpr int heads = 16, dimension = 128, width = heads * dimension, steps = 256;
std::vector<float> q(width), k(width), v(width), d(width), b(heads), a(width), c(width);
std::vector<double> state(heads * dimension * dimension, 0.0);
double npu_square_error = 0, cpu_square_error = 0, reference_square = 0;
double npu_max = 0, cpu_max = 0;
for (int t = 0; t < steps; ++t) {
FillGdnToken(t, q, k, v, d, b);
// Include near-unit retention as well as fast forgetting; a zero-state
// first-token test alone cannot exercise accumulated recurrence error.
for (int i = 0; i < width; ++i)
d[i] = i % 3 == 0 ? -0.0001F : (i % 3 == 1 ? -0.05F : d[i]);
npu.Run(q, k, v, d, b, a);
cpu.RunBatchCpu(q, k, v, d, b, c);
for (int head = 0; head < heads; ++head) {
std::array<double, dimension> factor;
for (int j = 0; j < dimension; ++j) factor[j] = std::exp(double(d[head * dimension + j]));
for (int col = 0; col < dimension; ++col) {
auto * row = state.data() + (head * dimension + col) * dimension;
double predicted = 0;
for (int j = 0; j < dimension; ++j) {
row[j] *= factor[j];
predicted += row[j] * double(k[head * dimension + j]);
}
const double delta = double(b[head]) * (double(v[head * dimension + col]) - predicted);
double result = 0;
for (int j = 0; j < dimension; ++j) {
row[j] += delta * double(k[head * dimension + j]);
result += row[j] * double(q[head * dimension + j]) / std::sqrt(128.0);
}
const int index = head * dimension + col;
if (!std::isfinite(a[index]) || !std::isfinite(c[index]))
throw std::runtime_error("non-finite recurrence");
const double ea = a[index] - result, ec = c[index] - result;
npu_square_error += ea * ea;
cpu_square_error += ec * ec;
reference_square += result * result;
npu_max = std::max(npu_max, std::abs(ea));
cpu_max = std::max(cpu_max, std::abs(ec));
}
}
}
std::cout << std::setprecision(12) << "steps=" << steps
<< " npu_relative_rms=" << std::sqrt(npu_square_error / reference_square)
<< " cpu_relative_rms=" << std::sqrt(cpu_square_error / reference_square)
<< " npu_max=" << npu_max << " cpu_max=" << cpu_max << '\n';
return cpu_square_error < npu_square_error && cpu_max < 1e-6 ? 0 : 2;
}
int RunGdnCpuStateCheck(const std::string & path) {
if (std::getenv("LING3_GDN_CPU_DECODE"))
throw std::invalid_argument("unset LING3_GDN_CPU_DECODE to test transitions to NPU");
ConfigureInferenceProcess();
const ling3::ModelPackage package(path);
const auto h6 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads6"));
const auto h5 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads5"));
ling3::GdnStep batch(h6, h5), chunks(h6, h5);
constexpr std::size_t rows = 97, width = 16 * 128;
std::vector<float> q(rows * width), k(q.size()), v(q.size()), d(q.size()), b(rows * 16);
std::vector<float> one(q.size()), many(q.size()), replay(q.size());
for (std::size_t t = 0; t < rows; ++t) {
FillGdnToken(t, std::span<float>(q).subspan(t * width, width),
std::span<float>(k).subspan(t * width, width),
std::span<float>(v).subspan(t * width, width),
std::span<float>(d).subspan(t * width, width),
std::span<float>(b).subspan(t * 16, 16));
}
batch.RunBatchCpu(q, k, v, d, b, one);
std::size_t offset = 0;
for (const std::size_t count : {1, 3, 16, 31, 46}) {
chunks.RunBatchCpu(std::span<const float>(q).subspan(offset * width, count * width),
std::span<const float>(k).subspan(offset * width, count * width),
std::span<const float>(v).subspan(offset * width, count * width),
std::span<const float>(d).subspan(offset * width, count * width),
std::span<const float>(b).subspan(offset * 16, count * 16),
std::span<float>(many).subspan(offset * width, count * width));
offset += count;
}
const auto error = CompareVectors(many, one);
std::vector<float> tail1(width), tail2(width);
batch.Run(std::span<const float>(q).last(width), std::span<const float>(k).last(width),
std::span<const float>(v).last(width), std::span<const float>(d).last(width),
std::span<const float>(b).last(16), tail1);
chunks.Run(std::span<const float>(q).last(width), std::span<const float>(k).last(width),
std::span<const float>(v).last(width), std::span<const float>(d).last(width),
std::span<const float>(b).last(16), tail2);
const auto transition = CompareVectors(tail1, tail2);
// After a device step, return to CPU and verify the captured shadow state.
batch.RunBatchCpu(q, k, v, d, b, replay);
chunks.RunBatchCpu(q, k, v, d, b, many);
const auto back = CompareVectors(replay, many);
batch.Reset();
batch.RunBatchCpu(q, k, v, d, b, replay);
const auto reset = CompareVectors(replay, one);
std::cout << "chunk_max_error=" << error[2] << " cpu_to_npu_max_error=" << transition[2]
<< " npu_to_cpu_max_error=" << back[2] << " reset_max_error=" << reset[2] << '\n';
return error[2] == 0 && transition[2] == 0 && back[2] == 0 && reset[2] == 0 ? 0 : 2;
}
int RunPrefill32Smoke(const std::string & package_path) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("prefill32-smoke requires a build with RKNN support");
}
ConfigureInferenceProcess();
const ling3::ModelPackage package(package_path);
const auto tokenizer = PackageTokenizer(package);
auto encoded = tokenizer.Encode(Utf8(
u8"<role>SYSTEM</role>detailed thinking off<|role_end|>"
u8"<role>HUMAN</role>请用中文简要说明今天的工作安排,并列出三个重点事项。"
u8"<|role_end|><role>ASSISTANT</role>\n<think></think>"));
if (encoded.size() < 32) throw std::runtime_error("prefill32 test prompt is too short");
encoded.resize(32);
ling3::Decoder decoder(package);
if (!decoder.has_dynamic_batch()) {
throw std::runtime_error("set LING3_GDN_PREFILL_DIR for prefill32-smoke");
}
std::vector<float> sequential_logits(package.header().vocab_size);
std::vector<float> sequential_tail(package.header().vocab_size);
std::vector<float> batch_logits(package.header().vocab_size);
std::vector<float> batch_tail(package.header().vocab_size);
double sequential_ms = 0.0;
for (const auto token : encoded) {
sequential_ms += decoder.Eval(token, sequential_logits).total_ms;
}
const auto next_token = static_cast<std::uint32_t>(
std::max_element(sequential_logits.begin(), sequential_logits.end()) -
sequential_logits.begin());
decoder.Eval(next_token, sequential_tail);
decoder.Reset();
const auto cold_batch = decoder.EvalBatch32(encoded, batch_logits);
decoder.Eval(next_token, batch_tail);
const auto logits_error = CompareVectors(batch_logits, sequential_logits);
const auto tail_error = CompareVectors(batch_tail, sequential_tail);
decoder.Reset();
const auto warm_batch = decoder.EvalBatch32(encoded, batch_logits);
std::cout << std::fixed << std::setprecision(8)
<< "tokens=32\n"
<< "sequential_ms=" << sequential_ms << '\n'
<< "cold_batch_ms=" << cold_batch.total_ms << '\n'
<< "warm_batch_ms=" << warm_batch.total_ms << '\n'
<< "speedup=" << sequential_ms / warm_batch.total_ms << '\n'
<< "logits_cosine=" << logits_error[0] << '\n'
<< "logits_mean_abs_error=" << logits_error[1] << '\n'
<< "logits_max_abs_error=" << logits_error[2] << '\n'
<< "tail_logits_cosine=" << tail_error[0] << '\n'
<< "tail_logits_mean_abs_error=" << tail_error[1] << '\n'
<< "tail_logits_max_abs_error=" << tail_error[2] << '\n'
<< "next_token=" << next_token << '\n';
return logits_error[0] >= 0.999 && tail_error[0] >= 0.999 ? 0 : 15;
}
std::vector<std::pair<std::uint32_t, float>> TopLogits(
std::span<const float> logits,
std::size_t count) {
count = std::min(count, logits.size());
std::vector<std::uint32_t> indices(logits.size());
std::iota(indices.begin(), indices.end(), 0U);
std::partial_sort(
indices.begin(), indices.begin() + static_cast<std::ptrdiff_t>(count), indices.end(),
[&logits](std::uint32_t left, std::uint32_t right) {
return logits[left] > logits[right];
});
std::vector<std::pair<std::uint32_t, float>> result;
result.reserve(count);
for (std::size_t index = 0; index < count; ++index) {
result.emplace_back(indices[index], logits[indices[index]]);
}
return result;
}
int RunLogitsSmoke(
const std::string & package_path,
std::uint32_t token,
const std::string * reference_path) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("logits-smoke requires a build with RKNN support");
}
ConfigureInferenceProcess();
const ling3::ModelPackage package(package_path);
const auto tokenizer = PackageTokenizer(package);
const auto initialize_begin = std::chrono::steady_clock::now();
ling3::Decoder decoder(package);
const auto initialize_end = std::chrono::steady_clock::now();
std::vector<float> logits(package.header().vocab_size);
const auto timings = decoder.Eval(token, logits);
const auto top = TopLogits(logits, 10);
std::cout << std::fixed << std::setprecision(6)
<< "token=" << token << '\n'
<< "initialization_ms="
<< std::chrono::duration<double, std::milli>(initialize_end - initialize_begin).count()
<< '\n'
<< "layers_ms=" << timings.layers_ms << '\n'
<< "output_head_ms=" << timings.output_head_ms << '\n'
<< "total_ms=" << timings.total_ms << '\n';
for (std::size_t index = 0; index < top.size(); ++index) {
const std::array<std::uint32_t, 1> id {top[index].first};
std::cout << "top" << index << "_id=" << top[index].first
<< " logit=" << top[index].second
<< " piece=" << std::quoted(tokenizer.Decode(id)) << '\n';
}
if (reference_path == nullptr) return 0;
const auto reference = ReadFloatFile(*reference_path, logits.size());
double dot = 0.0, norm = 0.0, reference_norm = 0.0;
double mean_error = 0.0, maximum_error = 0.0;
for (std::size_t index = 0; index < logits.size(); ++index) {
const double error = std::abs(static_cast<double>(logits[index]) - reference[index]);
mean_error += error;
maximum_error = std::max(maximum_error, error);
dot += static_cast<double>(logits[index]) * reference[index];
norm += static_cast<double>(logits[index]) * logits[index];
reference_norm += static_cast<double>(reference[index]) * reference[index];
}
mean_error /= static_cast<double>(logits.size());
const double cosine = dot / std::sqrt(norm * reference_norm);
std::cout << "cosine_vs_reference=" << cosine << '\n'
<< "mean_abs_error=" << mean_error << '\n'
<< "max_abs_error=" << maximum_error << '\n';
return cosine >= 0.90 ? 0 : 13;
}
int RunGenerate(
const std::string & package_path,
std::string_view user_text,
std::size_t maximum_new_tokens) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("generate requires a build with RKNN support");
}
if (maximum_new_tokens == 0) throw std::invalid_argument("MAX_NEW_TOKENS must be positive");
ConfigureInferenceProcess();
const ling3::ModelPackage package(package_path);
const auto tokenizer = PackageTokenizer(package);
const std::string prompt =
"<role>SYSTEM</role>detailed thinking off<|role_end|>"
"<role>HUMAN</role>" + std::string(user_text) +
"<|role_end|><role>ASSISTANT</role>\n<think></think>";
const auto tokens = tokenizer.Encode(prompt);
if (tokens.empty() || tokens.size() + maximum_new_tokens > package.header().max_context) {
throw std::runtime_error("prompt and output exceed the package context capacity");
}
const auto initialize_begin = std::chrono::steady_clock::now();
ling3::Decoder decoder(package);
const auto initialize_end = std::chrono::steady_clock::now();
std::vector<float> logits(package.header().vocab_size);
const auto prefill_begin = std::chrono::steady_clock::now();
for (const auto token : tokens) decoder.Eval(token, logits);
const auto prefill_end = std::chrono::steady_clock::now();
std::vector<double> decode_ms;
std::vector<std::uint32_t> generated;
decode_ms.reserve(maximum_new_tokens);
generated.reserve(maximum_new_tokens);
std::cout << "response=" << std::flush;
for (std::size_t index = 0; index < maximum_new_tokens; ++index) {
const auto found = std::max_element(logits.begin(), logits.end());
const auto token = static_cast<std::uint32_t>(found - logits.begin());
if (token == package.header().eos_token) break;
generated.push_back(token);
std::cout << tokenizer.Piece(token) << std::flush;
const auto timing = decoder.Eval(token, logits);
decode_ms.push_back(timing.total_ms);
std::cerr << "\ntoken=" << token << " position=" << decoder.position()
<< " layers_ms=" << timing.layers_ms
<< " head_ms=" << timing.output_head_ms
<< " total_ms=" << timing.total_ms << std::flush;
}
std::cout << '\n';
const double initialize_ms = std::chrono::duration<double, std::milli>(
initialize_end - initialize_begin).count();
const double prefill_ms = std::chrono::duration<double, std::milli>(
prefill_end - prefill_begin).count();
const double decode_total = std::accumulate(decode_ms.begin(), decode_ms.end(), 0.0);
std::cerr << '\n' << std::fixed << std::setprecision(3)
<< "prompt_tokens=" << tokens.size() << '\n'
<< "generated_tokens=" << generated.size() << '\n'
<< "initialization_ms=" << initialize_ms << '\n'
<< "prefill_ms=" << prefill_ms << '\n'
<< "prefill_tokens_per_second="
<< (prefill_ms == 0.0 ? 0.0 : 1000.0 * tokens.size() / prefill_ms) << '\n'
<< "decode_ms=" << decode_total << '\n'
<< "decode_tokens_per_second="
<< (decode_total == 0.0 ? 0.0 : 1000.0 * decode_ms.size() / decode_total) << '\n';
return 0;
}
struct ProcessMemory {
std::size_t rss_kb = 0;
std::size_t hwm_kb = 0;
};
ProcessMemory ReadProcessMemory() {
#if defined(__linux__)
std::ifstream status("/proc/self/status");
if (!status) throw std::runtime_error("cannot read /proc/self/status");
ProcessMemory memory;
std::string line;
while (std::getline(status, line)) {
std::istringstream fields(line);
std::string key;
std::size_t value = 0;
std::string unit;
if (!(fields >> key >> value >> unit)) continue;
if (key == "VmRSS:") memory.rss_kb = value;
if (key == "VmHWM:") memory.hwm_kb = value;
}
return memory;
#else
return {};
#endif
}
std::vector<std::uint32_t> BuildBenchmarkPrompt(
const ling3::Tokenizer & tokenizer,
std::size_t sequence_length) {
const auto prefix = tokenizer.Encode(
"<role>SYSTEM</role>detailed thinking off<|role_end|>"
"<role>HUMAN</role>");
const auto body = tokenizer.Encode(Utf8(
u8"请用中文详细说明如何安排一天的工作,依次讨论目标、执行步骤、风险和复盘方法。"
u8"回答需要完整、连贯并包含具体例子。"));
const auto suffix = tokenizer.Encode(
"<|role_end|><role>ASSISTANT</role>\n<think></think>");
if (body.empty() || prefix.size() + suffix.size() > sequence_length) {
throw std::invalid_argument("SEQLEN is too short for the deterministic benchmark prompt");
}
std::vector<std::uint32_t> prompt;
prompt.reserve(sequence_length);
prompt.insert(prompt.end(), prefix.begin(), prefix.end());
std::size_t body_index = 0;
while (prompt.size() + suffix.size() < sequence_length) {
prompt.push_back(body[body_index++ % body.size()]);
}
prompt.insert(prompt.end(), suffix.begin(), suffix.end());
return prompt;
}
std::uint32_t GreedyToken(std::span<const float> logits) {
return static_cast<std::uint32_t>(
std::max_element(logits.begin(), logits.end()) - logits.begin());
}
std::uint64_t TokenHash(std::span<const std::uint32_t> tokens) {
std::uint64_t hash = 1469598103934665603ULL;
for (const auto token : tokens) {
for (unsigned shift = 0; shift < 32; shift += 8) {
hash ^= (token >> shift) & 0xffU;
hash *= 1099511628211ULL;
}
}
return hash;
}
// Reference tokens are teacher-forced so small numerical differences cannot
// change the input sequence and invalidate comparisons of recurrent state.
int RunSequenceCheck(const std::string & path, std::size_t length,
std::size_t steps, const std::string & out,
const std::string & reference, const std::string & corpus = "") {
ConfigureInferenceProcess();
const ling3::ModelPackage package(path);
if (!steps || length + steps > package.header().max_context)
throw std::invalid_argument("invalid sequence-check lengths");
const auto tokenizer = PackageTokenizer(package);
std::vector<std::uint32_t> corpus_tokens;
auto prompt = BuildBenchmarkPrompt(tokenizer, length);
if (!corpus.empty()) {
std::ifstream file(corpus);
if (!file) throw std::runtime_error("cannot open evaluation corpus");
const std::string body{std::istreambuf_iterator<char>(file), std::istreambuf_iterator<char>()};
corpus_tokens = tokenizer.Encode(body);
if (corpus_tokens.size() < length + steps)
throw std::runtime_error("evaluation corpus is too short: " + std::to_string(corpus_tokens.size()));
prompt.assign(corpus_tokens.begin(), corpus_tokens.begin() + length);
std::ofstream teacher(out + ".teacher.ids");
for (std::size_t i = 0; i < steps; ++i) teacher << corpus_tokens[length + i] << '\n';
if (!teacher) throw std::runtime_error("cannot save evaluation teacher IDs");
}
std::ofstream prompt_ids(out + ".prompt.ids");
for (const auto token : prompt) prompt_ids << token << '\n';
if (!prompt_ids) throw std::runtime_error("cannot save sequence-check prompt IDs");
ling3::Decoder decoder(package);
std::vector<float> logits(package.header().vocab_size), expected(logits.size());
std::ofstream values(out + ".f32", std::ios::binary), ids(out + ".ids");
std::ifstream ref_values, ref_ids;
if (!reference.empty()) {
ref_values.open(reference + ".f32", std::ios::binary);
ref_ids.open(reference + ".ids");
if (!ref_values || !ref_ids) throw std::runtime_error("cannot open reference");
}
if (!values || !ids) throw std::runtime_error("cannot open sequence-check output");
std::size_t offset = 0;
while (offset < length) {
const auto rows = std::min<std::size_t>(128, length - offset);
if (rows == 1) decoder.Eval(prompt[offset], logits);
else {
decoder.PrepareBatch(rows);
decoder.EvalBatch(std::span<const std::uint32_t>(prompt).subspan(offset, rows), logits);
}
offset += rows;
}
double min_cosine = 1.0, max_error = 0.0;
std::size_t agreement = 0;
for (std::size_t step = 0; step < steps; ++step) {
for (const auto x : logits)
if (!std::isfinite(x)) throw std::runtime_error("non-finite logits");
const auto predicted = GreedyToken(logits);
auto token = predicted;
values.write(reinterpret_cast<const char *>(logits.data()), logits.size() * sizeof(float));
ids << predicted << '\n';
if (!reference.empty()) {
ref_values.read(reinterpret_cast<char *>(expected.data()), expected.size() * sizeof(float));
if (!ref_values || !(ref_ids >> token) || token >= logits.size())
throw std::runtime_error("invalid/truncated reference");
for (const auto x : expected)
if (!std::isfinite(x)) throw std::runtime_error("non-finite reference logits");
const auto error = CompareVectors(logits, expected);
min_cosine = std::min(min_cosine, error[0]);
max_error = std::max(max_error, error[2]);
agreement += token == predicted;
std::cout << "step=" << step << " cosine=" << std::setprecision(10)
<< error[0] << " mae=" << error[1] << " max=" << error[2]
<< " top1=" << (token == predicted) << '\n';
}
if (!corpus_tokens.empty()) token = corpus_tokens[length + step];
if (step + 1 < steps) decoder.Eval(token, logits);
}
if (!values || !ids) throw std::runtime_error("failed writing reference");
std::cout << "steps=" << steps << " min_cosine=" << min_cosine
<< " max_error=" << max_error << " top1_agreement=" << agreement << '\n';
const auto mla=decoder.AttentionStats();
std::cout << "mla_npu_calls=" << mla.npu_calls << " mla_cpu_calls=" << mla.cpu_calls
<< " mla_fallbacks=" << mla.fallbacks << " mla_npu_ms=" << mla.npu_ms << '\n';
return reference.empty() || min_cosine >= 0.999 ? 0 : 2;
}
struct BenchmarkSample {
double ttft_ms = 0.0;
double prefill_ms = 0.0;
double decode_ms = 0.0;
double tokens_per_second = 0.0;
std::size_t eos_tokens = 0;
std::uint64_t token_hash = 0;
ProcessMemory memory;
};
#if LING3_EXPERIMENTAL_MTP
// Same resident decoder for baseline and MTP. Populate every MTP prefix slot
// sequentially, and teacher-force baseline tokens so the work and histories
// stay comparable. This measures forward overhead, not speculative speedup.
int RunMtpBenchmark(const std::string & path, std::size_t length,
std::size_t steps, std::size_t repeats, std::size_t capacity,
const std::string & cases_path) {
ConfigureInferenceProcess();
const ling3::ModelPackage package(path);
if (!length || steps < 2 || !repeats || length + steps > capacity || capacity > 262144)
throw std::invalid_argument("invalid MTP benchmark lengths");
const auto tokenizer = PackageTokenizer(package);
std::vector<std::vector<std::uint32_t>> prompts{BuildBenchmarkPrompt(tokenizer, length)};
if (!cases_path.empty()) {
std::ifstream file(cases_path);
if (!file) throw std::runtime_error("cannot open MTP cases file");
std::string question;
while (std::getline(file, question)) {
if (question.empty()) continue;
auto ids = tokenizer.Encode("<role>SYSTEM</role>detailed thinking off<|role_end|>"
"<role>HUMAN</role>" + question + "<|role_end|><role>ASSISTANT</role>\n<think></think>");
if (ids.empty() || ids.size() + steps > capacity)
throw std::invalid_argument("case does not fit the requested context");
prompts.push_back(std::move(ids));
}
}
const auto started = std::chrono::steady_clock::now();
ling3::Decoder decoder(package, capacity);
decoder.EnableMtp();
const std::vector<bool> modes = decoder.HasMtp() ? std::vector<bool>{false, true} : std::vector<bool>{false};
std::vector<float> logits(package.header().vocab_size), draft(logits.size());
const auto milliseconds = [](auto a, auto b) {
return std::chrono::duration<double, std::milli>(b - a).count();
};
std::cout << std::fixed << std::setprecision(4)
<< "benchmark=resident_mtp_forward_not_speculative\n"
<< "context_capacity=" << capacity << " cases=" << prompts.size()
<< " mtp_enabled=" << decoder.HasMtp() << " new_tokens=" << steps << " iterations=" << repeats << '\n'
<< "initialization_ms=" << milliseconds(started, std::chrono::steady_clock::now()) << std::endl;
const auto initial_memory = ReadProcessMemory();
std::cout << "initial_rss_mib=" << initial_memory.rss_kb / 1024.0
<< " initial_hwm_mib=" << initial_memory.hwm_kb / 1024.0 << std::endl;
for (std::size_t case_index = 0; case_index < prompts.size(); ++case_index) {
const auto & prompt = prompts[case_index];
length = prompt.size();
for (std::size_t round = 0; round <= repeats; ++round) {
std::vector<std::uint32_t> teacher;
teacher.reserve(steps);
for (bool with_mtp : modes) {
decoder.Reset();
const auto attention_before = decoder.AttentionStats();
const auto begin = std::chrono::steady_clock::now();
double trunk = 0, mtp = 0, mtp_layers = 0, mtp_head = 0;
for (std::size_t i = 0; i < length; ++i) {
decoder.Eval(prompt[i], logits);
if (with_mtp && i + 1 < length) decoder.EvalMtp(prompt[i + 1], draft);
}
std::uint32_t token = GreedyToken(logits);
std::size_t target_agreement = 0, draft_hits = 0;
if (!with_mtp) teacher.push_back(token);
else {
target_agreement += token == teacher[0];
token = teacher[0];
}
const auto first = std::chrono::steady_clock::now();
std::vector<double> trunk_samples, mtp_samples;
for (std::size_t i = 1; i < steps; ++i) {
std::uint32_t proposed = 0;
if (with_mtp) {
const auto t = decoder.EvalMtp(token, draft);
mtp += t.total_ms; mtp_layers += t.layers_ms; mtp_head += t.output_head_ms;
mtp_samples.push_back(t.total_ms);
proposed = GreedyToken(draft);
}
const auto t = decoder.Eval(token, logits);
trunk += t.total_ms; trunk_samples.push_back(t.total_ms);
token = GreedyToken(logits);
if (!with_mtp) teacher.push_back(token);
else {
target_agreement += token == teacher[i];
draft_hits += proposed == teacher[i];
token = teacher[i];
}
}
const auto end = std::chrono::steady_clock::now();
const auto percentile = [](std::vector<double> values, double q) {
std::sort(values.begin(), values.end());
return values[static_cast<std::size_t>(q * (values.size() - 1))];
};
const auto memory = ReadProcessMemory();
const auto attention = decoder.AttentionStats();
std::cout << "phase=" << (round == 0 ? "warmup" : "measured")
<< " case=" << case_index << " seqlen=" << length
<< " round=" << round << " mode=" << (with_mtp ? "target_plus_mtp" : "target_only")
<< " sequential_prefill_ms=" << milliseconds(begin, first)
<< " decode_wall_ms=" << milliseconds(first, end)
<< " tokens_per_second=" << 1000.0 * (steps - 1) / milliseconds(first, end)
<< " target_mean_ms=" << trunk / (steps - 1)
<< " target_p50_ms=" << percentile(trunk_samples, 0.5)
<< " target_p95_ms=" << percentile(trunk_samples, 0.95);
if (with_mtp)
std::cout << " mtp_mean_ms=" << mtp / (steps - 1)
<< " mtp_p50_ms=" << percentile(mtp_samples, 0.5)
<< " mtp_p95_ms=" << percentile(mtp_samples, 0.95)
<< " mtp_layers_mean_ms=" << mtp_layers / (steps - 1)
<< " mtp_head_mean_ms=" << mtp_head / (steps - 1)
<< " target_agreement=" << target_agreement << '/' << steps
<< " draft_top1_hits=" << draft_hits << '/' << (steps - 1);
std::cout << " mla_npu_calls=" << attention.npu_calls - attention_before.npu_calls
<< " mla_cpu_calls=" << attention.cpu_calls - attention_before.cpu_calls
<< " mla_fallbacks=" << attention.fallbacks - attention_before.fallbacks
<< " rss_mib=" << memory.rss_kb / 1024.0
<< " hwm_mib=" << memory.hwm_kb / 1024.0
<< " teacher_hash=" << std::hex << TokenHash(teacher) << std::dec << std::endl;
if (with_mtp && target_agreement != steps)
throw std::runtime_error("MTP probe changed target predictions for the same teacher-forced tokens");
}
}
}
return 0;
}
#endif
int RunBenchmark(
const std::string & package_path,
std::size_t sequence_length,
std::size_t new_tokens,
std::size_t iterations) {
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("benchmark requires a build with RKNN support");
}
if (sequence_length == 0 || new_tokens < 2 || iterations == 0) {
throw std::invalid_argument(
"SEQLEN and ITERATIONS must be positive; NEW_TOKENS must be at least 2");
}
ConfigureInferenceProcess();
const ling3::ModelPackage package(package_path);
const auto tokenizer = PackageTokenizer(package);
if (sequence_length + new_tokens > package.header().max_context) {
throw std::invalid_argument("SEQLEN + NEW_TOKENS exceeds the package context capacity");
}
const auto prompt = BuildBenchmarkPrompt(tokenizer, sequence_length);
const auto initialization_begin = std::chrono::steady_clock::now();
ling3::Decoder decoder(package);
const std::size_t batch_granularity = decoder.batch_granularity();
std::size_t prefill_batch = std::min<std::size_t>(sequence_length, 128);
if (batch_granularity != 0) prefill_batch -= prefill_batch % batch_granularity;
if (prefill_batch >= 2) decoder.PrepareBatch(prefill_batch);
const auto initialization_end = std::chrono::steady_clock::now();
const double initialization_ms = std::chrono::duration<double, std::milli>(
initialization_end - initialization_begin).count();
std::vector<float> logits(package.header().vocab_size);
const auto run_once = [&]() {
BenchmarkSample sample;
decoder.Reset();
const auto request_begin = std::chrono::steady_clock::now();
std::size_t offset = 0;
while (batch_granularity != 0 && sequence_length - offset >= 2) {
std::size_t rows = std::min<std::size_t>(sequence_length - offset, 128);
rows -= rows % batch_granularity;
if (rows < 2) break;
const auto block = std::span<const std::uint32_t>(prompt).subspan(offset, rows);
const bool final_block = sequence_length - offset == rows;
if (final_block) {
decoder.EvalBatch(block, logits);
} else {
decoder.EvalBatchState(block);
}
offset += rows;
}
for (; offset < sequence_length; ++offset) decoder.Eval(prompt[offset], logits);
const auto first_token = GreedyToken(logits);
const auto first_token_at = std::chrono::steady_clock::now();
sample.ttft_ms = std::chrono::duration<double, std::milli>(
first_token_at - request_begin).count();
sample.prefill_ms = sample.ttft_ms;
std::vector<std::uint32_t> generated;
generated.reserve(new_tokens);
generated.push_back(first_token);
if (first_token == package.header().eos_token) ++sample.eos_tokens;
for (std::size_t index = 1; index < new_tokens; ++index) {
decoder.Eval(generated.back(), logits);
const auto token = GreedyToken(logits);
generated.push_back(token);
if (token == package.header().eos_token) ++sample.eos_tokens;
}
const auto final_token_at = std::chrono::steady_clock::now();
sample.decode_ms = std::chrono::duration<double, std::milli>(
final_token_at - first_token_at).count();
sample.tokens_per_second = 1000.0 * static_cast<double>(new_tokens - 1) /
sample.decode_ms;
sample.token_hash = TokenHash(generated);
sample.memory = ReadProcessMemory();
return sample;
};
const auto before_warmup = ReadProcessMemory();
const auto warmup = run_once();
std::vector<BenchmarkSample> samples;
samples.reserve(iterations);
for (std::size_t iteration = 0; iteration < iterations; ++iteration) {
samples.push_back(run_once());
}
const auto statistic = [&samples](auto member, bool minimum) {
double result = samples.front().*member;
for (const auto & sample : samples) {
result = minimum ? std::min(result, sample.*member) : std::max(result, sample.*member);
}
return result;
};
const auto mean = [&samples](auto member) {
double total = 0.0;
for (const auto & sample : samples) total += sample.*member;
return total / static_cast<double>(samples.size());
};
std::size_t maximum_rss_kb = 0;
std::size_t maximum_hwm_kb = 0;
for (const auto & sample : samples) {
maximum_rss_kb = std::max(maximum_rss_kb, sample.memory.rss_kb);
maximum_hwm_kb = std::max(maximum_hwm_kb, sample.memory.hwm_kb);
}
std::cout << std::fixed << std::setprecision(3)
<< "benchmark=official_style_autoregressive\n"
<< "model=Ling-3.0-tiny\n"
<< "model_total_parameters_b=7.9\n"
<< "model_active_parameters_b=1.3\n"
<< "dtype=W4A8+FP16/FP32_mixed\n"
<< "seqlen=" << sequence_length << '\n'
<< "new_tokens=" << new_tokens << '\n'
<< "iterations=" << iterations << '\n'
<< "cpu_cores=4,5,6,7\n"
<< "npu_cores=0,1,2\n"
<< "initialization_ms=" << initialization_ms << '\n'
<< "model_file_mb=" << package.mapped_bytes() / (1024.0 * 1024.0) << '\n'
<< "rss_before_warmup_mb=" << before_warmup.rss_kb / 1024.0 << '\n'
<< "warmup_ttft_ms=" << warmup.ttft_ms << '\n'
<< "warmup_tokens_per_second=" << warmup.tokens_per_second << '\n';
for (std::size_t index = 0; index < samples.size(); ++index) {
const auto & sample = samples[index];
std::cout << "sample_" << index + 1 << "_ttft_ms=" << sample.ttft_ms << '\n'
<< "sample_" << index + 1 << "_tokens_per_second="
<< sample.tokens_per_second << '\n'
<< "sample_" << index + 1 << "_rss_mb=" << sample.memory.rss_kb / 1024.0
<< '\n'
<< "sample_" << index + 1 << "_hwm_mb=" << sample.memory.hwm_kb / 1024.0
<< '\n'
<< "sample_" << index + 1 << "_eos_tokens=" << sample.eos_tokens << '\n'
<< "sample_" << index + 1 << "_token_hash=" << std::hex
<< sample.token_hash << std::dec << '\n';
}
std::cout << "ttft_mean_ms=" << mean(&BenchmarkSample::ttft_ms) << '\n'
<< "ttft_min_ms=" << statistic(&BenchmarkSample::ttft_ms, true) << '\n'
<< "ttft_max_ms=" << statistic(&BenchmarkSample::ttft_ms, false) << '\n'
<< "tokens_per_second_mean=" << mean(&BenchmarkSample::tokens_per_second) << '\n'
<< "tokens_per_second_min="
<< statistic(&BenchmarkSample::tokens_per_second, true) << '\n'
<< "tokens_per_second_max="
<< statistic(&BenchmarkSample::tokens_per_second, false) << '\n'
<< "memory_rss_mb=" << maximum_rss_kb / 1024.0 << '\n'
<< "memory_hwm_mb=" << maximum_hwm_kb / 1024.0 << '\n';
return 0;
}
} // namespace
int main(int argc, char ** argv) {
try {
if (argc < 2) {
PrintUsage(argv[0]);
return 1;
}
const std::string_view command(argv[1]);
if (command == "inspect" && argc == 3) {
const ling3::ModelPackage package(argv[2]);
ling3::ValidateLing3Tiny(package.header());
std::cout << "valid Ling-3.0-tiny package\n"
<< "bytes=" << package.mapped_bytes() << '\n'
<< "tensors=" << package.tensors().size() << '\n'
<< "max_context=" << package.header().max_context << '\n'
<< "weight_bits=" << package.header().default_weight_bits << '\n'
<< "activation_bits=" << package.header().default_activation_bits << '\n';
return 0;
}
if (command == "plan" && argc <= 3) {
const std::size_t max_context = argc == 3 ? ParseSize(argv[2], "context") : 4096;
const ling3::ExecutionPlan plan = ling3::BuildExecutionPlan(max_context);
std::cout << "layers=" << plan.layers.size() << '\n'
<< "decode_matmul_contexts=" << plan.decode_matmul_contexts << '\n'
<< "rknn_island_contexts=" << plan.rknn_island_contexts << '\n'
<< "kda_state_bytes_fp16=" << plan.kda_state_bytes_fp16 << '\n'
<< "mla_cache_bytes_fp16=" << plan.mla_cache_bytes_fp16 << '\n'
<< "activation_arena_bytes=" << plan.activation_arena_bytes << '\n';
return 0;
}
if (command == "rknn-smoke" && argc <= 3) {
const std::size_t iterations = argc == 3 ? ParseSize(argv[2], "iterations") : 10;
return ling3::RunRknnMatmulSmoke(iterations);
}
if (command == "tokenizer-smoke" && argc == 3) {
return RunTokenizerSmoke(argv[2]);
}
if (command == "gdn-smoke" && (argc == 3 || argc == 4)) {
const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 50;
return RunGdnSmoke(argv[2], iterations);
}
if (command == "gdn-batch16-smoke" && (argc == 3 || argc == 4)) {
const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 10;
return RunGdnBatch16Smoke(argv[2], iterations);
}
if (command == "layer0-smoke" && (argc == 5 || argc == 6)) {
const auto token = ParseSize(argv[3], "token");
if (token > std::numeric_limits<std::uint32_t>::max()) {
throw std::runtime_error("token is out of range");
}
const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 20;
return RunLayer0Smoke(
argv[2], static_cast<std::uint32_t>(token), argv[4], iterations);
}
if (command == "linear-smoke" && (argc == 4 || argc == 5)) {
const std::size_t iterations = argc == 5 ? ParseSize(argv[4], "iterations") : 20;
return RunLinearSmoke(argv[2], argv[3], iterations);
}
if (command == "linear-batch-smoke" && argc >= 5 && argc <= 7) {
const std::size_t rows = ParseSize(argv[4], "rows");
const std::size_t iterations = argc >= 6 ? ParseSize(argv[5], "iterations") : 20;
const std::string weight_base = argc == 7 ? argv[6] : argv[3];
return RunLinearBatchSmoke(argv[2], argv[3], rows, iterations, weight_base);
}
if (command == "prefill32-smoke" && argc == 3) {
return RunPrefill32Smoke(argv[2]);
}
if (command == "gdn-cpu-state-check" && argc == 3) {
return RunGdnCpuStateCheck(argv[2]);
}
if (command == "gdn-reference-check" && argc == 3) {
return RunGdnReferenceCheck(argv[2]);
}
if (command == "logits-smoke" && (argc == 4 || argc == 5)) {
const auto token = ParseSize(argv[3], "token");
if (token > std::numeric_limits<std::uint32_t>::max()) {
throw std::runtime_error("token is out of range");
}
const std::string reference = argc == 5 ? argv[4] : "";
return RunLogitsSmoke(
argv[2], static_cast<std::uint32_t>(token), argc == 5 ? &reference : nullptr);
}
if (command == "generate" && (argc == 4 || argc == 5)) {
const std::size_t maximum_new_tokens = argc == 5
? ParseSize(argv[4], "maximum new tokens")
: 64;
return RunGenerate(argv[2], argv[3], maximum_new_tokens);
}
if (command == "benchmark" && (argc == 5 || argc == 6)) {
const std::size_t sequence_length = ParseSize(argv[3], "sequence length");
const std::size_t new_tokens = ParseSize(argv[4], "new tokens");
const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 3;
return RunBenchmark(argv[2], sequence_length, new_tokens, iterations);
}
#if LING3_EXPERIMENTAL_MTP
if (command == "mtp-benchmark" && (argc == 7 || argc == 8)) {
return RunMtpBenchmark(argv[2], ParseSize(argv[3], "sequence length"),
ParseSize(argv[4], "new tokens"), ParseSize(argv[5], "iterations"),
ParseSize(argv[6], "context capacity"), argc == 8 ? argv[7] : "");
}
#endif
if (command == "sequence-check" && argc >= 6 && argc <= 8) {
return RunSequenceCheck(argv[2], ParseSize(argv[3], "sequence length"),
ParseSize(argv[4], "new tokens"), argv[5],
argc >= 7 ? argv[6] : "", argc == 8 ? argv[7] : "");
}
if (command == "serve" && (argc == 3 || argc == 4)) {
const std::size_t port = argc == 4 ? ParseSize(argv[3], "port") : 9091;
if (port == 0 || port > std::numeric_limits<std::uint16_t>::max()) {
throw std::runtime_error("port must be in [1, 65535]");
}
if (!ling3::RknnBackendAvailable()) {
throw std::runtime_error("serve requires a build with RKNN support");
}
ConfigureInferenceProcess();
return ling3::RunHttpService(argv[2], static_cast<std::uint16_t>(port));
}
PrintUsage(argv[0]);
return 1;
} catch (const std::exception & error) {
std::cerr << "error: " << error.what() << '\n';
return 1;
}
}