Download src/main.cpp from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 78.6 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/src/main.cpp
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/src/main.cpp
-
curl -L -o main.cpp https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/src/main.cpp
78.6 kB
| namespace { | |
| std::size_t ParseSize(std::string_view text, const char * label) { | |
| std::size_t value = 0; | |
| const auto [end, error] = std::from_chars(text.data(), text.data() + text.size(), value); | |
| if (error != std::errc {} || end != text.data() + text.size()) { | |
| throw std::runtime_error(std::string("invalid ") + label + ": " + std::string(text)); | |
| } | |
| return value; | |
| } | |
| void PrintUsage(const char * program) { | |
| std::cerr << "Usage:\n" | |
| << " " << program << " inspect MODEL.l3r\n" | |
| << " " << program << " plan [MAX_CONTEXT]\n" | |
| << " " << program << " rknn-smoke [ITERATIONS]\n" | |
| << " " << program << " tokenizer-smoke MODEL.l3r\n" | |
| << " " << program << " gdn-smoke MODEL.l3r [ITERATIONS]\n" | |
| << " " << program << " gdn-batch16-smoke MODEL.l3r [ITERATIONS]\n" | |
| << " " << program << " gdn-cpu-state-check MODEL.l3r\n" | |
| << " " << program << " gdn-reference-check MODEL.l3r\n" | |
| << " " << program << " layer0-smoke MODEL.l3r TOKEN REF.f32 [ITERATIONS]\n" | |
| << " " << program << " linear-smoke MODEL.l3r TENSOR_BASE [ITERATIONS]\n" | |
| << " " << program | |
| << " linear-batch-smoke MODEL.l3r TENSOR_BASE ROWS [ITERATIONS] [WEIGHT_BASE]\n" | |
| << " " << program << " prefill32-smoke MODEL.l3r\n" | |
| << " " << program << " logits-smoke MODEL.l3r TOKEN [REF.f32]\n" | |
| << " " << program << " generate MODEL.l3r USER_TEXT [MAX_NEW_TOKENS]\n" | |
| << " " << program << " benchmark MODEL.l3r SEQLEN NEW_TOKENS [ITERATIONS]\n" | |
| << " " << program << " mtp-benchmark MODEL.l3r SEQLEN NEW_TOKENS ITERATIONS CONTEXT [CASES.txt]\n" | |
| << " " << program << " sequence-check MODEL.l3r SEQLEN NEW_TOKENS OUT_PREFIX [REFERENCE_PREFIX [CORPUS.txt]]\n" | |
| << " " << program << " serve MODEL.l3r [PORT]\n"; | |
| } | |
| template <typename T> | |
| std::span<const T> TensorSpan(const ling3::TensorView & tensor); | |
| void PinToA76Cores() { | |
| cpu_set_t cores; | |
| CPU_ZERO(&cores); | |
| for (int core = 4; core <= 7; ++core) CPU_SET(core, &cores); | |
| if (sched_setaffinity(0, sizeof(cores), &cores) != 0) { | |
| std::cerr << "warning: cannot pin inference threads to CPU cores 4-7: " | |
| << std::strerror(errno) << '\n'; | |
| } | |
| } | |
| void RaiseFileDescriptorLimit() { | |
| rlimit limit {}; | |
| if (getrlimit(RLIMIT_NOFILE, &limit) != 0) { | |
| throw std::runtime_error( | |
| std::string("getrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno)); | |
| } | |
| constexpr rlim_t requested = 262144; | |
| const rlim_t target = std::min(requested, limit.rlim_max); | |
| if (limit.rlim_cur >= target) return; | |
| limit.rlim_cur = target; | |
| if (setrlimit(RLIMIT_NOFILE, &limit) != 0) { | |
| throw std::runtime_error( | |
| std::string("setrlimit(RLIMIT_NOFILE) failed: ") + std::strerror(errno)); | |
| } | |
| } | |
| void ConfigureInferenceProcess() { | |
| RaiseFileDescriptorLimit(); | |
| PinToA76Cores(); | |
| } | |
| ling3::Tokenizer PackageTokenizer(const ling3::ModelPackage & package) { | |
| const auto & asset = package.tensor("tokenizer"); | |
| if (asset.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kTokenizer)) { | |
| throw std::runtime_error("tokenizer tensor has the wrong role"); | |
| } | |
| return ling3::Tokenizer(TensorSpan<std::byte>(asset)); | |
| } | |
| std::string Utf8(std::u8string_view text) { | |
| return {reinterpret_cast<const char *>(text.data()), text.size()}; | |
| } | |
| int RunTokenizerSmoke(const std::string & package_path) { | |
| const ling3::ModelPackage package(package_path); | |
| const auto & asset = package.tensor("tokenizer"); | |
| if (asset.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kTokenizer)) { | |
| throw std::runtime_error("tokenizer tensor has the wrong role"); | |
| } | |
| const ling3::Tokenizer tokenizer(TensorSpan<std::byte>(asset)); | |
| struct Case { | |
| std::string text; | |
| std::vector<std::uint32_t> expected; | |
| }; | |
| const std::vector<Case> cases = { | |
| {Utf8(u8"你好,世界!"), {34355, 44291, 859}}, | |
| {"Hello world!", {14455, 1931, 0}}, | |
| {"I'm testing 123.", {40, 3180, 7750, 220, 16, 17, 18, 13}}, | |
| {Utf8(u8"ABC e\u0301\n第二行"), {83140, 270, 95, 93086, 18878, 198, 2378, 685}}, | |
| {Utf8(u8"<think>想一下</think>"), {156903, 122843, 156904}}, | |
| { | |
| Utf8(u8"<|startoftext|>user\n你好<|role_end|>\n<|startoftext|>assistant\n"), | |
| {156891, 3840, 198, 34355, 156895, 198, 156891, 598, 10450, 198}, | |
| }, | |
| }; | |
| for (std::size_t index = 0; index < cases.size(); ++index) { | |
| const auto actual = tokenizer.Encode(cases[index].text); | |
| if (actual != cases[index].expected) { | |
| std::cerr << "tokenizer mismatch in case " << index << "\nexpected:"; | |
| for (auto token : cases[index].expected) std::cerr << ' ' << token; | |
| std::cerr << "\nactual:"; | |
| for (auto token : actual) std::cerr << ' ' << token; | |
| std::cerr << '\n'; | |
| return 8; | |
| } | |
| if (tokenizer.Decode(actual) != cases[index].text && index != 3) { | |
| std::cerr << "tokenizer round-trip mismatch in case " << index << '\n'; | |
| return 9; | |
| } | |
| } | |
| if (!tokenizer.uses_official_unicode_rules()) { | |
| std::cerr << "token IDs matched, but this build lacks official ICU Unicode rules\n"; | |
| return 10; | |
| } | |
| std::cout << "tokenizer smoke passed (" << cases.size() | |
| << " official vectors, ICU rules enabled)\n"; | |
| return 0; | |
| } | |
| int RunGdnSmoke(const std::string & package_path, std::size_t iterations) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("gdn-smoke requires a build with RKNN support"); | |
| } | |
| if (iterations == 0) iterations = 1; | |
| const ling3::ModelPackage package(package_path); | |
| const auto & heads6 = package.tensor("rknn.gdn.heads6"); | |
| const auto & heads5 = package.tensor("rknn.gdn.heads5"); | |
| if (heads6.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kRknnIsland) || | |
| heads5.entry->role != static_cast<std::uint32_t>(ling3::TensorRole::kRknnIsland)) { | |
| throw std::runtime_error("GDN tensors do not contain RKNN islands"); | |
| } | |
| constexpr int heads = 16; | |
| constexpr int dimension = 128; | |
| constexpr int elements = heads * dimension; | |
| std::vector<float> query(elements), key(elements), value(elements), decay(elements); | |
| std::vector<float> beta(heads), output(elements), reference(elements); | |
| for (int head = 0; head < heads; ++head) { | |
| float q_norm = 1.0e-6F; | |
| float k_norm = 1.0e-6F; | |
| for (int index = 0; index < dimension; ++index) { | |
| const int offset = head * dimension + index; | |
| query[offset] = std::sin(static_cast<float>(offset + 1) * 0.013F); | |
| key[offset] = std::cos(static_cast<float>(offset + 3) * 0.017F); | |
| value[offset] = 0.2F * std::sin(static_cast<float>(offset + 5) * 0.021F); | |
| decay[offset] = -2.0F - 0.5F * std::cos(static_cast<float>(offset) * 0.009F); | |
| q_norm += query[offset] * query[offset]; | |
| k_norm += key[offset] * key[offset]; | |
| } | |
| q_norm = std::sqrt(q_norm); | |
| k_norm = std::sqrt(k_norm); | |
| for (int index = 0; index < dimension; ++index) { | |
| query[head * dimension + index] /= q_norm; | |
| key[head * dimension + index] /= k_norm; | |
| } | |
| beta[head] = 1.0F / (1.0F + std::exp(-0.2F * std::sin(static_cast<float>(head)))); | |
| } | |
| const auto initialize_begin = std::chrono::steady_clock::now(); | |
| ling3::GdnStep gdn(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5)); | |
| const auto initialize_end = std::chrono::steady_clock::now(); | |
| const double initialize_ms = std::chrono::duration<double, std::milli>( | |
| initialize_end - initialize_begin).count(); | |
| gdn.Run(query, key, value, decay, beta, output); | |
| const float query_scale = 1.0F / std::sqrt(static_cast<float>(dimension)); | |
| for (int head = 0; head < heads; ++head) { | |
| float key_dot_query = 0.0F; | |
| for (int index = 0; index < dimension; ++index) { | |
| key_dot_query += key[head * dimension + index] * | |
| query[head * dimension + index] * query_scale; | |
| } | |
| for (int index = 0; index < dimension; ++index) { | |
| reference[head * dimension + index] = | |
| beta[head] * value[head * dimension + index] * key_dot_query; | |
| } | |
| } | |
| double dot = 0.0; | |
| double norm_output = 0.0; | |
| double norm_reference = 0.0; | |
| double max_error = 0.0; | |
| double mean_error = 0.0; | |
| for (int index = 0; index < elements; ++index) { | |
| const double error = std::abs(static_cast<double>(output[index]) - reference[index]); | |
| max_error = std::max(max_error, error); | |
| mean_error += error; | |
| dot += static_cast<double>(output[index]) * reference[index]; | |
| norm_output += static_cast<double>(output[index]) * output[index]; | |
| norm_reference += static_cast<double>(reference[index]) * reference[index]; | |
| } | |
| mean_error /= elements; | |
| const double cosine = dot / std::sqrt(norm_output * norm_reference); | |
| gdn.Reset(); | |
| for (int index = 0; index < 5; ++index) { | |
| gdn.Run(query, key, value, decay, beta, output); | |
| } | |
| ling3::GdnRunTimings mean; | |
| for (std::size_t index = 0; index < iterations; ++index) { | |
| const auto sample = gdn.Run(query, key, value, decay, beta, output); | |
| mean.stage_ms += sample.stage_ms; | |
| mean.npu_ms += sample.npu_ms; | |
| mean.collect_ms += sample.collect_ms; | |
| mean.total_ms += sample.total_ms; | |
| } | |
| const double divisor = static_cast<double>(iterations); | |
| mean.stage_ms /= divisor; | |
| mean.npu_ms /= divisor; | |
| mean.collect_ms /= divisor; | |
| mean.total_ms /= divisor; | |
| std::cout << std::fixed << std::setprecision(6) | |
| << "heads=6+5+5\n" | |
| << "cores=0,1,2\n" | |
| << "initialization_ms=" << initialize_ms << '\n' | |
| << "state_bytes=" << gdn.state_bytes() << '\n' | |
| << "first_step_cosine=" << cosine << '\n' | |
| << "first_step_mean_abs_error=" << mean_error << '\n' | |
| << "first_step_max_abs_error=" << max_error << '\n' | |
| << "iterations=" << iterations << '\n' | |
| << "stage_mean_ms=" << mean.stage_ms << '\n' | |
| << "npu_mean_ms=" << mean.npu_ms << '\n' | |
| << "collect_mean_ms=" << mean.collect_ms << '\n' | |
| << "total_mean_ms=" << mean.total_ms << '\n'; | |
| return cosine >= 0.999 && max_error <= 0.005 ? 0 : 11; | |
| } | |
| void FillGdnToken( | |
| int token, | |
| std::span<float> query, | |
| std::span<float> key, | |
| std::span<float> value, | |
| std::span<float> decay, | |
| std::span<float> beta) { | |
| constexpr int heads = 16; | |
| constexpr int dimension = 128; | |
| for (int head = 0; head < heads; ++head) { | |
| float q_norm = 1.0e-6F; | |
| float k_norm = 1.0e-6F; | |
| for (int index = 0; index < dimension; ++index) { | |
| const int offset = head * dimension + index; | |
| query[offset] = std::sin(static_cast<float>(offset + 13 * token + 1) * 0.013F); | |
| key[offset] = std::cos(static_cast<float>(offset + 7 * token + 3) * 0.017F); | |
| value[offset] = 0.2F * | |
| std::sin(static_cast<float>(offset + 5 * token + 5) * 0.021F); | |
| decay[offset] = -2.0F - 0.5F * | |
| std::cos(static_cast<float>(offset + 3 * token) * 0.009F); | |
| q_norm += query[offset] * query[offset]; | |
| k_norm += key[offset] * key[offset]; | |
| } | |
| q_norm = std::sqrt(q_norm); | |
| k_norm = std::sqrt(k_norm); | |
| for (int index = 0; index < dimension; ++index) { | |
| query[head * dimension + index] /= q_norm; | |
| key[head * dimension + index] /= k_norm; | |
| } | |
| beta[head] = 1.0F / | |
| (1.0F + std::exp(-0.2F * std::sin(static_cast<float>(head + token)))); | |
| } | |
| } | |
| int RunGdnBatch16Smoke(const std::string & package_path, std::size_t iterations) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("gdn-batch16-smoke requires a build with RKNN support"); | |
| } | |
| if (iterations == 0) iterations = 1; | |
| const ling3::ModelPackage package(package_path); | |
| const auto & heads6 = package.tensor("rknn.gdn.heads6"); | |
| const auto & heads5 = package.tensor("rknn.gdn.heads5"); | |
| constexpr std::size_t width = 16 * 128; | |
| constexpr std::size_t rows = 16; | |
| ling3::GdnStep sequential(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5)); | |
| ling3::GdnStep batch(TensorSpan<std::byte>(heads6), TensorSpan<std::byte>(heads5)); | |
| if (!batch.has_batch16()) { | |
| throw std::runtime_error("set LING3_GDN_PREFILL_DIR to the 16-token RKNN models"); | |
| } | |
| std::vector<float> query(rows * width), key(rows * width), value(rows * width); | |
| std::vector<float> decay(rows * width), beta(rows * 16); | |
| std::vector<float> sequential_output(rows * width), batch_output(rows * width); | |
| std::vector<float> scratch(width), scratch_key(width), scratch_value(width), scratch_decay(width); | |
| std::vector<float> scratch_beta(16), sequential_tail(width), batch_tail(width); | |
| for (int token = 0; token < 3; ++token) { | |
| FillGdnToken(token, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta); | |
| sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail); | |
| batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail); | |
| } | |
| for (std::size_t row = 0; row < rows; ++row) { | |
| FillGdnToken( | |
| static_cast<int>(row + 3), | |
| std::span<float>(query).subspan(row * width, width), | |
| std::span<float>(key).subspan(row * width, width), | |
| std::span<float>(value).subspan(row * width, width), | |
| std::span<float>(decay).subspan(row * width, width), | |
| std::span<float>(beta).subspan(row * 16, 16)); | |
| sequential.Run( | |
| std::span<const float>(query).subspan(row * width, width), | |
| std::span<const float>(key).subspan(row * width, width), | |
| std::span<const float>(value).subspan(row * width, width), | |
| std::span<const float>(decay).subspan(row * width, width), | |
| std::span<const float>(beta).subspan(row * 16, 16), | |
| std::span<float>(sequential_output).subspan(row * width, width)); | |
| } | |
| const auto first_timing = batch.RunBatch16( | |
| query, key, value, decay, beta, batch_output); | |
| FillGdnToken(19, scratch, scratch_key, scratch_value, scratch_decay, scratch_beta); | |
| sequential.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, sequential_tail); | |
| batch.Run(scratch, scratch_key, scratch_value, scratch_decay, scratch_beta, batch_tail); | |
| auto compare = [](std::span<const float> actual, std::span<const float> expected) { | |
| std::array<double, 3> result {}; | |
| double actual_norm = 0.0; | |
| double expected_norm = 0.0; | |
| for (std::size_t index = 0; index < actual.size(); ++index) { | |
| const double a = actual[index]; | |
| const double b = expected[index]; | |
| result[0] += a * b; | |
| actual_norm += a * a; | |
| expected_norm += b * b; | |
| const double error = std::abs(a - b); | |
| result[1] += error; | |
| result[2] = std::max(result[2], error); | |
| } | |
| result[0] /= std::sqrt(actual_norm * expected_norm); | |
| result[1] /= static_cast<double>(actual.size()); | |
| return result; | |
| }; | |
| const auto output_error = compare(batch_output, sequential_output); | |
| const auto state_error = compare(batch_tail, sequential_tail); | |
| ling3::GdnRunTimings mean = first_timing; | |
| for (std::size_t iteration = 1; iteration < iterations; ++iteration) { | |
| const auto sample = batch.RunBatch16(query, key, value, decay, beta, batch_output); | |
| mean.stage_ms += sample.stage_ms; | |
| mean.npu_ms += sample.npu_ms; | |
| mean.collect_ms += sample.collect_ms; | |
| mean.total_ms += sample.total_ms; | |
| } | |
| const double divisor = static_cast<double>(iterations); | |
| mean.stage_ms /= divisor; | |
| mean.npu_ms /= divisor; | |
| mean.collect_ms /= divisor; | |
| mean.total_ms /= divisor; | |
| std::cout << std::fixed << std::setprecision(8) | |
| << "rows=16\n" | |
| << "history_tokens=3\n" | |
| << "output_cosine=" << output_error[0] << '\n' | |
| << "output_mean_abs_error=" << output_error[1] << '\n' | |
| << "output_max_abs_error=" << output_error[2] << '\n' | |
| << "tail_state_cosine=" << state_error[0] << '\n' | |
| << "tail_mean_abs_error=" << state_error[1] << '\n' | |
| << "tail_max_abs_error=" << state_error[2] << '\n' | |
| << "stage_mean_ms=" << mean.stage_ms << '\n' | |
| << "npu_mean_ms=" << mean.npu_ms << '\n' | |
| << "collect_mean_ms=" << mean.collect_ms << '\n' | |
| << "total_mean_ms=" << mean.total_ms << '\n'; | |
| return output_error[0] >= 0.99999 && state_error[0] >= 0.99999 ? 0 : 14; | |
| } | |
| std::vector<float> ReadFloatFile(const std::string & path, std::size_t count) { | |
| std::ifstream stream(path, std::ios::binary | std::ios::ate); | |
| if (!stream) throw std::runtime_error("cannot open " + path); | |
| const auto bytes = static_cast<std::size_t>(stream.tellg()); | |
| if (bytes != count * sizeof(float)) { | |
| throw std::runtime_error(path + " has an incompatible byte count"); | |
| } | |
| stream.seekg(0); | |
| std::vector<float> result(count); | |
| stream.read(reinterpret_cast<char *>(result.data()), static_cast<std::streamsize>(bytes)); | |
| if (!stream) throw std::runtime_error("cannot read " + path); | |
| return result; | |
| } | |
| int RunLayer0Smoke( | |
| const std::string & package_path, | |
| std::uint32_t token, | |
| const std::string & reference_path, | |
| std::size_t iterations) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("layer0-smoke requires a build with RKNN support"); | |
| } | |
| if (iterations == 0) iterations = 1; | |
| const ling3::ModelPackage package(package_path); | |
| const auto initialization_begin = std::chrono::steady_clock::now(); | |
| ling3::Layer0 layer(package); | |
| const auto initialization_end = std::chrono::steady_clock::now(); | |
| const double initialization_ms = std::chrono::duration<double, std::milli>( | |
| initialization_end - initialization_begin).count(); | |
| std::vector<float> output(1536); | |
| layer.Reset(); | |
| layer.DecodeToken(token, output); | |
| const auto reference = ReadFloatFile(reference_path, output.size()); | |
| double dot = 0.0, output_norm = 0.0, reference_norm = 0.0; | |
| double mean_error = 0.0, maximum_error = 0.0; | |
| for (std::size_t index = 0; index < output.size(); ++index) { | |
| const double error = std::abs(static_cast<double>(output[index]) - reference[index]); | |
| mean_error += error; | |
| maximum_error = std::max(maximum_error, error); | |
| dot += static_cast<double>(output[index]) * reference[index]; | |
| output_norm += static_cast<double>(output[index]) * output[index]; | |
| reference_norm += static_cast<double>(reference[index]) * reference[index]; | |
| } | |
| mean_error /= output.size(); | |
| const double cosine = dot / std::sqrt(output_norm * reference_norm); | |
| layer.Reset(); | |
| for (int index = 0; index < 3; ++index) layer.DecodeToken(token, output); | |
| ling3::Layer0Timings mean; | |
| for (std::size_t index = 0; index < iterations; ++index) { | |
| const auto sample = layer.DecodeToken(token, output); | |
| mean.attention_ms += sample.attention_ms; | |
| mean.dense_ffn_ms += sample.dense_ffn_ms; | |
| mean.total_ms += sample.total_ms; | |
| } | |
| const double divisor = static_cast<double>(iterations); | |
| mean.attention_ms /= divisor; | |
| mean.dense_ffn_ms /= divisor; | |
| mean.total_ms /= divisor; | |
| std::cout << std::fixed << std::setprecision(6) | |
| << "token=" << token << '\n' | |
| << "initialization_ms=" << initialization_ms << '\n' | |
| << "cosine_vs_official_bf16=" << cosine << '\n' | |
| << "mean_abs_error=" << mean_error << '\n' | |
| << "max_abs_error=" << maximum_error << '\n' | |
| << "iterations=" << iterations << '\n' | |
| << "attention_mean_ms=" << mean.attention_ms << '\n' | |
| << "dense_ffn_mean_ms=" << mean.dense_ffn_ms << '\n' | |
| << "total_mean_ms=" << mean.total_ms << '\n'; | |
| return cosine >= 0.98 ? 0 : 12; | |
| } | |
| template <typename T> | |
| std::span<const T> TensorSpan(const ling3::TensorView & tensor) { | |
| if (tensor.entry->data_bytes % sizeof(T) != 0) { | |
| throw std::runtime_error(std::string(tensor.name) + " has an invalid byte count"); | |
| } | |
| return { | |
| reinterpret_cast<const T *>(tensor.data), | |
| static_cast<std::size_t>(tensor.entry->data_bytes / sizeof(T)), | |
| }; | |
| } | |
| int RunLinearSmoke( | |
| const std::string & package_path, | |
| const std::string & base, | |
| std::size_t iterations) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("linear-smoke requires a build with RKNN support"); | |
| } | |
| if (iterations == 0) iterations = 1; | |
| const ling3::ModelPackage package(package_path); | |
| ling3::ValidateLing3Tiny(package.header()); | |
| const auto & weight = package.tensor(base + ".weight"); | |
| const auto & scale = package.tensor(base + ".scales"); | |
| const auto & correction = package.tensor(base + ".correction"); | |
| if (weight.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kInt4Low) || | |
| weight.entry->layout != static_cast<std::uint32_t>(ling3::TensorLayout::kPackedInt4Low) || | |
| weight.entry->rank != 2 || scale.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kFloat32) || | |
| correction.entry->dtype != static_cast<std::uint32_t>(ling3::DataType::kInt32)) { | |
| throw std::runtime_error("linear-smoke tensors have incompatible metadata"); | |
| } | |
| const int k = static_cast<int>(weight.entry->dims[0]); | |
| const int n = static_cast<int>(weight.entry->dims[1]); | |
| const int k_splits = static_cast<int>(weight.entry->flags); | |
| const auto weights = TensorSpan<std::byte>(weight); | |
| const auto scales = TensorSpan<float>(scale); | |
| const auto corrections = TensorSpan<std::int32_t>(correction); | |
| std::vector<float> input(k); | |
| for (int index = 0; index < k; ++index) { | |
| input[index] = 0.7F * std::sin(static_cast<float>(index) * 0.03125F) + | |
| 0.2F * std::cos(static_cast<float>(index) * 0.0078125F); | |
| } | |
| std::vector<float> output(n); | |
| const auto initialize_begin = std::chrono::steady_clock::now(); | |
| ling3::DynamicW4Linear linear( | |
| {k, n, k_splits, {0, 1, 2}}, weights, scales, corrections); | |
| const auto initialize_end = std::chrono::steady_clock::now(); | |
| const auto initialize_ms = std::chrono::duration<double, std::milli>( | |
| initialize_end - initialize_begin).count(); | |
| linear.Run(input, output); | |
| std::vector<ling3::W4RunTimings> samples; | |
| samples.reserve(iterations); | |
| for (std::size_t iteration = 0; iteration < iterations; ++iteration) { | |
| samples.push_back(linear.Run(input, output)); | |
| } | |
| std::vector<std::int8_t> input_codes(k); | |
| const auto quantization = ling3::QuantizeSymmetricInt8(input, input_codes); | |
| std::vector<std::int32_t> reference_accumulator(n); | |
| ling3::ReferenceW4Linear(input_codes, weights, n, reference_accumulator); | |
| std::vector<float> reference(n); | |
| ling3::DequantizePerChannel(reference_accumulator, quantization.scale, scales, reference); | |
| double maximum_error = 0.0; | |
| double mean_error = 0.0; | |
| for (int index = 0; index < n; ++index) { | |
| const double error = std::abs(static_cast<double>(output[index]) - reference[index]); | |
| maximum_error = std::max(maximum_error, error); | |
| mean_error += error; | |
| } | |
| mean_error /= n; | |
| ling3::W4RunTimings mean; | |
| for (const auto & sample : samples) { | |
| mean.quantize_pack_ms += sample.quantize_pack_ms; | |
| mean.input_sync_ms += sample.input_sync_ms; | |
| mean.npu_ms += sample.npu_ms; | |
| mean.gather_ms += sample.gather_ms; | |
| mean.total_ms += sample.total_ms; | |
| } | |
| const double divisor = static_cast<double>(samples.size()); | |
| mean.quantize_pack_ms /= divisor; | |
| mean.input_sync_ms /= divisor; | |
| mean.npu_ms /= divisor; | |
| mean.gather_ms /= divisor; | |
| mean.total_ms /= divisor; | |
| std::cout << std::fixed << std::setprecision(6) | |
| << "tensor=" << base << '\n' | |
| << "k=" << k << '\n' | |
| << "n=" << n << '\n' | |
| << "k_splits=" << k_splits << '\n' | |
| << "cores=0,1,2\n" | |
| << "initialization_ms=" << initialize_ms << '\n' | |
| << "resident_weight_bytes=" << linear.resident_weight_bytes() << '\n' | |
| << "iterations=" << iterations << '\n' | |
| << "quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n' | |
| << "input_sync_mean_ms=" << mean.input_sync_ms << '\n' | |
| << "npu_mean_ms=" << mean.npu_ms << '\n' | |
| << "gather_mean_ms=" << mean.gather_ms << '\n' | |
| << "total_mean_ms=" << mean.total_ms << '\n' | |
| << "reference_mean_abs_error=" << mean_error << '\n' | |
| << "reference_max_abs_error=" << maximum_error << '\n'; | |
| return maximum_error <= 1.0e-5 ? 0 : 7; | |
| } | |
| int RunLinearBatchSmoke( | |
| const std::string & package_path, | |
| const std::string & base, | |
| std::size_t rows, | |
| std::size_t iterations, | |
| const std::string & weight_base) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("linear-batch-smoke requires a build with RKNN support"); | |
| } | |
| if (rows < 1 || rows > 128) throw std::runtime_error("rows must be in [1, 128]"); | |
| if (iterations == 0) iterations = 1; | |
| const ling3::ModelPackage package(package_path); | |
| ling3::ValidateLing3Tiny(package.header()); | |
| const auto & weight = package.tensor(base + ".weight"); | |
| const auto & scale = package.tensor(base + ".scales"); | |
| const auto & correction = package.tensor(base + ".correction"); | |
| const int k = static_cast<int>(weight.entry->dims[0]); | |
| const int n = static_cast<int>(weight.entry->dims[1]); | |
| const int k_splits = static_cast<int>(weight.entry->flags); | |
| const std::vector<int> cores = std::getenv("LING3_LINEAR_SINGLE_CORE") == nullptr | |
| ? std::vector<int> {0, 1, 2} | |
| : std::vector<int> {0}; | |
| ling3::DynamicW4Linear linear( | |
| {k, n, k_splits, cores}, TensorSpan<std::byte>(weight), | |
| TensorSpan<float>(scale), TensorSpan<std::int32_t>(correction)); | |
| std::unique_ptr<ling3::DynamicW4Linear> alternate_weights; | |
| if (!weight_base.empty() && weight_base != base) { | |
| const auto & alternate_weight = package.tensor(weight_base + ".weight"); | |
| const auto & alternate_scale = package.tensor(weight_base + ".scales"); | |
| const auto & alternate_correction = package.tensor(weight_base + ".correction"); | |
| if (static_cast<int>(alternate_weight.entry->dims[0]) != k || | |
| static_cast<int>(alternate_weight.entry->dims[1]) != n || | |
| static_cast<int>(alternate_weight.entry->flags) != k_splits) { | |
| throw std::runtime_error("alternate weight tensor has an incompatible shape"); | |
| } | |
| alternate_weights = std::make_unique<ling3::DynamicW4Linear>( | |
| ling3::W4LinearConfig {k, n, k_splits, cores}, | |
| TensorSpan<std::byte>(alternate_weight), TensorSpan<float>(alternate_scale), | |
| TensorSpan<std::int32_t>(alternate_correction)); | |
| } | |
| auto & reference_linear = alternate_weights ? *alternate_weights : linear; | |
| std::vector<float> input(rows * static_cast<std::size_t>(k)); | |
| for (std::size_t row = 0; row < rows; ++row) { | |
| for (int column = 0; column < k; ++column) { | |
| input[row * k + column] = | |
| 0.7F * std::sin(static_cast<float>(column + 7 * row) * 0.03125F) + | |
| 0.2F * std::cos(static_cast<float>(3 * column + row) * 0.0078125F); | |
| } | |
| } | |
| std::vector<float> batch_output(rows * static_cast<std::size_t>(n)); | |
| std::vector<float> sequential_output(rows * static_cast<std::size_t>(n)); | |
| linear.RunBatchWithWeights(input, rows, reference_linear, batch_output); | |
| for (std::size_t row = 0; row < rows; ++row) { | |
| reference_linear.Run( | |
| std::span<const float>(input).subspan(row * k, k), | |
| std::span<float>(sequential_output).subspan(row * n, n)); | |
| } | |
| double maximum_error = 0.0; | |
| double mean_error = 0.0; | |
| for (std::size_t index = 0; index < batch_output.size(); ++index) { | |
| const double error = std::abs( | |
| static_cast<double>(batch_output[index]) - sequential_output[index]); | |
| maximum_error = std::max(maximum_error, error); | |
| mean_error += error; | |
| } | |
| mean_error /= static_cast<double>(batch_output.size()); | |
| // Check shared quantized inputs with reordered/repeated rows, a smaller | |
| // tail, and a return to the original capacity in the same context. | |
| double indexed_maximum_error = 0.0; | |
| if (std::getenv("LING3_PREFILL_W4A4") == nullptr) { | |
| std::vector<std::int8_t> quantized(input.size()); | |
| std::vector<float> input_scales(rows); | |
| std::vector<std::size_t> indices(rows); | |
| for (std::size_t row = 0; row < rows; ++row) { | |
| input_scales[row] = ling3::QuantizeSymmetricInt8( | |
| std::span<const float>(input).subspan(row * k, k), | |
| std::span<std::int8_t>(quantized).subspan(row * k, k)).scale; | |
| indices[row] = rows - 1 - row / 2; | |
| } | |
| for (const std::size_t count : {rows, std::max<std::size_t>(1, rows / 2), rows}) { | |
| linear.RunBatchQuantizedRows( | |
| quantized, input_scales, | |
| std::span<const std::size_t>(indices).first(count), reference_linear, | |
| std::span<float>(batch_output).first(count * n)); | |
| for (std::size_t row = 0; row < count; ++row) { | |
| for (int column = 0; column < n; ++column) { | |
| const double actual = batch_output[row * n + column]; | |
| if (!std::isfinite(actual)) { | |
| throw std::runtime_error("indexed W4 batch produced a non-finite output"); | |
| } | |
| indexed_maximum_error = std::max(indexed_maximum_error, std::abs( | |
| actual - sequential_output[indices[row] * n + column])); | |
| } | |
| } | |
| } | |
| } | |
| ling3::W4RunTimings mean; | |
| const auto sequential_begin = std::chrono::steady_clock::now(); | |
| for (std::size_t iteration = 0; iteration < iterations; ++iteration) { | |
| for (std::size_t row = 0; row < rows; ++row) { | |
| reference_linear.Run( | |
| std::span<const float>(input).subspan(row * k, k), | |
| std::span<float>(sequential_output).subspan( | |
| row * n, n)); | |
| } | |
| } | |
| const double sequential_ms = std::chrono::duration<double, std::milli>( | |
| std::chrono::steady_clock::now() - sequential_begin).count() / | |
| static_cast<double>(iterations); | |
| for (std::size_t iteration = 0; iteration < iterations; ++iteration) { | |
| const auto sample = linear.RunBatchWithWeights(input, rows, reference_linear, batch_output); | |
| mean.quantize_pack_ms += sample.quantize_pack_ms; | |
| mean.input_sync_ms += sample.input_sync_ms; | |
| mean.npu_ms += sample.npu_ms; | |
| mean.gather_ms += sample.gather_ms; | |
| mean.total_ms += sample.total_ms; | |
| } | |
| const double divisor = static_cast<double>(iterations); | |
| mean.quantize_pack_ms /= divisor; | |
| mean.input_sync_ms /= divisor; | |
| mean.npu_ms /= divisor; | |
| mean.gather_ms /= divisor; | |
| mean.total_ms /= divisor; | |
| std::cout << std::fixed << std::setprecision(6) | |
| << "tensor=" << base << '\n' | |
| << "weight_tensor=" << (weight_base.empty() ? base : weight_base) << '\n' | |
| << "rows=" << rows << '\n' | |
| << "cores=" << (cores.size() == 1 ? "0" : "0,1,2") << '\n' | |
| << "sequential_mean_ms=" << sequential_ms << '\n' | |
| << "batch_quantize_pack_mean_ms=" << mean.quantize_pack_ms << '\n' | |
| << "batch_input_sync_mean_ms=" << mean.input_sync_ms << '\n' | |
| << "batch_npu_mean_ms=" << mean.npu_ms << '\n' | |
| << "batch_gather_mean_ms=" << mean.gather_ms << '\n' | |
| << "batch_total_mean_ms=" << mean.total_ms << '\n' | |
| << "speedup=" << sequential_ms / mean.total_ms << '\n' | |
| << "sequential_mean_abs_error=" << mean_error << '\n' | |
| << "sequential_max_abs_error=" << maximum_error << '\n' | |
| << "indexed_max_abs_error=" << indexed_maximum_error << '\n'; | |
| return maximum_error <= 1.0e-5 && indexed_maximum_error <= 1.0e-5 ? 0 : 13; | |
| } | |
| std::array<double, 3> CompareVectors( | |
| std::span<const float> actual, | |
| std::span<const float> expected) { | |
| if (actual.size() != expected.size() || actual.empty()) { | |
| throw std::invalid_argument("comparison vectors have incompatible sizes"); | |
| } | |
| double dot = 0.0; | |
| double actual_norm = 0.0; | |
| double expected_norm = 0.0; | |
| double mean_error = 0.0; | |
| double maximum_error = 0.0; | |
| for (std::size_t index = 0; index < actual.size(); ++index) { | |
| const double a = actual[index]; | |
| const double b = expected[index]; | |
| const double error = std::abs(a - b); | |
| dot += a * b; | |
| actual_norm += a * a; | |
| expected_norm += b * b; | |
| mean_error += error; | |
| maximum_error = std::max(maximum_error, error); | |
| } | |
| return { | |
| dot / std::sqrt(actual_norm * expected_norm), | |
| mean_error / static_cast<double>(actual.size()), | |
| maximum_error, | |
| }; | |
| } | |
| int RunGdnReferenceCheck(const std::string & path) { | |
| if (std::getenv("LING3_GDN_CPU_DECODE")) | |
| throw std::invalid_argument("unset LING3_GDN_CPU_DECODE for the NPU comparator"); | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(path); | |
| const auto h6 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads6")); | |
| const auto h5 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads5")); | |
| ling3::GdnStep npu(h6, h5), cpu(h6, h5); | |
| constexpr int heads = 16, dimension = 128, width = heads * dimension, steps = 256; | |
| std::vector<float> q(width), k(width), v(width), d(width), b(heads), a(width), c(width); | |
| std::vector<double> state(heads * dimension * dimension, 0.0); | |
| double npu_square_error = 0, cpu_square_error = 0, reference_square = 0; | |
| double npu_max = 0, cpu_max = 0; | |
| for (int t = 0; t < steps; ++t) { | |
| FillGdnToken(t, q, k, v, d, b); | |
| // Include near-unit retention as well as fast forgetting; a zero-state | |
| // first-token test alone cannot exercise accumulated recurrence error. | |
| for (int i = 0; i < width; ++i) | |
| d[i] = i % 3 == 0 ? -0.0001F : (i % 3 == 1 ? -0.05F : d[i]); | |
| npu.Run(q, k, v, d, b, a); | |
| cpu.RunBatchCpu(q, k, v, d, b, c); | |
| for (int head = 0; head < heads; ++head) { | |
| std::array<double, dimension> factor; | |
| for (int j = 0; j < dimension; ++j) factor[j] = std::exp(double(d[head * dimension + j])); | |
| for (int col = 0; col < dimension; ++col) { | |
| auto * row = state.data() + (head * dimension + col) * dimension; | |
| double predicted = 0; | |
| for (int j = 0; j < dimension; ++j) { | |
| row[j] *= factor[j]; | |
| predicted += row[j] * double(k[head * dimension + j]); | |
| } | |
| const double delta = double(b[head]) * (double(v[head * dimension + col]) - predicted); | |
| double result = 0; | |
| for (int j = 0; j < dimension; ++j) { | |
| row[j] += delta * double(k[head * dimension + j]); | |
| result += row[j] * double(q[head * dimension + j]) / std::sqrt(128.0); | |
| } | |
| const int index = head * dimension + col; | |
| if (!std::isfinite(a[index]) || !std::isfinite(c[index])) | |
| throw std::runtime_error("non-finite recurrence"); | |
| const double ea = a[index] - result, ec = c[index] - result; | |
| npu_square_error += ea * ea; | |
| cpu_square_error += ec * ec; | |
| reference_square += result * result; | |
| npu_max = std::max(npu_max, std::abs(ea)); | |
| cpu_max = std::max(cpu_max, std::abs(ec)); | |
| } | |
| } | |
| } | |
| std::cout << std::setprecision(12) << "steps=" << steps | |
| << " npu_relative_rms=" << std::sqrt(npu_square_error / reference_square) | |
| << " cpu_relative_rms=" << std::sqrt(cpu_square_error / reference_square) | |
| << " npu_max=" << npu_max << " cpu_max=" << cpu_max << '\n'; | |
| return cpu_square_error < npu_square_error && cpu_max < 1e-6 ? 0 : 2; | |
| } | |
| int RunGdnCpuStateCheck(const std::string & path) { | |
| if (std::getenv("LING3_GDN_CPU_DECODE")) | |
| throw std::invalid_argument("unset LING3_GDN_CPU_DECODE to test transitions to NPU"); | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(path); | |
| const auto h6 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads6")); | |
| const auto h5 = TensorSpan<std::byte>(package.tensor("rknn.gdn.heads5")); | |
| ling3::GdnStep batch(h6, h5), chunks(h6, h5); | |
| constexpr std::size_t rows = 97, width = 16 * 128; | |
| std::vector<float> q(rows * width), k(q.size()), v(q.size()), d(q.size()), b(rows * 16); | |
| std::vector<float> one(q.size()), many(q.size()), replay(q.size()); | |
| for (std::size_t t = 0; t < rows; ++t) { | |
| FillGdnToken(t, std::span<float>(q).subspan(t * width, width), | |
| std::span<float>(k).subspan(t * width, width), | |
| std::span<float>(v).subspan(t * width, width), | |
| std::span<float>(d).subspan(t * width, width), | |
| std::span<float>(b).subspan(t * 16, 16)); | |
| } | |
| batch.RunBatchCpu(q, k, v, d, b, one); | |
| std::size_t offset = 0; | |
| for (const std::size_t count : {1, 3, 16, 31, 46}) { | |
| chunks.RunBatchCpu(std::span<const float>(q).subspan(offset * width, count * width), | |
| std::span<const float>(k).subspan(offset * width, count * width), | |
| std::span<const float>(v).subspan(offset * width, count * width), | |
| std::span<const float>(d).subspan(offset * width, count * width), | |
| std::span<const float>(b).subspan(offset * 16, count * 16), | |
| std::span<float>(many).subspan(offset * width, count * width)); | |
| offset += count; | |
| } | |
| const auto error = CompareVectors(many, one); | |
| std::vector<float> tail1(width), tail2(width); | |
| batch.Run(std::span<const float>(q).last(width), std::span<const float>(k).last(width), | |
| std::span<const float>(v).last(width), std::span<const float>(d).last(width), | |
| std::span<const float>(b).last(16), tail1); | |
| chunks.Run(std::span<const float>(q).last(width), std::span<const float>(k).last(width), | |
| std::span<const float>(v).last(width), std::span<const float>(d).last(width), | |
| std::span<const float>(b).last(16), tail2); | |
| const auto transition = CompareVectors(tail1, tail2); | |
| // After a device step, return to CPU and verify the captured shadow state. | |
| batch.RunBatchCpu(q, k, v, d, b, replay); | |
| chunks.RunBatchCpu(q, k, v, d, b, many); | |
| const auto back = CompareVectors(replay, many); | |
| batch.Reset(); | |
| batch.RunBatchCpu(q, k, v, d, b, replay); | |
| const auto reset = CompareVectors(replay, one); | |
| std::cout << "chunk_max_error=" << error[2] << " cpu_to_npu_max_error=" << transition[2] | |
| << " npu_to_cpu_max_error=" << back[2] << " reset_max_error=" << reset[2] << '\n'; | |
| return error[2] == 0 && transition[2] == 0 && back[2] == 0 && reset[2] == 0 ? 0 : 2; | |
| } | |
| int RunPrefill32Smoke(const std::string & package_path) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("prefill32-smoke requires a build with RKNN support"); | |
| } | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(package_path); | |
| const auto tokenizer = PackageTokenizer(package); | |
| auto encoded = tokenizer.Encode(Utf8( | |
| u8"<role>SYSTEM</role>detailed thinking off<|role_end|>" | |
| u8"<role>HUMAN</role>请用中文简要说明今天的工作安排,并列出三个重点事项。" | |
| u8"<|role_end|><role>ASSISTANT</role>\n<think></think>")); | |
| if (encoded.size() < 32) throw std::runtime_error("prefill32 test prompt is too short"); | |
| encoded.resize(32); | |
| ling3::Decoder decoder(package); | |
| if (!decoder.has_dynamic_batch()) { | |
| throw std::runtime_error("set LING3_GDN_PREFILL_DIR for prefill32-smoke"); | |
| } | |
| std::vector<float> sequential_logits(package.header().vocab_size); | |
| std::vector<float> sequential_tail(package.header().vocab_size); | |
| std::vector<float> batch_logits(package.header().vocab_size); | |
| std::vector<float> batch_tail(package.header().vocab_size); | |
| double sequential_ms = 0.0; | |
| for (const auto token : encoded) { | |
| sequential_ms += decoder.Eval(token, sequential_logits).total_ms; | |
| } | |
| const auto next_token = static_cast<std::uint32_t>( | |
| std::max_element(sequential_logits.begin(), sequential_logits.end()) - | |
| sequential_logits.begin()); | |
| decoder.Eval(next_token, sequential_tail); | |
| decoder.Reset(); | |
| const auto cold_batch = decoder.EvalBatch32(encoded, batch_logits); | |
| decoder.Eval(next_token, batch_tail); | |
| const auto logits_error = CompareVectors(batch_logits, sequential_logits); | |
| const auto tail_error = CompareVectors(batch_tail, sequential_tail); | |
| decoder.Reset(); | |
| const auto warm_batch = decoder.EvalBatch32(encoded, batch_logits); | |
| std::cout << std::fixed << std::setprecision(8) | |
| << "tokens=32\n" | |
| << "sequential_ms=" << sequential_ms << '\n' | |
| << "cold_batch_ms=" << cold_batch.total_ms << '\n' | |
| << "warm_batch_ms=" << warm_batch.total_ms << '\n' | |
| << "speedup=" << sequential_ms / warm_batch.total_ms << '\n' | |
| << "logits_cosine=" << logits_error[0] << '\n' | |
| << "logits_mean_abs_error=" << logits_error[1] << '\n' | |
| << "logits_max_abs_error=" << logits_error[2] << '\n' | |
| << "tail_logits_cosine=" << tail_error[0] << '\n' | |
| << "tail_logits_mean_abs_error=" << tail_error[1] << '\n' | |
| << "tail_logits_max_abs_error=" << tail_error[2] << '\n' | |
| << "next_token=" << next_token << '\n'; | |
| return logits_error[0] >= 0.999 && tail_error[0] >= 0.999 ? 0 : 15; | |
| } | |
| std::vector<std::pair<std::uint32_t, float>> TopLogits( | |
| std::span<const float> logits, | |
| std::size_t count) { | |
| count = std::min(count, logits.size()); | |
| std::vector<std::uint32_t> indices(logits.size()); | |
| std::iota(indices.begin(), indices.end(), 0U); | |
| std::partial_sort( | |
| indices.begin(), indices.begin() + static_cast<std::ptrdiff_t>(count), indices.end(), | |
| [&logits](std::uint32_t left, std::uint32_t right) { | |
| return logits[left] > logits[right]; | |
| }); | |
| std::vector<std::pair<std::uint32_t, float>> result; | |
| result.reserve(count); | |
| for (std::size_t index = 0; index < count; ++index) { | |
| result.emplace_back(indices[index], logits[indices[index]]); | |
| } | |
| return result; | |
| } | |
| int RunLogitsSmoke( | |
| const std::string & package_path, | |
| std::uint32_t token, | |
| const std::string * reference_path) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("logits-smoke requires a build with RKNN support"); | |
| } | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(package_path); | |
| const auto tokenizer = PackageTokenizer(package); | |
| const auto initialize_begin = std::chrono::steady_clock::now(); | |
| ling3::Decoder decoder(package); | |
| const auto initialize_end = std::chrono::steady_clock::now(); | |
| std::vector<float> logits(package.header().vocab_size); | |
| const auto timings = decoder.Eval(token, logits); | |
| const auto top = TopLogits(logits, 10); | |
| std::cout << std::fixed << std::setprecision(6) | |
| << "token=" << token << '\n' | |
| << "initialization_ms=" | |
| << std::chrono::duration<double, std::milli>(initialize_end - initialize_begin).count() | |
| << '\n' | |
| << "layers_ms=" << timings.layers_ms << '\n' | |
| << "output_head_ms=" << timings.output_head_ms << '\n' | |
| << "total_ms=" << timings.total_ms << '\n'; | |
| for (std::size_t index = 0; index < top.size(); ++index) { | |
| const std::array<std::uint32_t, 1> id {top[index].first}; | |
| std::cout << "top" << index << "_id=" << top[index].first | |
| << " logit=" << top[index].second | |
| << " piece=" << std::quoted(tokenizer.Decode(id)) << '\n'; | |
| } | |
| if (reference_path == nullptr) return 0; | |
| const auto reference = ReadFloatFile(*reference_path, logits.size()); | |
| double dot = 0.0, norm = 0.0, reference_norm = 0.0; | |
| double mean_error = 0.0, maximum_error = 0.0; | |
| for (std::size_t index = 0; index < logits.size(); ++index) { | |
| const double error = std::abs(static_cast<double>(logits[index]) - reference[index]); | |
| mean_error += error; | |
| maximum_error = std::max(maximum_error, error); | |
| dot += static_cast<double>(logits[index]) * reference[index]; | |
| norm += static_cast<double>(logits[index]) * logits[index]; | |
| reference_norm += static_cast<double>(reference[index]) * reference[index]; | |
| } | |
| mean_error /= static_cast<double>(logits.size()); | |
| const double cosine = dot / std::sqrt(norm * reference_norm); | |
| std::cout << "cosine_vs_reference=" << cosine << '\n' | |
| << "mean_abs_error=" << mean_error << '\n' | |
| << "max_abs_error=" << maximum_error << '\n'; | |
| return cosine >= 0.90 ? 0 : 13; | |
| } | |
| int RunGenerate( | |
| const std::string & package_path, | |
| std::string_view user_text, | |
| std::size_t maximum_new_tokens) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("generate requires a build with RKNN support"); | |
| } | |
| if (maximum_new_tokens == 0) throw std::invalid_argument("MAX_NEW_TOKENS must be positive"); | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(package_path); | |
| const auto tokenizer = PackageTokenizer(package); | |
| const std::string prompt = | |
| "<role>SYSTEM</role>detailed thinking off<|role_end|>" | |
| "<role>HUMAN</role>" + std::string(user_text) + | |
| "<|role_end|><role>ASSISTANT</role>\n<think></think>"; | |
| const auto tokens = tokenizer.Encode(prompt); | |
| if (tokens.empty() || tokens.size() + maximum_new_tokens > package.header().max_context) { | |
| throw std::runtime_error("prompt and output exceed the package context capacity"); | |
| } | |
| const auto initialize_begin = std::chrono::steady_clock::now(); | |
| ling3::Decoder decoder(package); | |
| const auto initialize_end = std::chrono::steady_clock::now(); | |
| std::vector<float> logits(package.header().vocab_size); | |
| const auto prefill_begin = std::chrono::steady_clock::now(); | |
| for (const auto token : tokens) decoder.Eval(token, logits); | |
| const auto prefill_end = std::chrono::steady_clock::now(); | |
| std::vector<double> decode_ms; | |
| std::vector<std::uint32_t> generated; | |
| decode_ms.reserve(maximum_new_tokens); | |
| generated.reserve(maximum_new_tokens); | |
| std::cout << "response=" << std::flush; | |
| for (std::size_t index = 0; index < maximum_new_tokens; ++index) { | |
| const auto found = std::max_element(logits.begin(), logits.end()); | |
| const auto token = static_cast<std::uint32_t>(found - logits.begin()); | |
| if (token == package.header().eos_token) break; | |
| generated.push_back(token); | |
| std::cout << tokenizer.Piece(token) << std::flush; | |
| const auto timing = decoder.Eval(token, logits); | |
| decode_ms.push_back(timing.total_ms); | |
| std::cerr << "\ntoken=" << token << " position=" << decoder.position() | |
| << " layers_ms=" << timing.layers_ms | |
| << " head_ms=" << timing.output_head_ms | |
| << " total_ms=" << timing.total_ms << std::flush; | |
| } | |
| std::cout << '\n'; | |
| const double initialize_ms = std::chrono::duration<double, std::milli>( | |
| initialize_end - initialize_begin).count(); | |
| const double prefill_ms = std::chrono::duration<double, std::milli>( | |
| prefill_end - prefill_begin).count(); | |
| const double decode_total = std::accumulate(decode_ms.begin(), decode_ms.end(), 0.0); | |
| std::cerr << '\n' << std::fixed << std::setprecision(3) | |
| << "prompt_tokens=" << tokens.size() << '\n' | |
| << "generated_tokens=" << generated.size() << '\n' | |
| << "initialization_ms=" << initialize_ms << '\n' | |
| << "prefill_ms=" << prefill_ms << '\n' | |
| << "prefill_tokens_per_second=" | |
| << (prefill_ms == 0.0 ? 0.0 : 1000.0 * tokens.size() / prefill_ms) << '\n' | |
| << "decode_ms=" << decode_total << '\n' | |
| << "decode_tokens_per_second=" | |
| << (decode_total == 0.0 ? 0.0 : 1000.0 * decode_ms.size() / decode_total) << '\n'; | |
| return 0; | |
| } | |
| struct ProcessMemory { | |
| std::size_t rss_kb = 0; | |
| std::size_t hwm_kb = 0; | |
| }; | |
| ProcessMemory ReadProcessMemory() { | |
| std::ifstream status("/proc/self/status"); | |
| if (!status) throw std::runtime_error("cannot read /proc/self/status"); | |
| ProcessMemory memory; | |
| std::string line; | |
| while (std::getline(status, line)) { | |
| std::istringstream fields(line); | |
| std::string key; | |
| std::size_t value = 0; | |
| std::string unit; | |
| if (!(fields >> key >> value >> unit)) continue; | |
| if (key == "VmRSS:") memory.rss_kb = value; | |
| if (key == "VmHWM:") memory.hwm_kb = value; | |
| } | |
| return memory; | |
| return {}; | |
| } | |
| std::vector<std::uint32_t> BuildBenchmarkPrompt( | |
| const ling3::Tokenizer & tokenizer, | |
| std::size_t sequence_length) { | |
| const auto prefix = tokenizer.Encode( | |
| "<role>SYSTEM</role>detailed thinking off<|role_end|>" | |
| "<role>HUMAN</role>"); | |
| const auto body = tokenizer.Encode(Utf8( | |
| u8"请用中文详细说明如何安排一天的工作,依次讨论目标、执行步骤、风险和复盘方法。" | |
| u8"回答需要完整、连贯并包含具体例子。")); | |
| const auto suffix = tokenizer.Encode( | |
| "<|role_end|><role>ASSISTANT</role>\n<think></think>"); | |
| if (body.empty() || prefix.size() + suffix.size() > sequence_length) { | |
| throw std::invalid_argument("SEQLEN is too short for the deterministic benchmark prompt"); | |
| } | |
| std::vector<std::uint32_t> prompt; | |
| prompt.reserve(sequence_length); | |
| prompt.insert(prompt.end(), prefix.begin(), prefix.end()); | |
| std::size_t body_index = 0; | |
| while (prompt.size() + suffix.size() < sequence_length) { | |
| prompt.push_back(body[body_index++ % body.size()]); | |
| } | |
| prompt.insert(prompt.end(), suffix.begin(), suffix.end()); | |
| return prompt; | |
| } | |
| std::uint32_t GreedyToken(std::span<const float> logits) { | |
| return static_cast<std::uint32_t>( | |
| std::max_element(logits.begin(), logits.end()) - logits.begin()); | |
| } | |
| std::uint64_t TokenHash(std::span<const std::uint32_t> tokens) { | |
| std::uint64_t hash = 1469598103934665603ULL; | |
| for (const auto token : tokens) { | |
| for (unsigned shift = 0; shift < 32; shift += 8) { | |
| hash ^= (token >> shift) & 0xffU; | |
| hash *= 1099511628211ULL; | |
| } | |
| } | |
| return hash; | |
| } | |
| // Reference tokens are teacher-forced so small numerical differences cannot | |
| // change the input sequence and invalidate comparisons of recurrent state. | |
| int RunSequenceCheck(const std::string & path, std::size_t length, | |
| std::size_t steps, const std::string & out, | |
| const std::string & reference, const std::string & corpus = "") { | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(path); | |
| if (!steps || length + steps > package.header().max_context) | |
| throw std::invalid_argument("invalid sequence-check lengths"); | |
| const auto tokenizer = PackageTokenizer(package); | |
| std::vector<std::uint32_t> corpus_tokens; | |
| auto prompt = BuildBenchmarkPrompt(tokenizer, length); | |
| if (!corpus.empty()) { | |
| std::ifstream file(corpus); | |
| if (!file) throw std::runtime_error("cannot open evaluation corpus"); | |
| const std::string body{std::istreambuf_iterator<char>(file), std::istreambuf_iterator<char>()}; | |
| corpus_tokens = tokenizer.Encode(body); | |
| if (corpus_tokens.size() < length + steps) | |
| throw std::runtime_error("evaluation corpus is too short: " + std::to_string(corpus_tokens.size())); | |
| prompt.assign(corpus_tokens.begin(), corpus_tokens.begin() + length); | |
| std::ofstream teacher(out + ".teacher.ids"); | |
| for (std::size_t i = 0; i < steps; ++i) teacher << corpus_tokens[length + i] << '\n'; | |
| if (!teacher) throw std::runtime_error("cannot save evaluation teacher IDs"); | |
| } | |
| std::ofstream prompt_ids(out + ".prompt.ids"); | |
| for (const auto token : prompt) prompt_ids << token << '\n'; | |
| if (!prompt_ids) throw std::runtime_error("cannot save sequence-check prompt IDs"); | |
| ling3::Decoder decoder(package); | |
| std::vector<float> logits(package.header().vocab_size), expected(logits.size()); | |
| std::ofstream values(out + ".f32", std::ios::binary), ids(out + ".ids"); | |
| std::ifstream ref_values, ref_ids; | |
| if (!reference.empty()) { | |
| ref_values.open(reference + ".f32", std::ios::binary); | |
| ref_ids.open(reference + ".ids"); | |
| if (!ref_values || !ref_ids) throw std::runtime_error("cannot open reference"); | |
| } | |
| if (!values || !ids) throw std::runtime_error("cannot open sequence-check output"); | |
| std::size_t offset = 0; | |
| while (offset < length) { | |
| const auto rows = std::min<std::size_t>(128, length - offset); | |
| if (rows == 1) decoder.Eval(prompt[offset], logits); | |
| else { | |
| decoder.PrepareBatch(rows); | |
| decoder.EvalBatch(std::span<const std::uint32_t>(prompt).subspan(offset, rows), logits); | |
| } | |
| offset += rows; | |
| } | |
| double min_cosine = 1.0, max_error = 0.0; | |
| std::size_t agreement = 0; | |
| for (std::size_t step = 0; step < steps; ++step) { | |
| for (const auto x : logits) | |
| if (!std::isfinite(x)) throw std::runtime_error("non-finite logits"); | |
| const auto predicted = GreedyToken(logits); | |
| auto token = predicted; | |
| values.write(reinterpret_cast<const char *>(logits.data()), logits.size() * sizeof(float)); | |
| ids << predicted << '\n'; | |
| if (!reference.empty()) { | |
| ref_values.read(reinterpret_cast<char *>(expected.data()), expected.size() * sizeof(float)); | |
| if (!ref_values || !(ref_ids >> token) || token >= logits.size()) | |
| throw std::runtime_error("invalid/truncated reference"); | |
| for (const auto x : expected) | |
| if (!std::isfinite(x)) throw std::runtime_error("non-finite reference logits"); | |
| const auto error = CompareVectors(logits, expected); | |
| min_cosine = std::min(min_cosine, error[0]); | |
| max_error = std::max(max_error, error[2]); | |
| agreement += token == predicted; | |
| std::cout << "step=" << step << " cosine=" << std::setprecision(10) | |
| << error[0] << " mae=" << error[1] << " max=" << error[2] | |
| << " top1=" << (token == predicted) << '\n'; | |
| } | |
| if (!corpus_tokens.empty()) token = corpus_tokens[length + step]; | |
| if (step + 1 < steps) decoder.Eval(token, logits); | |
| } | |
| if (!values || !ids) throw std::runtime_error("failed writing reference"); | |
| std::cout << "steps=" << steps << " min_cosine=" << min_cosine | |
| << " max_error=" << max_error << " top1_agreement=" << agreement << '\n'; | |
| const auto mla=decoder.AttentionStats(); | |
| std::cout << "mla_npu_calls=" << mla.npu_calls << " mla_cpu_calls=" << mla.cpu_calls | |
| << " mla_fallbacks=" << mla.fallbacks << " mla_npu_ms=" << mla.npu_ms << '\n'; | |
| return reference.empty() || min_cosine >= 0.999 ? 0 : 2; | |
| } | |
| struct BenchmarkSample { | |
| double ttft_ms = 0.0; | |
| double prefill_ms = 0.0; | |
| double decode_ms = 0.0; | |
| double tokens_per_second = 0.0; | |
| std::size_t eos_tokens = 0; | |
| std::uint64_t token_hash = 0; | |
| ProcessMemory memory; | |
| }; | |
| // Same resident decoder for baseline and MTP. Populate every MTP prefix slot | |
| // sequentially, and teacher-force baseline tokens so the work and histories | |
| // stay comparable. This measures forward overhead, not speculative speedup. | |
| int RunMtpBenchmark(const std::string & path, std::size_t length, | |
| std::size_t steps, std::size_t repeats, std::size_t capacity, | |
| const std::string & cases_path) { | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(path); | |
| if (!length || steps < 2 || !repeats || length + steps > capacity || capacity > 262144) | |
| throw std::invalid_argument("invalid MTP benchmark lengths"); | |
| const auto tokenizer = PackageTokenizer(package); | |
| std::vector<std::vector<std::uint32_t>> prompts{BuildBenchmarkPrompt(tokenizer, length)}; | |
| if (!cases_path.empty()) { | |
| std::ifstream file(cases_path); | |
| if (!file) throw std::runtime_error("cannot open MTP cases file"); | |
| std::string question; | |
| while (std::getline(file, question)) { | |
| if (question.empty()) continue; | |
| auto ids = tokenizer.Encode("<role>SYSTEM</role>detailed thinking off<|role_end|>" | |
| "<role>HUMAN</role>" + question + "<|role_end|><role>ASSISTANT</role>\n<think></think>"); | |
| if (ids.empty() || ids.size() + steps > capacity) | |
| throw std::invalid_argument("case does not fit the requested context"); | |
| prompts.push_back(std::move(ids)); | |
| } | |
| } | |
| const auto started = std::chrono::steady_clock::now(); | |
| ling3::Decoder decoder(package, capacity); | |
| decoder.EnableMtp(); | |
| const std::vector<bool> modes = decoder.HasMtp() ? std::vector<bool>{false, true} : std::vector<bool>{false}; | |
| std::vector<float> logits(package.header().vocab_size), draft(logits.size()); | |
| const auto milliseconds = [](auto a, auto b) { | |
| return std::chrono::duration<double, std::milli>(b - a).count(); | |
| }; | |
| std::cout << std::fixed << std::setprecision(4) | |
| << "benchmark=resident_mtp_forward_not_speculative\n" | |
| << "context_capacity=" << capacity << " cases=" << prompts.size() | |
| << " mtp_enabled=" << decoder.HasMtp() << " new_tokens=" << steps << " iterations=" << repeats << '\n' | |
| << "initialization_ms=" << milliseconds(started, std::chrono::steady_clock::now()) << std::endl; | |
| const auto initial_memory = ReadProcessMemory(); | |
| std::cout << "initial_rss_mib=" << initial_memory.rss_kb / 1024.0 | |
| << " initial_hwm_mib=" << initial_memory.hwm_kb / 1024.0 << std::endl; | |
| for (std::size_t case_index = 0; case_index < prompts.size(); ++case_index) { | |
| const auto & prompt = prompts[case_index]; | |
| length = prompt.size(); | |
| for (std::size_t round = 0; round <= repeats; ++round) { | |
| std::vector<std::uint32_t> teacher; | |
| teacher.reserve(steps); | |
| for (bool with_mtp : modes) { | |
| decoder.Reset(); | |
| const auto attention_before = decoder.AttentionStats(); | |
| const auto begin = std::chrono::steady_clock::now(); | |
| double trunk = 0, mtp = 0, mtp_layers = 0, mtp_head = 0; | |
| for (std::size_t i = 0; i < length; ++i) { | |
| decoder.Eval(prompt[i], logits); | |
| if (with_mtp && i + 1 < length) decoder.EvalMtp(prompt[i + 1], draft); | |
| } | |
| std::uint32_t token = GreedyToken(logits); | |
| std::size_t target_agreement = 0, draft_hits = 0; | |
| if (!with_mtp) teacher.push_back(token); | |
| else { | |
| target_agreement += token == teacher[0]; | |
| token = teacher[0]; | |
| } | |
| const auto first = std::chrono::steady_clock::now(); | |
| std::vector<double> trunk_samples, mtp_samples; | |
| for (std::size_t i = 1; i < steps; ++i) { | |
| std::uint32_t proposed = 0; | |
| if (with_mtp) { | |
| const auto t = decoder.EvalMtp(token, draft); | |
| mtp += t.total_ms; mtp_layers += t.layers_ms; mtp_head += t.output_head_ms; | |
| mtp_samples.push_back(t.total_ms); | |
| proposed = GreedyToken(draft); | |
| } | |
| const auto t = decoder.Eval(token, logits); | |
| trunk += t.total_ms; trunk_samples.push_back(t.total_ms); | |
| token = GreedyToken(logits); | |
| if (!with_mtp) teacher.push_back(token); | |
| else { | |
| target_agreement += token == teacher[i]; | |
| draft_hits += proposed == teacher[i]; | |
| token = teacher[i]; | |
| } | |
| } | |
| const auto end = std::chrono::steady_clock::now(); | |
| const auto percentile = [](std::vector<double> values, double q) { | |
| std::sort(values.begin(), values.end()); | |
| return values[static_cast<std::size_t>(q * (values.size() - 1))]; | |
| }; | |
| const auto memory = ReadProcessMemory(); | |
| const auto attention = decoder.AttentionStats(); | |
| std::cout << "phase=" << (round == 0 ? "warmup" : "measured") | |
| << " case=" << case_index << " seqlen=" << length | |
| << " round=" << round << " mode=" << (with_mtp ? "target_plus_mtp" : "target_only") | |
| << " sequential_prefill_ms=" << milliseconds(begin, first) | |
| << " decode_wall_ms=" << milliseconds(first, end) | |
| << " tokens_per_second=" << 1000.0 * (steps - 1) / milliseconds(first, end) | |
| << " target_mean_ms=" << trunk / (steps - 1) | |
| << " target_p50_ms=" << percentile(trunk_samples, 0.5) | |
| << " target_p95_ms=" << percentile(trunk_samples, 0.95); | |
| if (with_mtp) | |
| std::cout << " mtp_mean_ms=" << mtp / (steps - 1) | |
| << " mtp_p50_ms=" << percentile(mtp_samples, 0.5) | |
| << " mtp_p95_ms=" << percentile(mtp_samples, 0.95) | |
| << " mtp_layers_mean_ms=" << mtp_layers / (steps - 1) | |
| << " mtp_head_mean_ms=" << mtp_head / (steps - 1) | |
| << " target_agreement=" << target_agreement << '/' << steps | |
| << " draft_top1_hits=" << draft_hits << '/' << (steps - 1); | |
| std::cout << " mla_npu_calls=" << attention.npu_calls - attention_before.npu_calls | |
| << " mla_cpu_calls=" << attention.cpu_calls - attention_before.cpu_calls | |
| << " mla_fallbacks=" << attention.fallbacks - attention_before.fallbacks | |
| << " rss_mib=" << memory.rss_kb / 1024.0 | |
| << " hwm_mib=" << memory.hwm_kb / 1024.0 | |
| << " teacher_hash=" << std::hex << TokenHash(teacher) << std::dec << std::endl; | |
| if (with_mtp && target_agreement != steps) | |
| throw std::runtime_error("MTP probe changed target predictions for the same teacher-forced tokens"); | |
| } | |
| } | |
| } | |
| return 0; | |
| } | |
| int RunBenchmark( | |
| const std::string & package_path, | |
| std::size_t sequence_length, | |
| std::size_t new_tokens, | |
| std::size_t iterations) { | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("benchmark requires a build with RKNN support"); | |
| } | |
| if (sequence_length == 0 || new_tokens < 2 || iterations == 0) { | |
| throw std::invalid_argument( | |
| "SEQLEN and ITERATIONS must be positive; NEW_TOKENS must be at least 2"); | |
| } | |
| ConfigureInferenceProcess(); | |
| const ling3::ModelPackage package(package_path); | |
| const auto tokenizer = PackageTokenizer(package); | |
| if (sequence_length + new_tokens > package.header().max_context) { | |
| throw std::invalid_argument("SEQLEN + NEW_TOKENS exceeds the package context capacity"); | |
| } | |
| const auto prompt = BuildBenchmarkPrompt(tokenizer, sequence_length); | |
| const auto initialization_begin = std::chrono::steady_clock::now(); | |
| ling3::Decoder decoder(package); | |
| const std::size_t batch_granularity = decoder.batch_granularity(); | |
| std::size_t prefill_batch = std::min<std::size_t>(sequence_length, 128); | |
| if (batch_granularity != 0) prefill_batch -= prefill_batch % batch_granularity; | |
| if (prefill_batch >= 2) decoder.PrepareBatch(prefill_batch); | |
| const auto initialization_end = std::chrono::steady_clock::now(); | |
| const double initialization_ms = std::chrono::duration<double, std::milli>( | |
| initialization_end - initialization_begin).count(); | |
| std::vector<float> logits(package.header().vocab_size); | |
| const auto run_once = [&]() { | |
| BenchmarkSample sample; | |
| decoder.Reset(); | |
| const auto request_begin = std::chrono::steady_clock::now(); | |
| std::size_t offset = 0; | |
| while (batch_granularity != 0 && sequence_length - offset >= 2) { | |
| std::size_t rows = std::min<std::size_t>(sequence_length - offset, 128); | |
| rows -= rows % batch_granularity; | |
| if (rows < 2) break; | |
| const auto block = std::span<const std::uint32_t>(prompt).subspan(offset, rows); | |
| const bool final_block = sequence_length - offset == rows; | |
| if (final_block) { | |
| decoder.EvalBatch(block, logits); | |
| } else { | |
| decoder.EvalBatchState(block); | |
| } | |
| offset += rows; | |
| } | |
| for (; offset < sequence_length; ++offset) decoder.Eval(prompt[offset], logits); | |
| const auto first_token = GreedyToken(logits); | |
| const auto first_token_at = std::chrono::steady_clock::now(); | |
| sample.ttft_ms = std::chrono::duration<double, std::milli>( | |
| first_token_at - request_begin).count(); | |
| sample.prefill_ms = sample.ttft_ms; | |
| std::vector<std::uint32_t> generated; | |
| generated.reserve(new_tokens); | |
| generated.push_back(first_token); | |
| if (first_token == package.header().eos_token) ++sample.eos_tokens; | |
| for (std::size_t index = 1; index < new_tokens; ++index) { | |
| decoder.Eval(generated.back(), logits); | |
| const auto token = GreedyToken(logits); | |
| generated.push_back(token); | |
| if (token == package.header().eos_token) ++sample.eos_tokens; | |
| } | |
| const auto final_token_at = std::chrono::steady_clock::now(); | |
| sample.decode_ms = std::chrono::duration<double, std::milli>( | |
| final_token_at - first_token_at).count(); | |
| sample.tokens_per_second = 1000.0 * static_cast<double>(new_tokens - 1) / | |
| sample.decode_ms; | |
| sample.token_hash = TokenHash(generated); | |
| sample.memory = ReadProcessMemory(); | |
| return sample; | |
| }; | |
| const auto before_warmup = ReadProcessMemory(); | |
| const auto warmup = run_once(); | |
| std::vector<BenchmarkSample> samples; | |
| samples.reserve(iterations); | |
| for (std::size_t iteration = 0; iteration < iterations; ++iteration) { | |
| samples.push_back(run_once()); | |
| } | |
| const auto statistic = [&samples](auto member, bool minimum) { | |
| double result = samples.front().*member; | |
| for (const auto & sample : samples) { | |
| result = minimum ? std::min(result, sample.*member) : std::max(result, sample.*member); | |
| } | |
| return result; | |
| }; | |
| const auto mean = [&samples](auto member) { | |
| double total = 0.0; | |
| for (const auto & sample : samples) total += sample.*member; | |
| return total / static_cast<double>(samples.size()); | |
| }; | |
| std::size_t maximum_rss_kb = 0; | |
| std::size_t maximum_hwm_kb = 0; | |
| for (const auto & sample : samples) { | |
| maximum_rss_kb = std::max(maximum_rss_kb, sample.memory.rss_kb); | |
| maximum_hwm_kb = std::max(maximum_hwm_kb, sample.memory.hwm_kb); | |
| } | |
| std::cout << std::fixed << std::setprecision(3) | |
| << "benchmark=official_style_autoregressive\n" | |
| << "model=Ling-3.0-tiny\n" | |
| << "model_total_parameters_b=7.9\n" | |
| << "model_active_parameters_b=1.3\n" | |
| << "dtype=W4A8+FP16/FP32_mixed\n" | |
| << "seqlen=" << sequence_length << '\n' | |
| << "new_tokens=" << new_tokens << '\n' | |
| << "iterations=" << iterations << '\n' | |
| << "cpu_cores=4,5,6,7\n" | |
| << "npu_cores=0,1,2\n" | |
| << "initialization_ms=" << initialization_ms << '\n' | |
| << "model_file_mb=" << package.mapped_bytes() / (1024.0 * 1024.0) << '\n' | |
| << "rss_before_warmup_mb=" << before_warmup.rss_kb / 1024.0 << '\n' | |
| << "warmup_ttft_ms=" << warmup.ttft_ms << '\n' | |
| << "warmup_tokens_per_second=" << warmup.tokens_per_second << '\n'; | |
| for (std::size_t index = 0; index < samples.size(); ++index) { | |
| const auto & sample = samples[index]; | |
| std::cout << "sample_" << index + 1 << "_ttft_ms=" << sample.ttft_ms << '\n' | |
| << "sample_" << index + 1 << "_tokens_per_second=" | |
| << sample.tokens_per_second << '\n' | |
| << "sample_" << index + 1 << "_rss_mb=" << sample.memory.rss_kb / 1024.0 | |
| << '\n' | |
| << "sample_" << index + 1 << "_hwm_mb=" << sample.memory.hwm_kb / 1024.0 | |
| << '\n' | |
| << "sample_" << index + 1 << "_eos_tokens=" << sample.eos_tokens << '\n' | |
| << "sample_" << index + 1 << "_token_hash=" << std::hex | |
| << sample.token_hash << std::dec << '\n'; | |
| } | |
| std::cout << "ttft_mean_ms=" << mean(&BenchmarkSample::ttft_ms) << '\n' | |
| << "ttft_min_ms=" << statistic(&BenchmarkSample::ttft_ms, true) << '\n' | |
| << "ttft_max_ms=" << statistic(&BenchmarkSample::ttft_ms, false) << '\n' | |
| << "tokens_per_second_mean=" << mean(&BenchmarkSample::tokens_per_second) << '\n' | |
| << "tokens_per_second_min=" | |
| << statistic(&BenchmarkSample::tokens_per_second, true) << '\n' | |
| << "tokens_per_second_max=" | |
| << statistic(&BenchmarkSample::tokens_per_second, false) << '\n' | |
| << "memory_rss_mb=" << maximum_rss_kb / 1024.0 << '\n' | |
| << "memory_hwm_mb=" << maximum_hwm_kb / 1024.0 << '\n'; | |
| return 0; | |
| } | |
| } // namespace | |
| int main(int argc, char ** argv) { | |
| try { | |
| if (argc < 2) { | |
| PrintUsage(argv[0]); | |
| return 1; | |
| } | |
| const std::string_view command(argv[1]); | |
| if (command == "inspect" && argc == 3) { | |
| const ling3::ModelPackage package(argv[2]); | |
| ling3::ValidateLing3Tiny(package.header()); | |
| std::cout << "valid Ling-3.0-tiny package\n" | |
| << "bytes=" << package.mapped_bytes() << '\n' | |
| << "tensors=" << package.tensors().size() << '\n' | |
| << "max_context=" << package.header().max_context << '\n' | |
| << "weight_bits=" << package.header().default_weight_bits << '\n' | |
| << "activation_bits=" << package.header().default_activation_bits << '\n'; | |
| return 0; | |
| } | |
| if (command == "plan" && argc <= 3) { | |
| const std::size_t max_context = argc == 3 ? ParseSize(argv[2], "context") : 4096; | |
| const ling3::ExecutionPlan plan = ling3::BuildExecutionPlan(max_context); | |
| std::cout << "layers=" << plan.layers.size() << '\n' | |
| << "decode_matmul_contexts=" << plan.decode_matmul_contexts << '\n' | |
| << "rknn_island_contexts=" << plan.rknn_island_contexts << '\n' | |
| << "kda_state_bytes_fp16=" << plan.kda_state_bytes_fp16 << '\n' | |
| << "mla_cache_bytes_fp16=" << plan.mla_cache_bytes_fp16 << '\n' | |
| << "activation_arena_bytes=" << plan.activation_arena_bytes << '\n'; | |
| return 0; | |
| } | |
| if (command == "rknn-smoke" && argc <= 3) { | |
| const std::size_t iterations = argc == 3 ? ParseSize(argv[2], "iterations") : 10; | |
| return ling3::RunRknnMatmulSmoke(iterations); | |
| } | |
| if (command == "tokenizer-smoke" && argc == 3) { | |
| return RunTokenizerSmoke(argv[2]); | |
| } | |
| if (command == "gdn-smoke" && (argc == 3 || argc == 4)) { | |
| const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 50; | |
| return RunGdnSmoke(argv[2], iterations); | |
| } | |
| if (command == "gdn-batch16-smoke" && (argc == 3 || argc == 4)) { | |
| const std::size_t iterations = argc == 4 ? ParseSize(argv[3], "iterations") : 10; | |
| return RunGdnBatch16Smoke(argv[2], iterations); | |
| } | |
| if (command == "layer0-smoke" && (argc == 5 || argc == 6)) { | |
| const auto token = ParseSize(argv[3], "token"); | |
| if (token > std::numeric_limits<std::uint32_t>::max()) { | |
| throw std::runtime_error("token is out of range"); | |
| } | |
| const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 20; | |
| return RunLayer0Smoke( | |
| argv[2], static_cast<std::uint32_t>(token), argv[4], iterations); | |
| } | |
| if (command == "linear-smoke" && (argc == 4 || argc == 5)) { | |
| const std::size_t iterations = argc == 5 ? ParseSize(argv[4], "iterations") : 20; | |
| return RunLinearSmoke(argv[2], argv[3], iterations); | |
| } | |
| if (command == "linear-batch-smoke" && argc >= 5 && argc <= 7) { | |
| const std::size_t rows = ParseSize(argv[4], "rows"); | |
| const std::size_t iterations = argc >= 6 ? ParseSize(argv[5], "iterations") : 20; | |
| const std::string weight_base = argc == 7 ? argv[6] : argv[3]; | |
| return RunLinearBatchSmoke(argv[2], argv[3], rows, iterations, weight_base); | |
| } | |
| if (command == "prefill32-smoke" && argc == 3) { | |
| return RunPrefill32Smoke(argv[2]); | |
| } | |
| if (command == "gdn-cpu-state-check" && argc == 3) { | |
| return RunGdnCpuStateCheck(argv[2]); | |
| } | |
| if (command == "gdn-reference-check" && argc == 3) { | |
| return RunGdnReferenceCheck(argv[2]); | |
| } | |
| if (command == "logits-smoke" && (argc == 4 || argc == 5)) { | |
| const auto token = ParseSize(argv[3], "token"); | |
| if (token > std::numeric_limits<std::uint32_t>::max()) { | |
| throw std::runtime_error("token is out of range"); | |
| } | |
| const std::string reference = argc == 5 ? argv[4] : ""; | |
| return RunLogitsSmoke( | |
| argv[2], static_cast<std::uint32_t>(token), argc == 5 ? &reference : nullptr); | |
| } | |
| if (command == "generate" && (argc == 4 || argc == 5)) { | |
| const std::size_t maximum_new_tokens = argc == 5 | |
| ? ParseSize(argv[4], "maximum new tokens") | |
| : 64; | |
| return RunGenerate(argv[2], argv[3], maximum_new_tokens); | |
| } | |
| if (command == "benchmark" && (argc == 5 || argc == 6)) { | |
| const std::size_t sequence_length = ParseSize(argv[3], "sequence length"); | |
| const std::size_t new_tokens = ParseSize(argv[4], "new tokens"); | |
| const std::size_t iterations = argc == 6 ? ParseSize(argv[5], "iterations") : 3; | |
| return RunBenchmark(argv[2], sequence_length, new_tokens, iterations); | |
| } | |
| if (command == "mtp-benchmark" && (argc == 7 || argc == 8)) { | |
| return RunMtpBenchmark(argv[2], ParseSize(argv[3], "sequence length"), | |
| ParseSize(argv[4], "new tokens"), ParseSize(argv[5], "iterations"), | |
| ParseSize(argv[6], "context capacity"), argc == 8 ? argv[7] : ""); | |
| } | |
| if (command == "sequence-check" && argc >= 6 && argc <= 8) { | |
| return RunSequenceCheck(argv[2], ParseSize(argv[3], "sequence length"), | |
| ParseSize(argv[4], "new tokens"), argv[5], | |
| argc >= 7 ? argv[6] : "", argc == 8 ? argv[7] : ""); | |
| } | |
| if (command == "serve" && (argc == 3 || argc == 4)) { | |
| const std::size_t port = argc == 4 ? ParseSize(argv[3], "port") : 9091; | |
| if (port == 0 || port > std::numeric_limits<std::uint16_t>::max()) { | |
| throw std::runtime_error("port must be in [1, 65535]"); | |
| } | |
| if (!ling3::RknnBackendAvailable()) { | |
| throw std::runtime_error("serve requires a build with RKNN support"); | |
| } | |
| ConfigureInferenceProcess(); | |
| return ling3::RunHttpService(argv[2], static_cast<std::uint16_t>(port)); | |
| } | |
| PrintUsage(argv[0]); | |
| return 1; | |
| } catch (const std::exception & error) { | |
| std::cerr << "error: " << error.what() << '\n'; | |
| return 1; | |
| } | |
| } | |