Ling-3.0-tiny-RKNN / src /model_package.cpp
Sariel00's picture
Publish Ling-3.0-tiny RKNN engine and model
3fd1a35 verified
Raw History Blame Contribute Delete
8.64 kB
#include "ling3/model_package.h"
#include <algorithm>
#include <cerrno>
#include <cstring>
#include <fcntl.h>
#include <stdexcept>
#include <string>
#include <sys/mman.h>
#include <sys/stat.h>
#include <unistd.h>
namespace ling3 {
namespace {
bool RangeValid(std::uint64_t offset, std::uint64_t bytes, std::size_t file_bytes) {
return offset <= file_bytes && bytes <= file_bytes - offset;
}
std::runtime_error SystemError(const std::string & operation) {
return std::runtime_error(operation + ": " + std::strerror(errno));
}
} // namespace
std::uint32_t Crc32(const std::byte * data, std::size_t bytes) {
std::uint32_t crc = 0xFFFFFFFFU;
for (std::size_t index = 0; index < bytes; ++index) {
crc ^= static_cast<std::uint8_t>(data[index]);
for (int bit = 0; bit < 8; ++bit) {
crc = (crc >> 1U) ^ (0xEDB88320U & (0U - (crc & 1U)));
}
}
return ~crc;
}
ModelPackage::ModelPackage(const std::filesystem::path & path) {
try {
fd_ = open(path.c_str(), O_RDONLY | O_CLOEXEC);
if (fd_ < 0) throw SystemError("open " + path.string());
struct stat info {};
if (fstat(fd_, &info) != 0) throw SystemError("stat " + path.string());
if (info.st_size < static_cast<off_t>(sizeof(PackageHeader))) {
throw std::runtime_error("model package is smaller than its header");
}
mapped_bytes_ = static_cast<std::size_t>(info.st_size);
void * mapped = mmap(nullptr, mapped_bytes_, PROT_READ, MAP_PRIVATE, fd_, 0);
if (mapped == MAP_FAILED) throw SystemError("mmap " + path.string());
mapping_ = static_cast<const std::byte *>(mapped);
header_ = reinterpret_cast<const PackageHeader *>(mapping_);
if (!std::equal(kPackageMagic.begin(), kPackageMagic.end(), header_->magic) ||
header_->version != kPackageVersion ||
header_->header_bytes != sizeof(PackageHeader) ||
header_->tensor_entry_bytes != sizeof(TensorEntry) ||
header_->endian_tag != kEndianTag ||
header_->file_bytes != mapped_bytes_) {
throw std::runtime_error("invalid Ling3RKNN package header");
}
PackageHeader header_copy = *header_;
const std::uint32_t expected_crc = header_copy.header_crc32;
header_copy.header_crc32 = 0;
if (expected_crc != Crc32(reinterpret_cast<const std::byte *>(&header_copy), sizeof(header_copy))) {
throw std::runtime_error("model package header checksum mismatch");
}
const std::uint64_t table_bytes =
static_cast<std::uint64_t>(header_->tensor_count) * sizeof(TensorEntry);
if (!RangeValid(header_->tensor_table_offset, table_bytes, mapped_bytes_) ||
!RangeValid(header_->string_table_offset, header_->string_table_bytes, mapped_bytes_) ||
!RangeValid(header_->payload_offset, header_->payload_bytes, mapped_bytes_)) {
throw std::runtime_error("model package contains an out-of-range table");
}
const auto * entries = reinterpret_cast<const TensorEntry *>(mapping_ + header_->tensor_table_offset);
const char * strings = reinterpret_cast<const char *>(mapping_ + header_->string_table_offset);
tensors_.reserve(header_->tensor_count);
tensor_index_.reserve(header_->tensor_count);
for (std::uint32_t index = 0; index < header_->tensor_count; ++index) {
const TensorEntry & entry = entries[index];
if (entry.rank > 4 || entry.name_offset > header_->string_table_bytes ||
entry.name_bytes > header_->string_table_bytes - entry.name_offset ||
!RangeValid(entry.data_offset, entry.data_bytes, mapped_bytes_) ||
(entry.aux_bytes != 0 && !RangeValid(entry.aux_offset, entry.aux_bytes, mapped_bytes_))) {
throw std::runtime_error("model package contains an invalid tensor entry");
}
std::string_view name(strings + entry.name_offset, entry.name_bytes);
if (name.empty() || name.find('\0') != std::string_view::npos) {
throw std::runtime_error("model package contains an invalid tensor name");
}
const auto [_, inserted] = tensor_index_.emplace(name, tensors_.size());
if (!inserted) throw std::runtime_error("duplicate tensor name in model package");
tensors_.push_back({
name,
&entry,
mapping_ + entry.data_offset,
entry.aux_bytes == 0 ? nullptr : mapping_ + entry.aux_offset,
});
}
} catch (...) {
Reset();
throw;
}
}
ModelPackage::~ModelPackage() {
Reset();
}
void ModelPackage::Reset() noexcept {
tensors_.clear();
tensor_index_.clear();
header_ = nullptr;
if (mapping_ != nullptr) munmap(const_cast<std::byte *>(mapping_), mapped_bytes_);
mapping_ = nullptr;
mapped_bytes_ = 0;
if (fd_ >= 0) close(fd_);
fd_ = -1;
}
const TensorView & ModelPackage::tensor(std::string_view name) const {
const auto found = tensor_index_.find(name);
if (found == tensor_index_.end()) {
throw std::out_of_range("model package does not contain tensor " + std::string(name));
}
return tensors_[found->second];
}
void ModelPackage::DiscardCopiedLinearWeight(const TensorView & weight) const {
const auto & owned = tensor(weight.name);
if (owned.entry != weight.entry || owned.data != weight.data ||
weight.entry->role != static_cast<std::uint32_t>(TensorRole::kLinearWeight) ||
(weight.entry->dtype != static_cast<std::uint32_t>(DataType::kInt4Low) &&
weight.entry->dtype != static_cast<std::uint32_t>(DataType::kBFloat16))) {
throw std::invalid_argument("only this package's copied linear weights can be discarded");
}
const auto page = sysconf(_SC_PAGESIZE);
if (page <= 0) throw SystemError("sysconf page size");
const auto size = static_cast<std::size_t>(page);
const auto begin = ((weight.entry->data_offset + size - 1) / size) * size;
const auto end = ((weight.entry->data_offset + weight.entry->data_bytes) / size) * size;
if (end <= begin) return;
// MADV_DONTNEED removes this process's PTEs, then FADV_DONTNEED asks the
// filesystem to evict the clean backing pages as well. No global drop_caches.
if (madvise(const_cast<std::byte *>(mapping_ + begin), end - begin, MADV_DONTNEED) != 0)
throw SystemError("discard copied weight mapping");
const int status = posix_fadvise(fd_, begin, end - begin, POSIX_FADV_DONTNEED);
if (status != 0) ++cache_advice_failures_;
discarded_weight_bytes_ += end - begin;
}
std::size_t ModelPackage::resident_linear_weight_bytes() const {
const auto page = sysconf(_SC_PAGESIZE);
if (page <= 0) throw SystemError("sysconf page size");
const auto size = static_cast<std::size_t>(page);
std::vector<unsigned char> resident((mapped_bytes_ + size - 1) / size);
if (mincore(const_cast<std::byte *>(mapping_), mapped_bytes_, resident.data()) != 0)
throw SystemError("mincore model");
std::size_t bytes = 0;
for (const auto & tensor : tensors_) {
if (tensor.entry->role != static_cast<std::uint32_t>(TensorRole::kLinearWeight) ||
(tensor.entry->dtype != static_cast<std::uint32_t>(DataType::kInt4Low) &&
tensor.entry->dtype != static_cast<std::uint32_t>(DataType::kBFloat16))) continue;
const auto begin = (tensor.entry->data_offset + size - 1) / size;
const auto end = (tensor.entry->data_offset + tensor.entry->data_bytes) / size;
for (auto index = begin; index < end; ++index) if (resident[index] & 1) bytes += size;
}
return bytes;
}
void ValidateLing3Tiny(const PackageHeader & h) {
const bool valid =
h.vocab_size == 157184 && h.hidden_size == 1536 && h.layer_count == 24 &&
h.attention_heads == 16 && h.head_dim == 128 && h.kv_lora_rank == 512 &&
h.q_lora_rank == 256 && h.qk_nope_dim == 128 && h.qk_rope_dim == 64 &&
h.value_head_dim == 128 && h.dense_ffn_dim == 4608 && h.expert_ffn_dim == 512 &&
h.shared_ffn_dim == 512 && h.expert_count == 128 && h.experts_per_token == 8 &&
h.expert_group_count == 8 && h.selected_group_count == 4 &&
h.layer_group_size == 4 && h.leading_dense_layers == 1 &&
h.convolution_kernel == 4 && h.mla_layer_count == 6 && h.kda_layer_count == 18;
if (!valid) throw std::runtime_error("package model configuration is not Ling-3.0-tiny");
}
} // namespace ling3