#include "ling3/model_package.h" #include #include #include #include #include #include #include #include #include namespace ling3 { namespace { bool RangeValid(std::uint64_t offset, std::uint64_t bytes, std::size_t file_bytes) { return offset <= file_bytes && bytes <= file_bytes - offset; } std::runtime_error SystemError(const std::string & operation) { return std::runtime_error(operation + ": " + std::strerror(errno)); } } // namespace std::uint32_t Crc32(const std::byte * data, std::size_t bytes) { std::uint32_t crc = 0xFFFFFFFFU; for (std::size_t index = 0; index < bytes; ++index) { crc ^= static_cast(data[index]); for (int bit = 0; bit < 8; ++bit) { crc = (crc >> 1U) ^ (0xEDB88320U & (0U - (crc & 1U))); } } return ~crc; } ModelPackage::ModelPackage(const std::filesystem::path & path) { try { fd_ = open(path.c_str(), O_RDONLY | O_CLOEXEC); if (fd_ < 0) throw SystemError("open " + path.string()); struct stat info {}; if (fstat(fd_, &info) != 0) throw SystemError("stat " + path.string()); if (info.st_size < static_cast(sizeof(PackageHeader))) { throw std::runtime_error("model package is smaller than its header"); } mapped_bytes_ = static_cast(info.st_size); void * mapped = mmap(nullptr, mapped_bytes_, PROT_READ, MAP_PRIVATE, fd_, 0); if (mapped == MAP_FAILED) throw SystemError("mmap " + path.string()); mapping_ = static_cast(mapped); header_ = reinterpret_cast(mapping_); if (!std::equal(kPackageMagic.begin(), kPackageMagic.end(), header_->magic) || header_->version != kPackageVersion || header_->header_bytes != sizeof(PackageHeader) || header_->tensor_entry_bytes != sizeof(TensorEntry) || header_->endian_tag != kEndianTag || header_->file_bytes != mapped_bytes_) { throw std::runtime_error("invalid Ling3RKNN package header"); } PackageHeader header_copy = *header_; const std::uint32_t expected_crc = header_copy.header_crc32; header_copy.header_crc32 = 0; if (expected_crc != Crc32(reinterpret_cast(&header_copy), sizeof(header_copy))) { throw std::runtime_error("model package header checksum mismatch"); } const std::uint64_t table_bytes = static_cast(header_->tensor_count) * sizeof(TensorEntry); if (!RangeValid(header_->tensor_table_offset, table_bytes, mapped_bytes_) || !RangeValid(header_->string_table_offset, header_->string_table_bytes, mapped_bytes_) || !RangeValid(header_->payload_offset, header_->payload_bytes, mapped_bytes_)) { throw std::runtime_error("model package contains an out-of-range table"); } const auto * entries = reinterpret_cast(mapping_ + header_->tensor_table_offset); const char * strings = reinterpret_cast(mapping_ + header_->string_table_offset); tensors_.reserve(header_->tensor_count); tensor_index_.reserve(header_->tensor_count); for (std::uint32_t index = 0; index < header_->tensor_count; ++index) { const TensorEntry & entry = entries[index]; if (entry.rank > 4 || entry.name_offset > header_->string_table_bytes || entry.name_bytes > header_->string_table_bytes - entry.name_offset || !RangeValid(entry.data_offset, entry.data_bytes, mapped_bytes_) || (entry.aux_bytes != 0 && !RangeValid(entry.aux_offset, entry.aux_bytes, mapped_bytes_))) { throw std::runtime_error("model package contains an invalid tensor entry"); } std::string_view name(strings + entry.name_offset, entry.name_bytes); if (name.empty() || name.find('\0') != std::string_view::npos) { throw std::runtime_error("model package contains an invalid tensor name"); } const auto [_, inserted] = tensor_index_.emplace(name, tensors_.size()); if (!inserted) throw std::runtime_error("duplicate tensor name in model package"); tensors_.push_back({ name, &entry, mapping_ + entry.data_offset, entry.aux_bytes == 0 ? nullptr : mapping_ + entry.aux_offset, }); } } catch (...) { Reset(); throw; } } ModelPackage::~ModelPackage() { Reset(); } void ModelPackage::Reset() noexcept { tensors_.clear(); tensor_index_.clear(); header_ = nullptr; if (mapping_ != nullptr) munmap(const_cast(mapping_), mapped_bytes_); mapping_ = nullptr; mapped_bytes_ = 0; if (fd_ >= 0) close(fd_); fd_ = -1; } const TensorView & ModelPackage::tensor(std::string_view name) const { const auto found = tensor_index_.find(name); if (found == tensor_index_.end()) { throw std::out_of_range("model package does not contain tensor " + std::string(name)); } return tensors_[found->second]; } void ModelPackage::DiscardCopiedLinearWeight(const TensorView & weight) const { const auto & owned = tensor(weight.name); if (owned.entry != weight.entry || owned.data != weight.data || weight.entry->role != static_cast(TensorRole::kLinearWeight) || (weight.entry->dtype != static_cast(DataType::kInt4Low) && weight.entry->dtype != static_cast(DataType::kBFloat16))) { throw std::invalid_argument("only this package's copied linear weights can be discarded"); } const auto page = sysconf(_SC_PAGESIZE); if (page <= 0) throw SystemError("sysconf page size"); const auto size = static_cast(page); const auto begin = ((weight.entry->data_offset + size - 1) / size) * size; const auto end = ((weight.entry->data_offset + weight.entry->data_bytes) / size) * size; if (end <= begin) return; // MADV_DONTNEED removes this process's PTEs, then FADV_DONTNEED asks the // filesystem to evict the clean backing pages as well. No global drop_caches. if (madvise(const_cast(mapping_ + begin), end - begin, MADV_DONTNEED) != 0) throw SystemError("discard copied weight mapping"); const int status = posix_fadvise(fd_, begin, end - begin, POSIX_FADV_DONTNEED); if (status != 0) ++cache_advice_failures_; discarded_weight_bytes_ += end - begin; } std::size_t ModelPackage::resident_linear_weight_bytes() const { const auto page = sysconf(_SC_PAGESIZE); if (page <= 0) throw SystemError("sysconf page size"); const auto size = static_cast(page); std::vector resident((mapped_bytes_ + size - 1) / size); if (mincore(const_cast(mapping_), mapped_bytes_, resident.data()) != 0) throw SystemError("mincore model"); std::size_t bytes = 0; for (const auto & tensor : tensors_) { if (tensor.entry->role != static_cast(TensorRole::kLinearWeight) || (tensor.entry->dtype != static_cast(DataType::kInt4Low) && tensor.entry->dtype != static_cast(DataType::kBFloat16))) continue; const auto begin = (tensor.entry->data_offset + size - 1) / size; const auto end = (tensor.entry->data_offset + tensor.entry->data_bytes) / size; for (auto index = begin; index < end; ++index) if (resident[index] & 1) bytes += size; } return bytes; } void ValidateLing3Tiny(const PackageHeader & h) { const bool valid = h.vocab_size == 157184 && h.hidden_size == 1536 && h.layer_count == 24 && h.attention_heads == 16 && h.head_dim == 128 && h.kv_lora_rank == 512 && h.q_lora_rank == 256 && h.qk_nope_dim == 128 && h.qk_rope_dim == 64 && h.value_head_dim == 128 && h.dense_ffn_dim == 4608 && h.expert_ffn_dim == 512 && h.shared_ffn_dim == 512 && h.expert_count == 128 && h.experts_per_token == 8 && h.expert_group_count == 8 && h.selected_group_count == 4 && h.layer_group_size == 4 && h.leading_dense_layers == 1 && h.convolution_kernel == 4 && h.mla_layer_count == 6 && h.kda_layer_count == 18; if (!valid) throw std::runtime_error("package model configuration is not Ling-3.0-tiny"); } } // namespace ling3