#!/usr/bin/env python3 """Audit actual exported linear weights against pinned BF16, without inference. Read in row chunks; exclude alignment padding, raw embeddings, routers and norms. FP64 reductions; W8 reconstruction follows src/linear.cpp (FP32 per-row scale). No activation quantization or kernel errors are represented by these metrics. """ import argparse import json import math import mmap import time from pathlib import Path import numpy as np import torch from pack_model import HEADER, ENTRY from quantize_model import TensorSource, source_linear_names class Package: def __init__(self, path): self.file = path.open('rb') self.data = mmap.mmap(self.file.fileno(), 0, access=mmap.ACCESS_READ) h = HEADER.unpack_from(self.data) assert h[0] == b'L3RKNN1\0' and h[12] == len(self.data) self.info = dict(path=str(path.resolve()), bytes=len(self.data), revision=h[13].hex()) self.entries = {} for i in range(h[4]): e = ENTRY.unpack_from(self.data, h[7] + i * ENTRY.size) name = self.data[h[8]+e[0]:h[8]+e[0]+e[1]].decode() self.entries[name] = e def array(self, name, dtype): e = self.entries[name] dt = np.dtype(dtype) return np.frombuffer(self.data, dtype=dt, count=e[13]//dt.itemsize, offset=e[12]) def rows(self, base, lo, hi): e = self.entries[base+'.weight'] k, n = e[8:10] if e[2] == 1: assert e[6] == 0 b = self.array(base+'.weight', '> 4 q[q >= 8] -= 16 q = q.T.astype(np.float32) if e[5] == 4: assert e[7] == 32 b = self.array(base+'.scales', '