Download tools/quantize_model.py from Sariel00/Ling-3.0-tiny-RKNN: direct link, hf CLI and curl.
- Browser
- Download file 23.9 kB
-
https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/quantize_model.py
- Command line
-
hf download hf://Sariel00/Ling-3.0-tiny-RKNN/tools/quantize_model.py
-
curl -L -o quantize_model.py https://huggingface.co/Sariel00/Ling-3.0-tiny-RKNN/resolve/main/tools/quantize_model.py
23.9 kB
| #!/usr/bin/env python3 | |
| """Quantize the pinned Ling-3.0-tiny safetensors into standalone RKNN assets.""" | |
| from __future__ import annotations | |
| import argparse | |
| import hashlib | |
| import json | |
| import math | |
| import shutil | |
| import struct | |
| from pathlib import Path | |
| import torch | |
| from safetensors import safe_open | |
| REVISION = "e3a47d5b986e7141b6efd62597d598ebb392060d" | |
| HIDDEN = 1536 | |
| HEADS = 16 | |
| HEAD_DIM = 128 | |
| EXPERTS = 128 | |
| EXPERT_DIM = 512 | |
| VOCAB = 157184 | |
| N_ALIGNMENT = 64 | |
| K_ALIGNMENT = 32 | |
| def parse_layers(value: str) -> list[int]: | |
| result: set[int] = set() | |
| for item in value.split(","): | |
| item = item.strip() | |
| if not item: | |
| continue | |
| if "-" in item: | |
| begin, end = (int(part) for part in item.split("-", 1)) | |
| result.update(range(begin, end + 1)) | |
| else: | |
| result.add(int(item)) | |
| if not result or min(result) < 0 or max(result) >= 24: | |
| raise ValueError("--layers must select values in [0, 23]") | |
| return sorted(result) | |
| def relative(path: Path, root: Path) -> str: | |
| return str(path.relative_to(root)) | |
| class TensorSource: | |
| def __init__(self, root: Path): | |
| self.root = root | |
| index = json.loads((root / "model.safetensors.index.json").read_text()) | |
| self.weight_map: dict[str, str] = index["weight_map"] | |
| self.handles = { | |
| shard: safe_open(root / shard, framework="pt", device="cpu") | |
| for shard in sorted(set(self.weight_map.values())) | |
| } | |
| def tensor(self, name: str) -> torch.Tensor: | |
| try: | |
| return self.handles[self.weight_map[name]].get_tensor(name) | |
| except KeyError as error: | |
| raise KeyError(f"source model does not contain {name}") from error | |
| def source_linear_names(name: str) -> list[str]: | |
| """Map the runtime's fused matrices to the exact checkpoint row order.""" | |
| parent, _, suffix = name.rpartition('.') | |
| if suffix == 'qkvfgb': | |
| return [parent+'.'+s+'.weight' for s in ('q_proj','k_proj','v_proj','f_proj','g_proj','b_proj')] | |
| if suffix == 'qkv_gate_a': | |
| return [parent+'.'+s+'.weight' for s in ('q_a_proj','kv_a_proj_with_mqa','g_proj')] | |
| if suffix == 'gate_up': | |
| return [parent+'.'+s+'.weight' for s in ('gate_proj','up_proj')] | |
| if suffix == 'o_proj' and '.attention.' in name and (int(name.split('.')[2])+1)%4 == 0: | |
| return [parent+'.dense.weight'] | |
| return [name+'.weight'] | |
| class AssetWriter: | |
| def __init__(self, root: Path, calibration_scales: Path | None = None): | |
| self.root = root | |
| self.assets = root / "assets" | |
| self.assets.mkdir(parents=True, exist_ok=True) | |
| self.entries: list[dict[str, object]] = [] | |
| self.calibration_scales = None | |
| self.calibration_sha256 = None | |
| if calibration_scales is not None: | |
| from safetensors.torch import load_file | |
| self.calibration_scales = load_file(str(calibration_scales), device='cpu') | |
| self.calibration_sha256 = hashlib.sha256(calibration_scales.read_bytes()).hexdigest() | |
| def add_file( | |
| self, | |
| name: str, | |
| path: Path, | |
| dtype: str, | |
| dims: list[int], | |
| role: str, | |
| *, | |
| quant: str = "none", | |
| layout: str = "row_major", | |
| flags: int = 0, | |
| layer: int = 0xFFFFFFFF, | |
| expert: int = 0xFFFFFFFF, | |
| core: int = 0xFFFFFFFF, | |
| ) -> None: | |
| self.entries.append( | |
| { | |
| "name": name, | |
| "path": relative(path, self.root), | |
| "dtype": dtype, | |
| "dims": dims, | |
| "role": role, | |
| "quant": quant, | |
| "layout": layout, | |
| "flags": flags, | |
| "layer": layer, | |
| "expert": expert, | |
| "core": core, | |
| } | |
| ) | |
| def raw(self, source: TensorSource, name: str, *, role: str, layer: int = 0xFFFFFFFF) -> None: | |
| tensor = source.tensor(name).contiguous() | |
| suffix = "f32" if tensor.dtype == torch.float32 else "bf16" | |
| if tensor.dtype not in (torch.float32, torch.bfloat16): | |
| raise TypeError(f"unsupported raw dtype for {name}: {tensor.dtype}") | |
| path = self.assets / "raw" / f"{name}.{suffix}.bin" | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| expected = tensor.numel() * (4 if tensor.dtype == torch.float32 else 2) | |
| if not path.is_file() or path.stat().st_size != expected: | |
| temporary = path.with_suffix(path.suffix + ".partial") | |
| values = tensor.numpy() if tensor.dtype == torch.float32 else tensor.view(torch.uint16).numpy() | |
| values.tofile(temporary) | |
| temporary.replace(path) | |
| self.add_file(name, path, suffix, list(tensor.shape), role, layer=layer) | |
| def blob( | |
| self, | |
| name: str, | |
| path: Path, | |
| *, | |
| role: str, | |
| core: int = 0xFFFFFFFF, | |
| ) -> None: | |
| self.add_file( | |
| name, | |
| path, | |
| "unknown", | |
| [path.stat().st_size], | |
| role, | |
| layout="opaque", | |
| core=core, | |
| ) | |
| def copied_blob( | |
| self, | |
| name: str, | |
| source: Path, | |
| *, | |
| role: str, | |
| core: int, | |
| ) -> None: | |
| filename = name.replace(".", "-") + source.suffix | |
| destination = self.assets / "rknn" / filename | |
| destination.parent.mkdir(parents=True, exist_ok=True) | |
| if not destination.is_file() or destination.stat().st_size != source.stat().st_size: | |
| temporary = destination.with_suffix(destination.suffix + ".partial") | |
| shutil.copyfile(source, temporary) | |
| temporary.replace(destination) | |
| self.blob(name, destination, role=role, core=core) | |
| def _quantize_rows(weight_nk: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: | |
| weight = weight_nk.float() | |
| positive = weight.amax(dim=1).clamp_min(0.0) / 7.0 | |
| negative = (-weight.amin(dim=1)).clamp_min(0.0) / 8.0 | |
| scales = torch.maximum(positive, negative) | |
| scales = torch.where(scales == 0, torch.ones_like(scales), scales) | |
| codes = torch.round(weight / scales[:, None]).clamp(-8, 7).to(torch.int8) | |
| return codes, scales | |
| def _safe_splits(codes_nk: torch.Tensor, required_parts: int | None = None) -> int: | |
| n, k = codes_nk.shape | |
| if k % K_ALIGNMENT: | |
| raise ValueError(f"K={k} is not aligned to {K_ALIGNMENT}") | |
| blocks = k // K_ALIGNMENT | |
| if required_parts is not None and not 1 <= required_parts <= blocks: | |
| raise ValueError('invalid required K partitions') | |
| for parts in ([required_parts] if required_parts is not None else range(1, blocks + 1)): | |
| base, extra = divmod(blocks, parts) | |
| begin = 0 | |
| safe = True | |
| for part in range(parts): | |
| end = begin + (base + (part < extra)) * K_ALIGNMENT | |
| for row in range(0, n, 1024): | |
| block = codes_nk[row : row + 1024, begin:end].to(torch.int32) | |
| positive = block.clamp_min(0) | |
| negative = (-block).clamp_min(0) | |
| high = (7 * positive + 8 * negative).sum(dim=1) | |
| low = (8 * positive + 7 * negative).sum(dim=1) | |
| if int(high.max()) > 32767 or int(low.max()) > 32768: | |
| safe = False | |
| break | |
| if not safe: | |
| break | |
| begin = end | |
| if safe: | |
| return parts | |
| raise RuntimeError(f"cannot find an INT16-safe split for {tuple(codes_nk.shape)}") | |
| def _write_packed_kn(path: Path, codes_nk: torch.Tensor) -> None: | |
| temporary = path.with_suffix(path.suffix + ".partial") | |
| with temporary.open("wb") as output: | |
| _, k = codes_nk.shape | |
| for begin in range(0, k, 32): | |
| values = codes_nk[:, begin : begin + 32].t().contiguous().view(-1).numpy() | |
| nibbles = values.view("uint8") & 0x0F | |
| packed = nibbles[0::2] | (nibbles[1::2] << 4) | |
| output.write(packed.tobytes()) | |
| temporary.replace(path) | |
| def linear( | |
| self, | |
| name: str, | |
| tensors_nk: list[torch.Tensor], | |
| *, | |
| layer: int = 0xFFFFFFFF, | |
| expert: int = 0xFFFFFFFF, | |
| pad_n: bool = False, | |
| required_k_splits: int | None = None, | |
| ) -> None: | |
| if not tensors_nk: | |
| raise ValueError("linear requires at least one source tensor") | |
| k = tensors_nk[0].shape[1] | |
| if any(tensor.ndim != 2 or tensor.shape[1] != k for tensor in tensors_nk): | |
| raise ValueError(f"incompatible tensors for fused linear {name}") | |
| logical_n = sum(tensor.shape[0] for tensor in tensors_nk) | |
| n = math.ceil(logical_n / N_ALIGNMENT) * N_ALIGNMENT if pad_n else logical_n | |
| if k % K_ALIGNMENT or n % N_ALIGNMENT: | |
| raise ValueError(f"unaligned W4 linear {name}: K={k} N={n}") | |
| directory = self.assets / "linear" / name | |
| weight_path = directory / "weight-int4-kn.bin" | |
| scale_path = directory / "scales-f32.bin" | |
| correction_path = directory / "correction-i32.bin" | |
| spec_path = directory / "spec.json" | |
| expected_weight_bytes = k * n // 2 | |
| valid = False | |
| if spec_path.is_file() and weight_path.is_file() and scale_path.is_file() and correction_path.is_file(): | |
| spec = json.loads(spec_path.read_text()) | |
| valid = ( | |
| spec.get("k") == k | |
| and spec.get("n") == n | |
| and spec.get("logical_n") == logical_n | |
| and weight_path.stat().st_size == expected_weight_bytes | |
| and scale_path.stat().st_size == n * 4 | |
| and correction_path.stat().st_size == n * 4 | |
| and spec.get('calibration_sha256') == self.calibration_sha256 | |
| and (required_k_splits is None or spec.get('k_splits') == required_k_splits) | |
| ) | |
| if not valid: | |
| directory.mkdir(parents=True, exist_ok=True) | |
| weight = torch.cat([tensor.float() for tensor in tensors_nk], dim=0) | |
| if n != logical_n: | |
| weight = torch.cat((weight, torch.zeros((n - logical_n, k))), dim=0) | |
| if self.calibration_scales is None: | |
| codes, scales = self._quantize_rows(weight) | |
| else: | |
| source_names = source_linear_names(name) | |
| if len(source_names) != len(tensors_nk): | |
| raise ValueError(f'calibration source mapping mismatch: {name}') | |
| values = [] | |
| for source_name, tensor in zip(source_names, tensors_nk): | |
| scale = self.calibration_scales[source_name] | |
| if scale.shape != (tensor.shape[0],) or not torch.isfinite(scale).all() or not (scale > 0).all(): | |
| raise ValueError(f'invalid calibration scale: {source_name}') | |
| values.append(scale) | |
| scales = torch.cat(values).float() | |
| if n != logical_n: | |
| scales = torch.cat((scales, torch.ones(n-logical_n))) | |
| codes = torch.round(weight/scales[:,None]).clamp(-8,7).to(torch.int8) | |
| splits = self._safe_splits(codes,required_k_splits) | |
| correction = 8 * codes.to(torch.int32).sum(dim=1) | |
| self._write_packed_kn(weight_path, codes) | |
| scales.numpy().astype("float32", copy=False).tofile(scale_path) | |
| correction.numpy().astype("int32", copy=False).tofile(correction_path) | |
| spec = {"k": k, "n": n, "logical_n": logical_n, "k_splits": splits} | |
| if self.calibration_sha256 is not None: | |
| spec['calibration_sha256'] = self.calibration_sha256 | |
| spec_path.write_text(json.dumps(spec, sort_keys=True) + "\n") | |
| else: | |
| splits = int(spec["k_splits"]) | |
| core = expert % 3 if expert != 0xFFFFFFFF else 0xFFFFFFFF | |
| self.add_file( | |
| name + ".weight", | |
| weight_path, | |
| "int4", | |
| [k, n], | |
| "linear_weight", | |
| quant="per_output_channel", | |
| layout="packed_int4", | |
| flags=splits, | |
| layer=layer, | |
| expert=expert, | |
| core=core, | |
| ) | |
| self.add_file( | |
| name + ".scales", | |
| scale_path, | |
| "fp32", | |
| [n], | |
| "linear_scale", | |
| layer=layer, | |
| expert=expert, | |
| core=core, | |
| ) | |
| self.add_file( | |
| name + ".correction", | |
| correction_path, | |
| "int32", | |
| [n], | |
| "bias", | |
| layer=layer, | |
| expert=expert, | |
| core=core, | |
| ) | |
| def byte_alphabet() -> tuple[dict[int, str], dict[str, int]]: | |
| values = list(range(ord("!"), ord("~") + 1)) | |
| values += list(range(ord("¡"), ord("¬") + 1)) | |
| values += list(range(ord("®"), ord("ÿ") + 1)) | |
| encoded = values[:] | |
| extra = 0 | |
| for byte in range(256): | |
| if byte not in values: | |
| values.append(byte) | |
| encoded.append(256 + extra) | |
| extra += 1 | |
| forward = {byte: chr(codepoint) for byte, codepoint in zip(values, encoded)} | |
| return forward, {symbol: byte for byte, symbol in forward.items()} | |
| def export_tokenizer(source_root: Path, output: Path) -> None: | |
| tokenizer = json.loads((source_root / "tokenizer.json").read_text()) | |
| vocab: dict[str, int] = tokenizer["model"]["vocab"] | |
| merges: list[list[str]] = tokenizer["model"]["merges"] | |
| added = tokenizer["added_tokens"] | |
| tokenizer_count = max(max(vocab.values()), max(item["id"] for item in added)) + 1 | |
| if tokenizer_count > VOCAB: | |
| raise ValueError(f"tokenizer vocabulary is {tokenizer_count}, larger than model vocab {VOCAB}") | |
| full_count = VOCAB | |
| forward, inverse = byte_alphabet() | |
| byte_ids = [vocab[forward[byte]] for byte in range(256)] | |
| pieces = [b""] * full_count | |
| for text, token_id in vocab.items(): | |
| try: | |
| pieces[token_id] = bytes(inverse[character] for character in text) | |
| except KeyError as error: | |
| raise ValueError(f"vocabulary token contains a non-byte-alphabet character: {text!r}") from error | |
| added_tokens: list[tuple[int, bytes, int]] = [] | |
| for item in added: | |
| content = item["content"].encode("utf-8") | |
| pieces[item["id"]] = content | |
| flags = 1 if item.get("special") else 0 | |
| added_tokens.append((item["id"], content, flags)) | |
| merge_rows = [] | |
| for rank, pair in enumerate(merges): | |
| left, right = pair | |
| result = left + right | |
| if left not in vocab or right not in vocab or result not in vocab: | |
| raise ValueError(f"merge {rank} cannot be represented with vocabulary IDs") | |
| merge_rows.append((vocab[left], vocab[right], vocab[result])) | |
| output.parent.mkdir(parents=True, exist_ok=True) | |
| temporary = output.with_suffix(output.suffix + ".partial") | |
| with temporary.open("wb") as stream: | |
| stream.write(struct.pack("<8sIIII", b"L3TOK2\0\0", full_count, len(merge_rows), len(added_tokens), 0)) | |
| stream.write(struct.pack("<256I", *byte_ids)) | |
| for piece in pieces: | |
| stream.write(struct.pack("<I", len(piece))) | |
| stream.write(piece) | |
| for left, right, result in merge_rows: | |
| stream.write(struct.pack("<III", left, right, result)) | |
| for token_id, content, flags in added_tokens: | |
| stream.write(struct.pack("<III", token_id, len(content), flags)) | |
| stream.write(content) | |
| temporary.replace(output) | |
| def source_tensors(source: TensorSource, names: list[str]) -> list[torch.Tensor]: | |
| return [source.tensor(name) for name in names] | |
| def quantize_layer(source: TensorSource, writer: AssetWriter, layer: int, include_experts: bool) -> None: | |
| prefix = f"model.layers.{layer}" | |
| writer.raw(source, prefix + ".input_layernorm.weight", role="norm", layer=layer) | |
| writer.raw(source, prefix + ".post_attention_layernorm.weight", role="norm", layer=layer) | |
| attention = prefix + ".attention" | |
| if (layer + 1) % 4: | |
| projection_names = [ | |
| attention + ".q_proj.weight", | |
| attention + ".k_proj.weight", | |
| attention + ".v_proj.weight", | |
| attention + ".f_proj.weight", | |
| attention + ".g_proj.weight", | |
| attention + ".b_proj.weight", | |
| ] | |
| writer.linear(attention + ".qkvfgb", source_tensors(source, projection_names), layer=layer, pad_n=True) | |
| writer.linear(attention + ".o_proj", [source.tensor(attention + ".o_proj.weight")], layer=layer) | |
| for name in ("q_conv1d.weight", "k_conv1d.weight", "v_conv1d.weight", "A_log", "dt_bias", "o_norm.weight"): | |
| writer.raw(source, attention + "." + name, role="norm" if name == "o_norm.weight" else "unknown", layer=layer) | |
| else: | |
| fused_names = [ | |
| attention + ".q_a_proj.weight", | |
| attention + ".kv_a_proj_with_mqa.weight", | |
| attention + ".g_proj.weight", | |
| ] | |
| writer.linear(attention + ".qkv_gate_a", source_tensors(source, fused_names), layer=layer, pad_n=True) | |
| writer.linear(attention + ".q_b_proj", [source.tensor(attention + ".q_b_proj.weight")], layer=layer) | |
| writer.linear(attention + ".kv_b_proj", [source.tensor(attention + ".kv_b_proj.weight")], layer=layer) | |
| writer.linear(attention + ".o_proj", [source.tensor(attention + ".dense.weight")], layer=layer) | |
| writer.raw(source, attention + ".q_a_layernorm.weight", role="norm", layer=layer) | |
| writer.raw(source, attention + ".kv_a_layernorm.weight", role="norm", layer=layer) | |
| mlp = prefix + ".mlp" | |
| if layer == 0: | |
| writer.linear( | |
| mlp + ".gate_up", | |
| source_tensors(source, [mlp + ".gate_proj.weight", mlp + ".up_proj.weight"]), | |
| layer=layer, | |
| ) | |
| writer.linear(mlp + ".down_proj", [source.tensor(mlp + ".down_proj.weight")], layer=layer) | |
| return | |
| writer.raw(source, mlp + ".gate.weight", role="unknown", layer=layer) | |
| writer.raw(source, mlp + ".gate.expert_bias", role="bias", layer=layer) | |
| shared = mlp + ".shared_experts" | |
| writer.linear( | |
| shared + ".gate_up", | |
| source_tensors(source, [shared + ".gate_proj.weight", shared + ".up_proj.weight"]), | |
| layer=layer, | |
| ) | |
| writer.linear(shared + ".down_proj", [source.tensor(shared + ".down_proj.weight")], layer=layer) | |
| if not include_experts: | |
| return | |
| for expert in range(EXPERTS): | |
| root = mlp + f".experts.{expert}" | |
| writer.linear( | |
| root + ".gate_up", | |
| source_tensors(source, [root + ".gate_proj.weight", root + ".up_proj.weight"]), | |
| layer=layer, | |
| expert=expert, | |
| ) | |
| writer.linear( | |
| root + ".down_proj", | |
| [source.tensor(root + ".down_proj.weight")], | |
| layer=layer, | |
| expert=expert, | |
| ) | |
| if expert % 16 == 15: | |
| print(f"layer {layer}: quantized experts 0-{expert}", flush=True) | |
| def make_manifest(writer: AssetWriter, max_context: int, complete: bool) -> dict[str, object]: | |
| return { | |
| "source_revision": REVISION, | |
| "flags": 1 if complete else 0, | |
| "model": { | |
| "vocab_size": VOCAB, | |
| "hidden_size": HIDDEN, | |
| "layer_count": 24, | |
| "attention_heads": HEADS, | |
| "head_dim": HEAD_DIM, | |
| "kv_lora_rank": 512, | |
| "q_lora_rank": 256, | |
| "qk_nope_dim": 128, | |
| "qk_rope_dim": 64, | |
| "value_head_dim": 128, | |
| "dense_ffn_dim": 4608, | |
| "expert_ffn_dim": EXPERT_DIM, | |
| "shared_ffn_dim": EXPERT_DIM, | |
| "expert_count": EXPERTS, | |
| "experts_per_token": 8, | |
| "expert_group_count": 8, | |
| "selected_group_count": 4, | |
| "layer_group_size": 4, | |
| "leading_dense_layers": 1, | |
| "convolution_kernel": 4, | |
| "max_context": max_context, | |
| "eos_token": 156895, | |
| "pad_token": 156892, | |
| "bos_token": 156891, | |
| "mla_layer_count": 6, | |
| "kda_layer_count": 18, | |
| "default_weight_bits": 4, | |
| "default_activation_bits": 8, | |
| "rms_epsilon": 0.000001, | |
| "rope_theta": 6000000.0, | |
| "routed_scale": 2.5, | |
| "kda_lower_bound": -5.0, | |
| }, | |
| "tensors": writer.entries, | |
| } | |
| def main() -> int: | |
| parser = argparse.ArgumentParser(description=__doc__) | |
| parser.add_argument("--source", type=Path, required=True) | |
| parser.add_argument("--output", type=Path, required=True) | |
| parser.add_argument("--layers", default="0-23") | |
| parser.add_argument("--skip-experts", action="store_true") | |
| parser.add_argument("--skip-lm-head", action="store_true") | |
| parser.add_argument("--max-context", type=int, default=4096) | |
| parser.add_argument('--calibration-scales', type=Path, | |
| help='Activation-calibrated per-source-output-row scales; changes no runtime layout') | |
| parser.add_argument( | |
| "--gdn-dir", | |
| type=Path, | |
| help="directory containing heads5/heads6 flat-GDN RKNN models", | |
| ) | |
| args = parser.parse_args() | |
| source_root = args.source.resolve() | |
| output_root = args.output.resolve() | |
| layers = parse_layers(args.layers) | |
| if not (source_root / "model.safetensors.index.json").is_file(): | |
| raise FileNotFoundError("source model is incomplete") | |
| config = json.loads((source_root / "config.json").read_text()) | |
| if config.get("model_type") != "bailing_hybrid" or config.get("vocab_size") != VOCAB: | |
| raise ValueError("source config is not Ling-3.0-tiny") | |
| writer = AssetWriter(output_root, args.calibration_scales) | |
| source = TensorSource(source_root) | |
| tokenizer_path = writer.assets / "tokenizer.l3tok" | |
| tokenizer_valid = ( | |
| tokenizer_path.is_file() | |
| and tokenizer_path.read_bytes()[:8] == b"L3TOK2\0\0" | |
| ) | |
| if not tokenizer_valid: | |
| export_tokenizer(source_root, tokenizer_path) | |
| writer.blob("tokenizer", tokenizer_path, role="tokenizer") | |
| if args.gdn_dir is not None: | |
| gdn_root = args.gdn_dir.resolve() | |
| writer.copied_blob( | |
| "rknn.gdn.heads6", | |
| gdn_root / "heads6" / "ling3_gdn_step_fp16_rk3588.rknn", | |
| role="rknn_island", | |
| core=0, | |
| ) | |
| writer.copied_blob( | |
| "rknn.gdn.heads5", | |
| gdn_root / "heads5" / "ling3_gdn_step_fp16_rk3588.rknn", | |
| role="rknn_island", | |
| core=1, | |
| ) | |
| writer.raw(source, "model.word_embeddings.weight", role="embedding") | |
| writer.raw(source, "model.norm.weight", role="norm") | |
| for layer in layers: | |
| print(f"quantizing layer {layer}", flush=True) | |
| quantize_layer(source, writer, layer, not args.skip_experts) | |
| if not args.skip_lm_head: | |
| print("quantizing lm_head", flush=True) | |
| writer.linear("lm_head", [source.tensor("lm_head.weight")]) | |
| complete = layers == list(range(24)) and not args.skip_experts and not args.skip_lm_head | |
| manifest = make_manifest(writer, args.max_context, complete) | |
| if writer.calibration_sha256 is not None: | |
| manifest['calibration'] = {'method':'activation_diagonal_weighted_scale_search_v1', | |
| 'scales_sha256':writer.calibration_sha256} | |
| manifest_path = output_root / "package-manifest.json" | |
| manifest_path.write_text(json.dumps(manifest, ensure_ascii=True, indent=2) + "\n") | |
| print(f"wrote {len(writer.entries)} entries to {manifest_path}") | |
| print(f"complete={complete}") | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |