File size: 2,265 Bytes
cf0f656
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
#!/usr/bin/env python3
"""Build the independent, weight-only MSE-fitted W4 package."""
from __future__ import annotations

import argparse
import json
import sys
import time
from pathlib import Path

import torch

sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from mamba2_recall import runtime
from mamba2_recall.w4 import DEFAULT_CHUNK_ROWS, quantize_package


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--source", required=True, help="Pinned official NVIDIA checkpoint file or directory")
    parser.add_argument("--output", required=True, help="Packed package directory")
    parser.add_argument("--device", default="cuda")
    parser.add_argument("--chunk-rows", type=int, default=DEFAULT_CHUNK_ROWS)
    parser.add_argument("--cpu-threads", type=int, default=8)
    parser.add_argument("--resume", action="store_true")
    parser.add_argument("--protocol", default=str(Path(__file__).resolve().parents[1] / "docs" / "PROTOCOL.md"))
    args = parser.parse_args()
    if args.cpu_threads < 1 or args.chunk_rows < 1:
        parser.error("CPU threads and chunk rows must be positive")
    torch.set_num_threads(args.cpu_threads)
    if str(args.device).startswith("cuda"):
        torch.cuda.reset_peak_memory_stats(args.device)
    start = time.monotonic()

    def progress(event):
        print(json.dumps({"event": "tensor_complete", "elapsed_seconds": time.monotonic() - start, **event}), flush=True)

    manifest = quantize_package(args.source, args.output, device=args.device,
                                chunk_rows=args.chunk_rows, resume=args.resume,
                                progress=progress, protocol_path=args.protocol)
    receipt = {"event": "package_complete", "elapsed_seconds": time.monotonic() - start,
               "manifest_sha256": runtime.sha256_file(Path(args.output) / "manifest.json"),
               "tensor_count": manifest["tensor_count"], "tensor_bytes": manifest["tensor_bytes"],
               "peak_gpu_allocated_bytes": torch.cuda.max_memory_allocated(args.device) if str(args.device).startswith("cuda") else None,
               "complete": manifest["complete"]}
    print(json.dumps(receipt), flush=True)


if __name__ == "__main__":
    main()