File size: 6,033 Bytes
ff5f59d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
"""Stage the McBopomofoLM v2 runtime files (Core ML only, ANE) into SlothE/Bundle/SlothE/
(folder reference -> Contents/Resources/SlothE/).

Inputs (read-only):
  * $W/ane/enc/models/enc25m_multi_pal2_emb.mlpackage   SlothE-T 25M encoder, 2-bit ternary, functions L8/L16/L32/L64/L256,
                                                        caller-side embedding (ids = fp16 rows [1, L, 352], mask [1, L])
  * $W/ane/enc/work/embed_f16_25m.bin                   the fp16 embedding table [1539, 352] (== w25m.npz["embed"])
  * $W/ane/dec/mf/dec_mf_fp16.mlpackage                 SlothE decoder pred_q35_60m, fp16, functions t16/t32/t64/t96, B=3
  * HF cache Luigi/sloth-ime-models @ e7d13c9 pred_q35_60m/tokenizer.json (the decoder's byte-level BPE)
  * the v1 staging's vocabulary files (syl_vocab.tsv, char2id.tsv, syl2legal.bin, variants.tsv; same vocab for 12M/25M)
Outputs (SlothE/Bundle/SlothE/):
  enc25m.mlmodelc/        xcrun coremlcompiler compile of the encoder package
  dec60m.mlmodelc/        xcrun coremlcompiler compile of the decoder package
  enc25m_embed_f16.bin    embedding table
  dec_tokenizer.json      tokenizer.json, byte copy
  syl_vocab.tsv char2id.tsv syl2legal.bin variants.tsv
  runtime-manifest.txt    "<sha256> <bytes> <path>" for every runtime file, including every file inside the
                          two .mlmodelc directories; the app verifies it before loading (SlothEEngine.cpp)
  MANIFEST.txt NOTICE.txt OpenCC-LICENSE.txt
usage: python3 prep_resources_v2.py   (needs only the standard library + xcrun)
"""
import hashlib
import shutil
import subprocess
import tempfile
from pathlib import Path

HERE = Path(__file__).resolve().parent
W = HERE.parent.parent
OUT = HERE / "Bundle" / "SlothE"
ENC_PKG = W / "ane" / "enc" / "models" / "enc25m_multi_pal2_emb.mlpackage"
EMBED = W / "ane" / "enc" / "work" / "embed_f16_25m.bin"
DEC_PKG = W / "ane" / "dec" / "mf" / "dec_mf_fp16.mlpackage"
HF = Path.home() / ".cache/huggingface/hub/models--Luigi--sloth-ime-models/snapshots/e7d13c9c451dc01ab9546be53c574f4dbe0fdd54"
TOKENIZER = HF / "pred_q35_60m" / "tokenizer.json"
KEEP = ["syl_vocab.tsv", "char2id.tsv", "syl2legal.bin", "variants.tsv", "OpenCC-LICENSE.txt"]


def sha256(p):
    h = hashlib.sha256()
    with open(p, "rb") as f:
        for b in iter(lambda: f.read(1 << 20), b""):
            h.update(b)
    return h.hexdigest()


def tree_sha(d):
    h = hashlib.sha256()
    for p in sorted(x for x in Path(d).rglob("*") if x.is_file()):
        h.update(str(p.relative_to(d)).encode() + b"\0" + sha256(p).encode() + b"\n")
    return h.hexdigest()


def compile_pkg(pkg, name, tmp):
    src = Path(tmp) / f"{name}.mlpackage"
    shutil.copytree(pkg, src)
    subprocess.run(["xcrun", "coremlcompiler", "compile", str(src), str(tmp)], check=True, capture_output=True)
    dst = OUT / f"{name}.mlmodelc"
    shutil.rmtree(dst, ignore_errors=True)
    shutil.move(str(Path(tmp) / f"{name}.mlmodelc"), dst)
    return dst


def main():
    OUT.mkdir(parents=True, exist_ok=True)
    keep = {n: (OUT / n).read_bytes() for n in KEEP}
    for p in OUT.iterdir():   # v2 ships no GGUF, no ggml/llama.cpp licence files
        shutil.rmtree(p) if p.is_dir() else p.unlink()
    for n, b in keep.items():
        (OUT / n).write_bytes(b)
    with tempfile.TemporaryDirectory() as tmp:
        compile_pkg(ENC_PKG, "enc25m", tmp)
        compile_pkg(DEC_PKG, "dec60m", tmp)
    shutil.copyfile(EMBED, OUT / "enc25m_embed_f16.bin")
    shutil.copyfile(TOKENIZER, OUT / "dec_tokenizer.json")
    assert (OUT / "enc25m_embed_f16.bin").stat().st_size == 1539 * 352 * 2
    (OUT / "NOTICE.txt").write_text(
        "McBopomofoLM v2 bundles third-party parts (side-by-side build next to stock McBopomofo):\n"
        "- SlothE-T 25M encoder (enc25m.mlmodelc, enc25m_embed_f16.bin; converted to Core ML from the published "
        "weights) and its syl_vocab / syl2legal tables: huggingface.co/Luigi/sloth-ime-models, Apache-2.0.\n"
        "- SlothE decoder pred_q35_60m (dec60m.mlmodelc, dec_tokenizer.json; converted to Core ML from the published "
        "weights): huggingface.co/Luigi/sloth-ime-models, Apache-2.0.\n"
        "- char2id table (char2id.tsv, converted from enc/char2id.json): huggingface.co/spaces/Luigi/slothing-web, Apache-2.0.\n"
        "- Orthographic variant classes derived from OpenCC TWVariants.txt / HKVariants.txt, Apache-2.0 "
        "(OpenCC-LICENSE.txt).\n"
        "v2 runs the models with Core ML on the Apple Neural Engine only. It contains no ggml, llama.cpp or "
        "libslothe code.\n", encoding="utf-8")
    files = sorted(str(p.relative_to(OUT)) for p in OUT.rglob("*")
                   if p.is_file() and p.name not in ("runtime-manifest.txt", "MANIFEST.txt", "NOTICE.txt", "OpenCC-LICENSE.txt"))
    (OUT / "runtime-manifest.txt").write_text(
        "# McBopomofoLM v2 checks the byte size and sha256 of every file below before loading a model.\n"
        "# Any mismatch or missing file = that model is not loaded (encoder -> stock McBopomofo,\n"
        "# decoder -> encoder-only in-walk). Format: <sha256> <bytes> <path>\n"
        + "".join(f"{sha256(OUT / n)} {(OUT / n).stat().st_size} {n}\n" for n in files), encoding="utf-8")
    lines = [
        "SlothE runtime files for McBopomofoLM v2 (Core ML, ANE only)",
        f"encoder source: {ENC_PKG.relative_to(W)} tree sha256 {tree_sha(ENC_PKG)}",
        f"decoder source: {DEC_PKG.relative_to(W)} tree sha256 {tree_sha(DEC_PKG)}",
        f"embedding: {EMBED.relative_to(W)} sha256 {sha256(EMBED)}",
        f"tokenizer: HF pred_q35_60m/tokenizer.json sha256 {sha256(TOKENIZER)}",
        "compiled with: xcrun coremlcompiler compile",
        "config A' (walk2/frozen_final2.json): 25M, beta 0.3, gamma 0, pen -15, variant guard; decoder lambda 2, tau 0.5, top-3",
        "",
    ]
    (OUT / "MANIFEST.txt").write_text("\n".join(lines) + "\n", encoding="utf-8")
    print("\n".join(lines))
    print(len(files), "files in runtime-manifest.txt")


if __name__ == "__main__":
    main()