File size: 11,161 Bytes
61b6fb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ed926f1
 
61b6fb9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
"""Assemble the Hugging Face release bundle of Ines-1 (codename mini-v41-Decisions) from COMMITTED code only.

    python scripts/decisions/release/build_release.py --checkpoint CKPT_DIR --run RUN_DIR \
        --parent-run PARENT_RUN_DIR --repo REPO --model-commit SHA --release-commit SHA \
        --audit AUDIT_DIR --out OUT_DIR

Needs torch and safetensors. Every code file in the bundle comes from `git archive` of a commit:
  mini_v41/            <- --model-commit   (the modelling code)
  mini_v41_jev/, scripts/, examples/, README.md, requirements.txt, .gitattributes
                       <- --release-commit (scripts/decisions/release/, decisions.py resolved)
  training/code/       <- --release-commit (the scripts that built the data and trained the model)
Weights: the checkpoint's fp32 state dict cast to bf16 (inference already runs the model in bf16 on
CUDA, so this is the evaluated model). Internal absolute paths are reduced to file names.
"""
from __future__ import annotations

import argparse
import codecs
import hashlib
import io
import json
import re
import shutil
import subprocess
import tarfile
from pathlib import Path


# absolute data paths (not the "clean50k/da" "ta/" folder inside a dataset repo), home dirs, internal addresses, and the internal
# names below. They are stored rot13-encoded so that this file, which ships in training/code/, does not trip itself.
_NAMES = [codecs.decode(t, "rot13") for t in ['rivpvbfb', 'freire-vn', 'k630590', 'fjncraretvn', 'phfgbzwri',
                                               'p4svk', 'x4v', 'frpergvn', 'frpergcvybg', 'fnovpb', 'freivat-pbasvt', '<guvf ercb']]
LEAK = re.compile(r"(?<![\w.])/da" r"ta/|(?<![\w.])/ho" r"me/|10\.20[01]\.\d|" + "|".join(map(re.escape, _NAMES)), re.I)


def sha256(p: Path) -> str:
    h = hashlib.sha256()
    with open(p, "rb") as f:
        for chunk in iter(lambda: f.read(1 << 24), b""):
            h.update(chunk)
    return h.hexdigest()


def git_extract(repo: Path, commit: str, path: str, dest: Path, strip: str = "") -> list[str]:
    """Extract `path` at `commit` into `dest`, removing the `strip` prefix. Symlinks are resolved
    against the same commit, so the bundle never contains a dangling link."""
    data = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit, path],
                          check=True, capture_output=True).stdout
    names = []
    with tarfile.open(fileobj=io.BytesIO(data)) as tar:
        for m in tar.getmembers():
            if not m.isfile() and not m.issym():
                continue
            rel = m.name[len(strip):] if strip and m.name.startswith(strip) else m.name
            target = dest / rel
            target.parent.mkdir(parents=True, exist_ok=True)
            if m.issym():
                src = "/".join(_normpath((Path(m.name).parent / m.linkname).as_posix()))
                blob = subprocess.run(["git", "-C", str(repo), "show", "%s:%s" % (commit, src)],
                                      check=True, capture_output=True).stdout
                target.write_bytes(blob)
            else:
                target.write_bytes(tar.extractfile(m).read())
            names.append(rel)
    return names


def _normpath(p: str) -> list[str]:
    out = []
    for part in p.split("/"):
        if part == "..":
            out.pop()
        elif part not in ("", "."):
            out.append(part)
    return out


def strip_paths(x):
    if isinstance(x, dict):
        return {k: strip_paths(v) for k, v in x.items()}
    if isinstance(x, list):
        return [strip_paths(v) for v in x]
    if isinstance(x, str) and x.startswith("/"):
        return Path(x).name
    return x


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--checkpoint", type=Path, required=True)
    ap.add_argument("--run", type=Path, required=True, help="the training run directory (results.json)")
    ap.add_argument("--parent-run", type=Path, required=True)
    ap.add_argument("--repo", type=Path, required=True)
    ap.add_argument("--model-commit", required=True)
    ap.add_argument("--release-commit", required=True)
    ap.add_argument("--audit", type=Path, required=True)
    ap.add_argument("--out", type=Path, required=True)
    ap.add_argument("--weights-from", type=Path, help="copy model.safetensors verbatim from this bundle (no re-export)")
    ap.add_argument("--expect-weights-sha", help="required sha256 prefix of model.safetensors")
    a = ap.parse_args()
    import torch
    from safetensors.torch import save_file

    out = a.out
    if out.exists():
        raise SystemExit("%s exists: the bundle is always built into a new directory" % out)
    out.mkdir(parents=True)
    ck = a.checkpoint
    for c in (a.model_commit, a.release_commit):
        subprocess.run(["git", "-C", str(a.repo), "cat-file", "-e", c + "^{commit}"], check=True)

    # weights
    src_pt = ck / "model" / "model.pt"
    if a.weights_from:  # rename-only rebuilds: the published weights file is copied, never regenerated
        shutil.copy(a.weights_from / "model.safetensors", out / "model.safetensors")
        import math
        from safetensors import safe_open
        with safe_open(str(out / "model.safetensors"), "pt") as f:
            n_params = sum(math.prod(f.get_slice(k).get_shape()) for k in f.keys())
    else:
        sd = torch.load(src_pt, map_location="cpu", weights_only=False)
        tensors = {k: (v.detach().to(torch.bfloat16) if v.is_floating_point() else v.detach()).contiguous().clone()
                   for k, v in sd.items()}
        save_file(tensors, str(out / "model.safetensors"), metadata={"format": "pt", "dtype": "bfloat16"})
        n_params = sum(v.numel() for v in tensors.values())
    if a.expect_weights_sha:
        got = sha256(out / "model.safetensors")
        if not got.startswith(a.expect_weights_sha):
            raise SystemExit("model.safetensors sha256 %s != expected %s" % (got, a.expect_weights_sha))

    cfg = json.loads((ck / "config.json").read_text())
    man = json.loads((ck / "manifest.json").read_text())
    ident = man["identity"]
    (out / "config.json").write_text(json.dumps({
        "name": "Ines-1",
        "codename": "mini-v41 / mini-v41-Decisions",
        "model_type": "mini-v41",
        "architectures": ["MiniV41"],
        "description": "1.59B-parameter MoE causal encoder-decoder (405M active per token) with Engram n-gram "
                       "memory, fine-tuned to answer typed decisions (choice / score / noul)",
        "parameters_total": n_params,
        "parameters_active_per_token": 405052224,
        "model": cfg["model"],
        "decision_reader": {"options": "letters A-Z read at the first assistant position", "max_options": 26,
                            "languages": ["en", "es"], "max_prompt_tokens": cfg["model"]["max_sequence_length"],
                            "questions_per_prompt": 1},
        "provenance": {
            "run_id": ident["run_id"], "parent_run_id": ident.get("parent_run_id"),
            "parent_weights_sha256": ident.get("parent_checkpoint_sha256"),
            "source_weights_sha256_fp32_pt": sha256(src_pt),
            "selected_epoch": man["extra"].get("selected_epoch"), "val_score": man["extra"].get("val_score"),
            "training_recipe_hash": ident.get("training_recipe_hash"),
            "dataset_subset_id": ident.get("dataset_subset_id"),
            "tokenizer_sha256": man["tokenizer_identity"]["sha256"],
            "engram_backend_identity": man["engram_backend_identity"],
            "training_code_commit": ident.get("code_commit"),
            "bundle_model_code_commit": a.model_commit, "bundle_release_code_commit": a.release_commit,
            "torch_version_at_training": man["torch_version"]},
    }, indent=2) + "\n")
    shutil.copytree(ck / "tokenizer", out / "tokenizer")

    # code, from commits only
    git_extract(a.repo, a.model_commit, "mini_v41", out)
    (out / "LICENSE-CODE").write_bytes(subprocess.run(["git", "-C", str(a.repo), "show", a.model_commit + ":LICENSE"],
                                                      check=True, capture_output=True).stdout)  # MIT, for the code
    for p in list((out / "mini_v41").rglob("__pycache__")):
        shutil.rmtree(p)
    git_extract(a.repo, a.release_commit, "scripts/decisions/release", out, strip="scripts/decisions/release/")
    (out / "build_release.py").unlink(missing_ok=True)  # the builder is in training/code/, not in the package
    for f in ("prepare_public.py", "prepare_mix.py", "train_decisions.py", "decisions.py", "eval_items.py",
              "metrics.py", "reference_rows.py", "audit_data.py", "audit_report.py", "release/build_release.py"):
        git_extract(a.repo, a.release_commit, "scripts/decisions/" + f, out / "training" / "code",
                    strip="scripts/decisions/")
    git_extract(a.repo, a.release_commit, "scripts/decisions/launch", out / "training" / "code",
                strip="scripts/decisions/")

    # recipe and provenance of both fine-tuning stages
    tr = out / "training"
    for name, run in (("stage2_mix", a.run), ("stage1_typed_jev", a.parent_run)):
        res = json.loads((run / "results.json").read_text())
        (tr / ("%s.results.json" % name)).write_text(json.dumps(strip_paths(res), indent=1, ensure_ascii=False) + "\n")
        m = json.loads((run / "checkpoint" / "manifest.json").read_text())
        (tr / ("%s.manifest.json" % name)).write_text(json.dumps(strip_paths(m), indent=1) + "\n")
    shutil.copy(a.audit / "audit_data.json", tr / "data_audit.json")
    # evaluation: summary + raw per-item outputs of this model
    ev = out / "eval"
    ev.mkdir()
    for f in ("report.json", "report.md", "reference_rows.json", "speed.md", "speed_raw.log",
              "fullcase_report.json", "fullcase_report.md"):
        shutil.copy(a.audit / f, ev / f)
    (ev / "items").mkdir()
    for p in sorted((a.audit / "items").glob("v3*__*.jsonl")):
        if p.name.startswith(("v3__", "v3-engram_off__")):
            shutil.copy(p, ev / "items" / p.name)
    for p in sorted((a.audit / "fullcase").glob("v3-full_*__*.jsonl")):
        shutil.copy(p, ev / "items" / p.name)
    if (a.audit / "btzsc22").is_dir():  # external evaluation package (its own REPORT, PROTOCOL, SHA256SUMS)
        shutil.copytree(a.audit / "btzsc22", ev / "btzsc22")

    # guard: no internal path, user, host or address may leave in the bundle
    leaks = []
    for p in sorted(out.rglob("*")):
        if p.is_file() and p.suffix != ".safetensors":
            text = p.read_text(errors="ignore")
            for m in LEAK.finditer(text):
                leaks.append("%s: %s" % (p.relative_to(out), m.group(0)))
    if leaks:
        raise SystemExit("internal identifiers in the bundle:\n  " + "\n  ".join(leaks[:50]))

    # hashes, last
    files = sorted(p for p in out.rglob("*") if p.is_file() and p.name != "SHA256SUMS")
    (out / "SHA256SUMS").write_text("".join("%s  %s\n" % (sha256(p), p.relative_to(out).as_posix()) for p in files))
    print("bundle", out, "files", len(files) + 1, "params", n_params)


if __name__ == "__main__":
    main()