File size: 11,161 Bytes
61b6fb9 ed926f1 61b6fb9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 | """Assemble the Hugging Face release bundle of Ines-1 (codename mini-v41-Decisions) from COMMITTED code only.
python scripts/decisions/release/build_release.py --checkpoint CKPT_DIR --run RUN_DIR \
--parent-run PARENT_RUN_DIR --repo REPO --model-commit SHA --release-commit SHA \
--audit AUDIT_DIR --out OUT_DIR
Needs torch and safetensors. Every code file in the bundle comes from `git archive` of a commit:
mini_v41/ <- --model-commit (the modelling code)
mini_v41_jev/, scripts/, examples/, README.md, requirements.txt, .gitattributes
<- --release-commit (scripts/decisions/release/, decisions.py resolved)
training/code/ <- --release-commit (the scripts that built the data and trained the model)
Weights: the checkpoint's fp32 state dict cast to bf16 (inference already runs the model in bf16 on
CUDA, so this is the evaluated model). Internal absolute paths are reduced to file names.
"""
from __future__ import annotations
import argparse
import codecs
import hashlib
import io
import json
import re
import shutil
import subprocess
import tarfile
from pathlib import Path
# absolute data paths (not the "clean50k/da" "ta/" folder inside a dataset repo), home dirs, internal addresses, and the internal
# names below. They are stored rot13-encoded so that this file, which ships in training/code/, does not trip itself.
_NAMES = [codecs.decode(t, "rot13") for t in ['rivpvbfb', 'freire-vn', 'k630590', 'fjncraretvn', 'phfgbzwri',
'p4svk', 'x4v', 'frpergvn', 'frpergcvybg', 'fnovpb', 'freivat-pbasvt', '<guvf ercb']]
LEAK = re.compile(r"(?<![\w.])/da" r"ta/|(?<![\w.])/ho" r"me/|10\.20[01]\.\d|" + "|".join(map(re.escape, _NAMES)), re.I)
def sha256(p: Path) -> str:
h = hashlib.sha256()
with open(p, "rb") as f:
for chunk in iter(lambda: f.read(1 << 24), b""):
h.update(chunk)
return h.hexdigest()
def git_extract(repo: Path, commit: str, path: str, dest: Path, strip: str = "") -> list[str]:
"""Extract `path` at `commit` into `dest`, removing the `strip` prefix. Symlinks are resolved
against the same commit, so the bundle never contains a dangling link."""
data = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit, path],
check=True, capture_output=True).stdout
names = []
with tarfile.open(fileobj=io.BytesIO(data)) as tar:
for m in tar.getmembers():
if not m.isfile() and not m.issym():
continue
rel = m.name[len(strip):] if strip and m.name.startswith(strip) else m.name
target = dest / rel
target.parent.mkdir(parents=True, exist_ok=True)
if m.issym():
src = "/".join(_normpath((Path(m.name).parent / m.linkname).as_posix()))
blob = subprocess.run(["git", "-C", str(repo), "show", "%s:%s" % (commit, src)],
check=True, capture_output=True).stdout
target.write_bytes(blob)
else:
target.write_bytes(tar.extractfile(m).read())
names.append(rel)
return names
def _normpath(p: str) -> list[str]:
out = []
for part in p.split("/"):
if part == "..":
out.pop()
elif part not in ("", "."):
out.append(part)
return out
def strip_paths(x):
if isinstance(x, dict):
return {k: strip_paths(v) for k, v in x.items()}
if isinstance(x, list):
return [strip_paths(v) for v in x]
if isinstance(x, str) and x.startswith("/"):
return Path(x).name
return x
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--checkpoint", type=Path, required=True)
ap.add_argument("--run", type=Path, required=True, help="the training run directory (results.json)")
ap.add_argument("--parent-run", type=Path, required=True)
ap.add_argument("--repo", type=Path, required=True)
ap.add_argument("--model-commit", required=True)
ap.add_argument("--release-commit", required=True)
ap.add_argument("--audit", type=Path, required=True)
ap.add_argument("--out", type=Path, required=True)
ap.add_argument("--weights-from", type=Path, help="copy model.safetensors verbatim from this bundle (no re-export)")
ap.add_argument("--expect-weights-sha", help="required sha256 prefix of model.safetensors")
a = ap.parse_args()
import torch
from safetensors.torch import save_file
out = a.out
if out.exists():
raise SystemExit("%s exists: the bundle is always built into a new directory" % out)
out.mkdir(parents=True)
ck = a.checkpoint
for c in (a.model_commit, a.release_commit):
subprocess.run(["git", "-C", str(a.repo), "cat-file", "-e", c + "^{commit}"], check=True)
# weights
src_pt = ck / "model" / "model.pt"
if a.weights_from: # rename-only rebuilds: the published weights file is copied, never regenerated
shutil.copy(a.weights_from / "model.safetensors", out / "model.safetensors")
import math
from safetensors import safe_open
with safe_open(str(out / "model.safetensors"), "pt") as f:
n_params = sum(math.prod(f.get_slice(k).get_shape()) for k in f.keys())
else:
sd = torch.load(src_pt, map_location="cpu", weights_only=False)
tensors = {k: (v.detach().to(torch.bfloat16) if v.is_floating_point() else v.detach()).contiguous().clone()
for k, v in sd.items()}
save_file(tensors, str(out / "model.safetensors"), metadata={"format": "pt", "dtype": "bfloat16"})
n_params = sum(v.numel() for v in tensors.values())
if a.expect_weights_sha:
got = sha256(out / "model.safetensors")
if not got.startswith(a.expect_weights_sha):
raise SystemExit("model.safetensors sha256 %s != expected %s" % (got, a.expect_weights_sha))
cfg = json.loads((ck / "config.json").read_text())
man = json.loads((ck / "manifest.json").read_text())
ident = man["identity"]
(out / "config.json").write_text(json.dumps({
"name": "Ines-1",
"codename": "mini-v41 / mini-v41-Decisions",
"model_type": "mini-v41",
"architectures": ["MiniV41"],
"description": "1.59B-parameter MoE causal encoder-decoder (405M active per token) with Engram n-gram "
"memory, fine-tuned to answer typed decisions (choice / score / noul)",
"parameters_total": n_params,
"parameters_active_per_token": 405052224,
"model": cfg["model"],
"decision_reader": {"options": "letters A-Z read at the first assistant position", "max_options": 26,
"languages": ["en", "es"], "max_prompt_tokens": cfg["model"]["max_sequence_length"],
"questions_per_prompt": 1},
"provenance": {
"run_id": ident["run_id"], "parent_run_id": ident.get("parent_run_id"),
"parent_weights_sha256": ident.get("parent_checkpoint_sha256"),
"source_weights_sha256_fp32_pt": sha256(src_pt),
"selected_epoch": man["extra"].get("selected_epoch"), "val_score": man["extra"].get("val_score"),
"training_recipe_hash": ident.get("training_recipe_hash"),
"dataset_subset_id": ident.get("dataset_subset_id"),
"tokenizer_sha256": man["tokenizer_identity"]["sha256"],
"engram_backend_identity": man["engram_backend_identity"],
"training_code_commit": ident.get("code_commit"),
"bundle_model_code_commit": a.model_commit, "bundle_release_code_commit": a.release_commit,
"torch_version_at_training": man["torch_version"]},
}, indent=2) + "\n")
shutil.copytree(ck / "tokenizer", out / "tokenizer")
# code, from commits only
git_extract(a.repo, a.model_commit, "mini_v41", out)
(out / "LICENSE-CODE").write_bytes(subprocess.run(["git", "-C", str(a.repo), "show", a.model_commit + ":LICENSE"],
check=True, capture_output=True).stdout) # MIT, for the code
for p in list((out / "mini_v41").rglob("__pycache__")):
shutil.rmtree(p)
git_extract(a.repo, a.release_commit, "scripts/decisions/release", out, strip="scripts/decisions/release/")
(out / "build_release.py").unlink(missing_ok=True) # the builder is in training/code/, not in the package
for f in ("prepare_public.py", "prepare_mix.py", "train_decisions.py", "decisions.py", "eval_items.py",
"metrics.py", "reference_rows.py", "audit_data.py", "audit_report.py", "release/build_release.py"):
git_extract(a.repo, a.release_commit, "scripts/decisions/" + f, out / "training" / "code",
strip="scripts/decisions/")
git_extract(a.repo, a.release_commit, "scripts/decisions/launch", out / "training" / "code",
strip="scripts/decisions/")
# recipe and provenance of both fine-tuning stages
tr = out / "training"
for name, run in (("stage2_mix", a.run), ("stage1_typed_jev", a.parent_run)):
res = json.loads((run / "results.json").read_text())
(tr / ("%s.results.json" % name)).write_text(json.dumps(strip_paths(res), indent=1, ensure_ascii=False) + "\n")
m = json.loads((run / "checkpoint" / "manifest.json").read_text())
(tr / ("%s.manifest.json" % name)).write_text(json.dumps(strip_paths(m), indent=1) + "\n")
shutil.copy(a.audit / "audit_data.json", tr / "data_audit.json")
# evaluation: summary + raw per-item outputs of this model
ev = out / "eval"
ev.mkdir()
for f in ("report.json", "report.md", "reference_rows.json", "speed.md", "speed_raw.log",
"fullcase_report.json", "fullcase_report.md"):
shutil.copy(a.audit / f, ev / f)
(ev / "items").mkdir()
for p in sorted((a.audit / "items").glob("v3*__*.jsonl")):
if p.name.startswith(("v3__", "v3-engram_off__")):
shutil.copy(p, ev / "items" / p.name)
for p in sorted((a.audit / "fullcase").glob("v3-full_*__*.jsonl")):
shutil.copy(p, ev / "items" / p.name)
if (a.audit / "btzsc22").is_dir(): # external evaluation package (its own REPORT, PROTOCOL, SHA256SUMS)
shutil.copytree(a.audit / "btzsc22", ev / "btzsc22")
# guard: no internal path, user, host or address may leave in the bundle
leaks = []
for p in sorted(out.rglob("*")):
if p.is_file() and p.suffix != ".safetensors":
text = p.read_text(errors="ignore")
for m in LEAK.finditer(text):
leaks.append("%s: %s" % (p.relative_to(out), m.group(0)))
if leaks:
raise SystemExit("internal identifiers in the bundle:\n " + "\n ".join(leaks[:50]))
# hashes, last
files = sorted(p for p in out.rglob("*") if p.is_file() and p.name != "SHA256SUMS")
(out / "SHA256SUMS").write_text("".join("%s %s\n" % (sha256(p), p.relative_to(out).as_posix()) for p in files))
print("bundle", out, "files", len(files) + 1, "params", n_params)
if __name__ == "__main__":
main()
|