"""Assemble the Hugging Face release bundle of Ines-1 (codename mini-v41-Decisions) from COMMITTED code only. python scripts/decisions/release/build_release.py --checkpoint CKPT_DIR --run RUN_DIR \ --parent-run PARENT_RUN_DIR --repo REPO --model-commit SHA --release-commit SHA \ --audit AUDIT_DIR --out OUT_DIR Needs torch and safetensors. Every code file in the bundle comes from `git archive` of a commit: mini_v41/ <- --model-commit (the modelling code) mini_v41_jev/, scripts/, examples/, README.md, requirements.txt, .gitattributes <- --release-commit (scripts/decisions/release/, decisions.py resolved) training/code/ <- --release-commit (the scripts that built the data and trained the model) Weights: the checkpoint's fp32 state dict cast to bf16 (inference already runs the model in bf16 on CUDA, so this is the evaluated model). Internal absolute paths are reduced to file names. """ from __future__ import annotations import argparse import codecs import hashlib import io import json import re import shutil import subprocess import tarfile from pathlib import Path # absolute data paths (not the "clean50k/da" "ta/" folder inside a dataset repo), home dirs, internal addresses, and the internal # names below. They are stored rot13-encoded so that this file, which ships in training/code/, does not trip itself. _NAMES = [codecs.decode(t, "rot13") for t in ['rivpvbfb', 'freire-vn', 'k630590', 'fjncraretvn', 'phfgbzwri', 'p4svk', 'x4v', 'frpergvn', 'frpergcvybg', 'fnovpb', 'freivat-pbasvt', ' str: h = hashlib.sha256() with open(p, "rb") as f: for chunk in iter(lambda: f.read(1 << 24), b""): h.update(chunk) return h.hexdigest() def git_extract(repo: Path, commit: str, path: str, dest: Path, strip: str = "") -> list[str]: """Extract `path` at `commit` into `dest`, removing the `strip` prefix. Symlinks are resolved against the same commit, so the bundle never contains a dangling link.""" data = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit, path], check=True, capture_output=True).stdout names = [] with tarfile.open(fileobj=io.BytesIO(data)) as tar: for m in tar.getmembers(): if not m.isfile() and not m.issym(): continue rel = m.name[len(strip):] if strip and m.name.startswith(strip) else m.name target = dest / rel target.parent.mkdir(parents=True, exist_ok=True) if m.issym(): src = "/".join(_normpath((Path(m.name).parent / m.linkname).as_posix())) blob = subprocess.run(["git", "-C", str(repo), "show", "%s:%s" % (commit, src)], check=True, capture_output=True).stdout target.write_bytes(blob) else: target.write_bytes(tar.extractfile(m).read()) names.append(rel) return names def _normpath(p: str) -> list[str]: out = [] for part in p.split("/"): if part == "..": out.pop() elif part not in ("", "."): out.append(part) return out def strip_paths(x): if isinstance(x, dict): return {k: strip_paths(v) for k, v in x.items()} if isinstance(x, list): return [strip_paths(v) for v in x] if isinstance(x, str) and x.startswith("/"): return Path(x).name return x def main(): ap = argparse.ArgumentParser() ap.add_argument("--checkpoint", type=Path, required=True) ap.add_argument("--run", type=Path, required=True, help="the training run directory (results.json)") ap.add_argument("--parent-run", type=Path, required=True) ap.add_argument("--repo", type=Path, required=True) ap.add_argument("--model-commit", required=True) ap.add_argument("--release-commit", required=True) ap.add_argument("--audit", type=Path, required=True) ap.add_argument("--out", type=Path, required=True) ap.add_argument("--weights-from", type=Path, help="copy model.safetensors verbatim from this bundle (no re-export)") ap.add_argument("--expect-weights-sha", help="required sha256 prefix of model.safetensors") a = ap.parse_args() import torch from safetensors.torch import save_file out = a.out if out.exists(): raise SystemExit("%s exists: the bundle is always built into a new directory" % out) out.mkdir(parents=True) ck = a.checkpoint for c in (a.model_commit, a.release_commit): subprocess.run(["git", "-C", str(a.repo), "cat-file", "-e", c + "^{commit}"], check=True) # weights src_pt = ck / "model" / "model.pt" if a.weights_from: # rename-only rebuilds: the published weights file is copied, never regenerated shutil.copy(a.weights_from / "model.safetensors", out / "model.safetensors") import math from safetensors import safe_open with safe_open(str(out / "model.safetensors"), "pt") as f: n_params = sum(math.prod(f.get_slice(k).get_shape()) for k in f.keys()) else: sd = torch.load(src_pt, map_location="cpu", weights_only=False) tensors = {k: (v.detach().to(torch.bfloat16) if v.is_floating_point() else v.detach()).contiguous().clone() for k, v in sd.items()} save_file(tensors, str(out / "model.safetensors"), metadata={"format": "pt", "dtype": "bfloat16"}) n_params = sum(v.numel() for v in tensors.values()) if a.expect_weights_sha: got = sha256(out / "model.safetensors") if not got.startswith(a.expect_weights_sha): raise SystemExit("model.safetensors sha256 %s != expected %s" % (got, a.expect_weights_sha)) cfg = json.loads((ck / "config.json").read_text()) man = json.loads((ck / "manifest.json").read_text()) ident = man["identity"] (out / "config.json").write_text(json.dumps({ "name": "Ines-1", "codename": "mini-v41 / mini-v41-Decisions", "model_type": "mini-v41", "architectures": ["MiniV41"], "description": "1.59B-parameter MoE causal encoder-decoder (405M active per token) with Engram n-gram " "memory, fine-tuned to answer typed decisions (choice / score / noul)", "parameters_total": n_params, "parameters_active_per_token": 405052224, "model": cfg["model"], "decision_reader": {"options": "letters A-Z read at the first assistant position", "max_options": 26, "languages": ["en", "es"], "max_prompt_tokens": cfg["model"]["max_sequence_length"], "questions_per_prompt": 1}, "provenance": { "run_id": ident["run_id"], "parent_run_id": ident.get("parent_run_id"), "parent_weights_sha256": ident.get("parent_checkpoint_sha256"), "source_weights_sha256_fp32_pt": sha256(src_pt), "selected_epoch": man["extra"].get("selected_epoch"), "val_score": man["extra"].get("val_score"), "training_recipe_hash": ident.get("training_recipe_hash"), "dataset_subset_id": ident.get("dataset_subset_id"), "tokenizer_sha256": man["tokenizer_identity"]["sha256"], "engram_backend_identity": man["engram_backend_identity"], "training_code_commit": ident.get("code_commit"), "bundle_model_code_commit": a.model_commit, "bundle_release_code_commit": a.release_commit, "torch_version_at_training": man["torch_version"]}, }, indent=2) + "\n") shutil.copytree(ck / "tokenizer", out / "tokenizer") # code, from commits only git_extract(a.repo, a.model_commit, "mini_v41", out) (out / "LICENSE-CODE").write_bytes(subprocess.run(["git", "-C", str(a.repo), "show", a.model_commit + ":LICENSE"], check=True, capture_output=True).stdout) # MIT, for the code for p in list((out / "mini_v41").rglob("__pycache__")): shutil.rmtree(p) git_extract(a.repo, a.release_commit, "scripts/decisions/release", out, strip="scripts/decisions/release/") (out / "build_release.py").unlink(missing_ok=True) # the builder is in training/code/, not in the package for f in ("prepare_public.py", "prepare_mix.py", "train_decisions.py", "decisions.py", "eval_items.py", "metrics.py", "reference_rows.py", "audit_data.py", "audit_report.py", "release/build_release.py"): git_extract(a.repo, a.release_commit, "scripts/decisions/" + f, out / "training" / "code", strip="scripts/decisions/") git_extract(a.repo, a.release_commit, "scripts/decisions/launch", out / "training" / "code", strip="scripts/decisions/") # recipe and provenance of both fine-tuning stages tr = out / "training" for name, run in (("stage2_mix", a.run), ("stage1_typed_jev", a.parent_run)): res = json.loads((run / "results.json").read_text()) (tr / ("%s.results.json" % name)).write_text(json.dumps(strip_paths(res), indent=1, ensure_ascii=False) + "\n") m = json.loads((run / "checkpoint" / "manifest.json").read_text()) (tr / ("%s.manifest.json" % name)).write_text(json.dumps(strip_paths(m), indent=1) + "\n") shutil.copy(a.audit / "audit_data.json", tr / "data_audit.json") # evaluation: summary + raw per-item outputs of this model ev = out / "eval" ev.mkdir() for f in ("report.json", "report.md", "reference_rows.json", "speed.md", "speed_raw.log", "fullcase_report.json", "fullcase_report.md"): shutil.copy(a.audit / f, ev / f) (ev / "items").mkdir() for p in sorted((a.audit / "items").glob("v3*__*.jsonl")): if p.name.startswith(("v3__", "v3-engram_off__")): shutil.copy(p, ev / "items" / p.name) for p in sorted((a.audit / "fullcase").glob("v3-full_*__*.jsonl")): shutil.copy(p, ev / "items" / p.name) if (a.audit / "btzsc22").is_dir(): # external evaluation package (its own REPORT, PROTOCOL, SHA256SUMS) shutil.copytree(a.audit / "btzsc22", ev / "btzsc22") # guard: no internal path, user, host or address may leave in the bundle leaks = [] for p in sorted(out.rglob("*")): if p.is_file() and p.suffix != ".safetensors": text = p.read_text(errors="ignore") for m in LEAK.finditer(text): leaks.append("%s: %s" % (p.relative_to(out), m.group(0))) if leaks: raise SystemExit("internal identifiers in the bundle:\n " + "\n ".join(leaks[:50])) # hashes, last files = sorted(p for p in out.rglob("*") if p.is_file() and p.name != "SHA256SUMS") (out / "SHA256SUMS").write_text("".join("%s %s\n" % (sha256(p), p.relative_to(out).as_posix()) for p in files)) print("bundle", out, "files", len(files) + 1, "params", n_params) if __name__ == "__main__": main()