Ines-1 / training /code /release /build_release.py
Endikavi's picture
BTZSC-22 external zero-shot classification evaluation (eval/btzsc22/) and README section (release commit 329aac4)
ed926f1 verified
Raw History Blame Contribute Delete
11.2 kB
"""Assemble the Hugging Face release bundle of Ines-1 (codename mini-v41-Decisions) from COMMITTED code only.
python scripts/decisions/release/build_release.py --checkpoint CKPT_DIR --run RUN_DIR \
--parent-run PARENT_RUN_DIR --repo REPO --model-commit SHA --release-commit SHA \
--audit AUDIT_DIR --out OUT_DIR
Needs torch and safetensors. Every code file in the bundle comes from `git archive` of a commit:
mini_v41/ <- --model-commit (the modelling code)
mini_v41_jev/, scripts/, examples/, README.md, requirements.txt, .gitattributes
<- --release-commit (scripts/decisions/release/, decisions.py resolved)
training/code/ <- --release-commit (the scripts that built the data and trained the model)
Weights: the checkpoint's fp32 state dict cast to bf16 (inference already runs the model in bf16 on
CUDA, so this is the evaluated model). Internal absolute paths are reduced to file names.
"""
from __future__ import annotations
import argparse
import codecs
import hashlib
import io
import json
import re
import shutil
import subprocess
import tarfile
from pathlib import Path
# absolute data paths (not the "clean50k/da" "ta/" folder inside a dataset repo), home dirs, internal addresses, and the internal
# names below. They are stored rot13-encoded so that this file, which ships in training/code/, does not trip itself.
_NAMES = [codecs.decode(t, "rot13") for t in ['rivpvbfb', 'freire-vn', 'k630590', 'fjncraretvn', 'phfgbzwri',
'p4svk', 'x4v', 'frpergvn', 'frpergcvybg', 'fnovpb', 'freivat-pbasvt', '<guvf ercb']]
LEAK = re.compile(r"(?<![\w.])/da" r"ta/|(?<![\w.])/ho" r"me/|10\.20[01]\.\d|" + "|".join(map(re.escape, _NAMES)), re.I)
def sha256(p: Path) -> str:
h = hashlib.sha256()
with open(p, "rb") as f:
for chunk in iter(lambda: f.read(1 << 24), b""):
h.update(chunk)
return h.hexdigest()
def git_extract(repo: Path, commit: str, path: str, dest: Path, strip: str = "") -> list[str]:
"""Extract `path` at `commit` into `dest`, removing the `strip` prefix. Symlinks are resolved
against the same commit, so the bundle never contains a dangling link."""
data = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit, path],
check=True, capture_output=True).stdout
names = []
with tarfile.open(fileobj=io.BytesIO(data)) as tar:
for m in tar.getmembers():
if not m.isfile() and not m.issym():
continue
rel = m.name[len(strip):] if strip and m.name.startswith(strip) else m.name
target = dest / rel
target.parent.mkdir(parents=True, exist_ok=True)
if m.issym():
src = "/".join(_normpath((Path(m.name).parent / m.linkname).as_posix()))
blob = subprocess.run(["git", "-C", str(repo), "show", "%s:%s" % (commit, src)],
check=True, capture_output=True).stdout
target.write_bytes(blob)
else:
target.write_bytes(tar.extractfile(m).read())
names.append(rel)
return names
def _normpath(p: str) -> list[str]:
out = []
for part in p.split("/"):
if part == "..":
out.pop()
elif part not in ("", "."):
out.append(part)
return out
def strip_paths(x):
if isinstance(x, dict):
return {k: strip_paths(v) for k, v in x.items()}
if isinstance(x, list):
return [strip_paths(v) for v in x]
if isinstance(x, str) and x.startswith("/"):
return Path(x).name
return x
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--checkpoint", type=Path, required=True)
ap.add_argument("--run", type=Path, required=True, help="the training run directory (results.json)")
ap.add_argument("--parent-run", type=Path, required=True)
ap.add_argument("--repo", type=Path, required=True)
ap.add_argument("--model-commit", required=True)
ap.add_argument("--release-commit", required=True)
ap.add_argument("--audit", type=Path, required=True)
ap.add_argument("--out", type=Path, required=True)
ap.add_argument("--weights-from", type=Path, help="copy model.safetensors verbatim from this bundle (no re-export)")
ap.add_argument("--expect-weights-sha", help="required sha256 prefix of model.safetensors")
a = ap.parse_args()
import torch
from safetensors.torch import save_file
out = a.out
if out.exists():
raise SystemExit("%s exists: the bundle is always built into a new directory" % out)
out.mkdir(parents=True)
ck = a.checkpoint
for c in (a.model_commit, a.release_commit):
subprocess.run(["git", "-C", str(a.repo), "cat-file", "-e", c + "^{commit}"], check=True)
# weights
src_pt = ck / "model" / "model.pt"
if a.weights_from: # rename-only rebuilds: the published weights file is copied, never regenerated
shutil.copy(a.weights_from / "model.safetensors", out / "model.safetensors")
import math
from safetensors import safe_open
with safe_open(str(out / "model.safetensors"), "pt") as f:
n_params = sum(math.prod(f.get_slice(k).get_shape()) for k in f.keys())
else:
sd = torch.load(src_pt, map_location="cpu", weights_only=False)
tensors = {k: (v.detach().to(torch.bfloat16) if v.is_floating_point() else v.detach()).contiguous().clone()
for k, v in sd.items()}
save_file(tensors, str(out / "model.safetensors"), metadata={"format": "pt", "dtype": "bfloat16"})
n_params = sum(v.numel() for v in tensors.values())
if a.expect_weights_sha:
got = sha256(out / "model.safetensors")
if not got.startswith(a.expect_weights_sha):
raise SystemExit("model.safetensors sha256 %s != expected %s" % (got, a.expect_weights_sha))
cfg = json.loads((ck / "config.json").read_text())
man = json.loads((ck / "manifest.json").read_text())
ident = man["identity"]
(out / "config.json").write_text(json.dumps({
"name": "Ines-1",
"codename": "mini-v41 / mini-v41-Decisions",
"model_type": "mini-v41",
"architectures": ["MiniV41"],
"description": "1.59B-parameter MoE causal encoder-decoder (405M active per token) with Engram n-gram "
"memory, fine-tuned to answer typed decisions (choice / score / noul)",
"parameters_total": n_params,
"parameters_active_per_token": 405052224,
"model": cfg["model"],
"decision_reader": {"options": "letters A-Z read at the first assistant position", "max_options": 26,
"languages": ["en", "es"], "max_prompt_tokens": cfg["model"]["max_sequence_length"],
"questions_per_prompt": 1},
"provenance": {
"run_id": ident["run_id"], "parent_run_id": ident.get("parent_run_id"),
"parent_weights_sha256": ident.get("parent_checkpoint_sha256"),
"source_weights_sha256_fp32_pt": sha256(src_pt),
"selected_epoch": man["extra"].get("selected_epoch"), "val_score": man["extra"].get("val_score"),
"training_recipe_hash": ident.get("training_recipe_hash"),
"dataset_subset_id": ident.get("dataset_subset_id"),
"tokenizer_sha256": man["tokenizer_identity"]["sha256"],
"engram_backend_identity": man["engram_backend_identity"],
"training_code_commit": ident.get("code_commit"),
"bundle_model_code_commit": a.model_commit, "bundle_release_code_commit": a.release_commit,
"torch_version_at_training": man["torch_version"]},
}, indent=2) + "\n")
shutil.copytree(ck / "tokenizer", out / "tokenizer")
# code, from commits only
git_extract(a.repo, a.model_commit, "mini_v41", out)
(out / "LICENSE-CODE").write_bytes(subprocess.run(["git", "-C", str(a.repo), "show", a.model_commit + ":LICENSE"],
check=True, capture_output=True).stdout) # MIT, for the code
for p in list((out / "mini_v41").rglob("__pycache__")):
shutil.rmtree(p)
git_extract(a.repo, a.release_commit, "scripts/decisions/release", out, strip="scripts/decisions/release/")
(out / "build_release.py").unlink(missing_ok=True) # the builder is in training/code/, not in the package
for f in ("prepare_public.py", "prepare_mix.py", "train_decisions.py", "decisions.py", "eval_items.py",
"metrics.py", "reference_rows.py", "audit_data.py", "audit_report.py", "release/build_release.py"):
git_extract(a.repo, a.release_commit, "scripts/decisions/" + f, out / "training" / "code",
strip="scripts/decisions/")
git_extract(a.repo, a.release_commit, "scripts/decisions/launch", out / "training" / "code",
strip="scripts/decisions/")
# recipe and provenance of both fine-tuning stages
tr = out / "training"
for name, run in (("stage2_mix", a.run), ("stage1_typed_jev", a.parent_run)):
res = json.loads((run / "results.json").read_text())
(tr / ("%s.results.json" % name)).write_text(json.dumps(strip_paths(res), indent=1, ensure_ascii=False) + "\n")
m = json.loads((run / "checkpoint" / "manifest.json").read_text())
(tr / ("%s.manifest.json" % name)).write_text(json.dumps(strip_paths(m), indent=1) + "\n")
shutil.copy(a.audit / "audit_data.json", tr / "data_audit.json")
# evaluation: summary + raw per-item outputs of this model
ev = out / "eval"
ev.mkdir()
for f in ("report.json", "report.md", "reference_rows.json", "speed.md", "speed_raw.log",
"fullcase_report.json", "fullcase_report.md"):
shutil.copy(a.audit / f, ev / f)
(ev / "items").mkdir()
for p in sorted((a.audit / "items").glob("v3*__*.jsonl")):
if p.name.startswith(("v3__", "v3-engram_off__")):
shutil.copy(p, ev / "items" / p.name)
for p in sorted((a.audit / "fullcase").glob("v3-full_*__*.jsonl")):
shutil.copy(p, ev / "items" / p.name)
if (a.audit / "btzsc22").is_dir(): # external evaluation package (its own REPORT, PROTOCOL, SHA256SUMS)
shutil.copytree(a.audit / "btzsc22", ev / "btzsc22")
# guard: no internal path, user, host or address may leave in the bundle
leaks = []
for p in sorted(out.rglob("*")):
if p.is_file() and p.suffix != ".safetensors":
text = p.read_text(errors="ignore")
for m in LEAK.finditer(text):
leaks.append("%s: %s" % (p.relative_to(out), m.group(0)))
if leaks:
raise SystemExit("internal identifiers in the bundle:\n " + "\n ".join(leaks[:50]))
# hashes, last
files = sorted(p for p in out.rglob("*") if p.is_file() and p.name != "SHA256SUMS")
(out / "SHA256SUMS").write_text("".join("%s %s\n" % (sha256(p), p.relative_to(out).as_posix()) for p in files))
print("bundle", out, "files", len(files) + 1, "params", n_params)
if __name__ == "__main__":
main()