Download training/code/release/build_release.py from Endikavi/Ines-1: direct link, hf CLI and curl.
- Browser
- Download file 11.2 kB
-
https://huggingface.co/Endikavi/Ines-1/resolve/main/training/code/release/build_release.py
- Command line
-
hf download hf://Endikavi/Ines-1/training/code/release/build_release.py
-
curl -L -o build_release.py https://huggingface.co/Endikavi/Ines-1/resolve/main/training/code/release/build_release.py
11.2 kB
| """Assemble the Hugging Face release bundle of Ines-1 (codename mini-v41-Decisions) from COMMITTED code only. | |
| python scripts/decisions/release/build_release.py --checkpoint CKPT_DIR --run RUN_DIR \ | |
| --parent-run PARENT_RUN_DIR --repo REPO --model-commit SHA --release-commit SHA \ | |
| --audit AUDIT_DIR --out OUT_DIR | |
| Needs torch and safetensors. Every code file in the bundle comes from `git archive` of a commit: | |
| mini_v41/ <- --model-commit (the modelling code) | |
| mini_v41_jev/, scripts/, examples/, README.md, requirements.txt, .gitattributes | |
| <- --release-commit (scripts/decisions/release/, decisions.py resolved) | |
| training/code/ <- --release-commit (the scripts that built the data and trained the model) | |
| Weights: the checkpoint's fp32 state dict cast to bf16 (inference already runs the model in bf16 on | |
| CUDA, so this is the evaluated model). Internal absolute paths are reduced to file names. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import codecs | |
| import hashlib | |
| import io | |
| import json | |
| import re | |
| import shutil | |
| import subprocess | |
| import tarfile | |
| from pathlib import Path | |
| # absolute data paths (not the "clean50k/da" "ta/" folder inside a dataset repo), home dirs, internal addresses, and the internal | |
| # names below. They are stored rot13-encoded so that this file, which ships in training/code/, does not trip itself. | |
| _NAMES = [codecs.decode(t, "rot13") for t in ['rivpvbfb', 'freire-vn', 'k630590', 'fjncraretvn', 'phfgbzwri', | |
| 'p4svk', 'x4v', 'frpergvn', 'frpergcvybg', 'fnovpb', 'freivat-pbasvt', '<guvf ercb']] | |
| LEAK = re.compile(r"(?<![\w.])/da" r"ta/|(?<![\w.])/ho" r"me/|10\.20[01]\.\d|" + "|".join(map(re.escape, _NAMES)), re.I) | |
| def sha256(p: Path) -> str: | |
| h = hashlib.sha256() | |
| with open(p, "rb") as f: | |
| for chunk in iter(lambda: f.read(1 << 24), b""): | |
| h.update(chunk) | |
| return h.hexdigest() | |
| def git_extract(repo: Path, commit: str, path: str, dest: Path, strip: str = "") -> list[str]: | |
| """Extract `path` at `commit` into `dest`, removing the `strip` prefix. Symlinks are resolved | |
| against the same commit, so the bundle never contains a dangling link.""" | |
| data = subprocess.run(["git", "-C", str(repo), "archive", "--format=tar", commit, path], | |
| check=True, capture_output=True).stdout | |
| names = [] | |
| with tarfile.open(fileobj=io.BytesIO(data)) as tar: | |
| for m in tar.getmembers(): | |
| if not m.isfile() and not m.issym(): | |
| continue | |
| rel = m.name[len(strip):] if strip and m.name.startswith(strip) else m.name | |
| target = dest / rel | |
| target.parent.mkdir(parents=True, exist_ok=True) | |
| if m.issym(): | |
| src = "/".join(_normpath((Path(m.name).parent / m.linkname).as_posix())) | |
| blob = subprocess.run(["git", "-C", str(repo), "show", "%s:%s" % (commit, src)], | |
| check=True, capture_output=True).stdout | |
| target.write_bytes(blob) | |
| else: | |
| target.write_bytes(tar.extractfile(m).read()) | |
| names.append(rel) | |
| return names | |
| def _normpath(p: str) -> list[str]: | |
| out = [] | |
| for part in p.split("/"): | |
| if part == "..": | |
| out.pop() | |
| elif part not in ("", "."): | |
| out.append(part) | |
| return out | |
| def strip_paths(x): | |
| if isinstance(x, dict): | |
| return {k: strip_paths(v) for k, v in x.items()} | |
| if isinstance(x, list): | |
| return [strip_paths(v) for v in x] | |
| if isinstance(x, str) and x.startswith("/"): | |
| return Path(x).name | |
| return x | |
| def main(): | |
| ap = argparse.ArgumentParser() | |
| ap.add_argument("--checkpoint", type=Path, required=True) | |
| ap.add_argument("--run", type=Path, required=True, help="the training run directory (results.json)") | |
| ap.add_argument("--parent-run", type=Path, required=True) | |
| ap.add_argument("--repo", type=Path, required=True) | |
| ap.add_argument("--model-commit", required=True) | |
| ap.add_argument("--release-commit", required=True) | |
| ap.add_argument("--audit", type=Path, required=True) | |
| ap.add_argument("--out", type=Path, required=True) | |
| ap.add_argument("--weights-from", type=Path, help="copy model.safetensors verbatim from this bundle (no re-export)") | |
| ap.add_argument("--expect-weights-sha", help="required sha256 prefix of model.safetensors") | |
| a = ap.parse_args() | |
| import torch | |
| from safetensors.torch import save_file | |
| out = a.out | |
| if out.exists(): | |
| raise SystemExit("%s exists: the bundle is always built into a new directory" % out) | |
| out.mkdir(parents=True) | |
| ck = a.checkpoint | |
| for c in (a.model_commit, a.release_commit): | |
| subprocess.run(["git", "-C", str(a.repo), "cat-file", "-e", c + "^{commit}"], check=True) | |
| # weights | |
| src_pt = ck / "model" / "model.pt" | |
| if a.weights_from: # rename-only rebuilds: the published weights file is copied, never regenerated | |
| shutil.copy(a.weights_from / "model.safetensors", out / "model.safetensors") | |
| import math | |
| from safetensors import safe_open | |
| with safe_open(str(out / "model.safetensors"), "pt") as f: | |
| n_params = sum(math.prod(f.get_slice(k).get_shape()) for k in f.keys()) | |
| else: | |
| sd = torch.load(src_pt, map_location="cpu", weights_only=False) | |
| tensors = {k: (v.detach().to(torch.bfloat16) if v.is_floating_point() else v.detach()).contiguous().clone() | |
| for k, v in sd.items()} | |
| save_file(tensors, str(out / "model.safetensors"), metadata={"format": "pt", "dtype": "bfloat16"}) | |
| n_params = sum(v.numel() for v in tensors.values()) | |
| if a.expect_weights_sha: | |
| got = sha256(out / "model.safetensors") | |
| if not got.startswith(a.expect_weights_sha): | |
| raise SystemExit("model.safetensors sha256 %s != expected %s" % (got, a.expect_weights_sha)) | |
| cfg = json.loads((ck / "config.json").read_text()) | |
| man = json.loads((ck / "manifest.json").read_text()) | |
| ident = man["identity"] | |
| (out / "config.json").write_text(json.dumps({ | |
| "name": "Ines-1", | |
| "codename": "mini-v41 / mini-v41-Decisions", | |
| "model_type": "mini-v41", | |
| "architectures": ["MiniV41"], | |
| "description": "1.59B-parameter MoE causal encoder-decoder (405M active per token) with Engram n-gram " | |
| "memory, fine-tuned to answer typed decisions (choice / score / noul)", | |
| "parameters_total": n_params, | |
| "parameters_active_per_token": 405052224, | |
| "model": cfg["model"], | |
| "decision_reader": {"options": "letters A-Z read at the first assistant position", "max_options": 26, | |
| "languages": ["en", "es"], "max_prompt_tokens": cfg["model"]["max_sequence_length"], | |
| "questions_per_prompt": 1}, | |
| "provenance": { | |
| "run_id": ident["run_id"], "parent_run_id": ident.get("parent_run_id"), | |
| "parent_weights_sha256": ident.get("parent_checkpoint_sha256"), | |
| "source_weights_sha256_fp32_pt": sha256(src_pt), | |
| "selected_epoch": man["extra"].get("selected_epoch"), "val_score": man["extra"].get("val_score"), | |
| "training_recipe_hash": ident.get("training_recipe_hash"), | |
| "dataset_subset_id": ident.get("dataset_subset_id"), | |
| "tokenizer_sha256": man["tokenizer_identity"]["sha256"], | |
| "engram_backend_identity": man["engram_backend_identity"], | |
| "training_code_commit": ident.get("code_commit"), | |
| "bundle_model_code_commit": a.model_commit, "bundle_release_code_commit": a.release_commit, | |
| "torch_version_at_training": man["torch_version"]}, | |
| }, indent=2) + "\n") | |
| shutil.copytree(ck / "tokenizer", out / "tokenizer") | |
| # code, from commits only | |
| git_extract(a.repo, a.model_commit, "mini_v41", out) | |
| (out / "LICENSE-CODE").write_bytes(subprocess.run(["git", "-C", str(a.repo), "show", a.model_commit + ":LICENSE"], | |
| check=True, capture_output=True).stdout) # MIT, for the code | |
| for p in list((out / "mini_v41").rglob("__pycache__")): | |
| shutil.rmtree(p) | |
| git_extract(a.repo, a.release_commit, "scripts/decisions/release", out, strip="scripts/decisions/release/") | |
| (out / "build_release.py").unlink(missing_ok=True) # the builder is in training/code/, not in the package | |
| for f in ("prepare_public.py", "prepare_mix.py", "train_decisions.py", "decisions.py", "eval_items.py", | |
| "metrics.py", "reference_rows.py", "audit_data.py", "audit_report.py", "release/build_release.py"): | |
| git_extract(a.repo, a.release_commit, "scripts/decisions/" + f, out / "training" / "code", | |
| strip="scripts/decisions/") | |
| git_extract(a.repo, a.release_commit, "scripts/decisions/launch", out / "training" / "code", | |
| strip="scripts/decisions/") | |
| # recipe and provenance of both fine-tuning stages | |
| tr = out / "training" | |
| for name, run in (("stage2_mix", a.run), ("stage1_typed_jev", a.parent_run)): | |
| res = json.loads((run / "results.json").read_text()) | |
| (tr / ("%s.results.json" % name)).write_text(json.dumps(strip_paths(res), indent=1, ensure_ascii=False) + "\n") | |
| m = json.loads((run / "checkpoint" / "manifest.json").read_text()) | |
| (tr / ("%s.manifest.json" % name)).write_text(json.dumps(strip_paths(m), indent=1) + "\n") | |
| shutil.copy(a.audit / "audit_data.json", tr / "data_audit.json") | |
| # evaluation: summary + raw per-item outputs of this model | |
| ev = out / "eval" | |
| ev.mkdir() | |
| for f in ("report.json", "report.md", "reference_rows.json", "speed.md", "speed_raw.log", | |
| "fullcase_report.json", "fullcase_report.md"): | |
| shutil.copy(a.audit / f, ev / f) | |
| (ev / "items").mkdir() | |
| for p in sorted((a.audit / "items").glob("v3*__*.jsonl")): | |
| if p.name.startswith(("v3__", "v3-engram_off__")): | |
| shutil.copy(p, ev / "items" / p.name) | |
| for p in sorted((a.audit / "fullcase").glob("v3-full_*__*.jsonl")): | |
| shutil.copy(p, ev / "items" / p.name) | |
| if (a.audit / "btzsc22").is_dir(): # external evaluation package (its own REPORT, PROTOCOL, SHA256SUMS) | |
| shutil.copytree(a.audit / "btzsc22", ev / "btzsc22") | |
| # guard: no internal path, user, host or address may leave in the bundle | |
| leaks = [] | |
| for p in sorted(out.rglob("*")): | |
| if p.is_file() and p.suffix != ".safetensors": | |
| text = p.read_text(errors="ignore") | |
| for m in LEAK.finditer(text): | |
| leaks.append("%s: %s" % (p.relative_to(out), m.group(0))) | |
| if leaks: | |
| raise SystemExit("internal identifiers in the bundle:\n " + "\n ".join(leaks[:50])) | |
| # hashes, last | |
| files = sorted(p for p in out.rglob("*") if p.is_file() and p.name != "SHA256SUMS") | |
| (out / "SHA256SUMS").write_text("".join("%s %s\n" % (sha256(p), p.relative_to(out).as_posix()) for p in files)) | |
| print("bundle", out, "files", len(files) + 1, "params", n_params) | |
| if __name__ == "__main__": | |
| main() | |