Cion-lab commited on
Commit
c12486d
·
verified ·
1 Parent(s): 9f37a2c

add publish_mix: LFS-first upload, generated card + attribution, Hub-side verification

Browse files
Files changed (1) hide show
  1. build/publish_mix.py +198 -0
build/publish_mix.py ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Phase 2 publisher: mix + manifest + card -> public `ounce100m-mix-v1`, then verified from the Hub.
2
+
3
+ import argparse
4
+ import hashlib
5
+ import json
6
+ import os
7
+ import struct
8
+ import sys
9
+ import urllib.request
10
+
11
+ # licence per (repo, config), as read from the Hub cards on 2026-09-19. The mix contains CC-BY-SA-4.0
12
+ # material, so the derived dataset inherits SA -- see docs/02-mix-plan.md §6.
13
+ LICENCES = {
14
+ "HuggingFaceFW/fineweb-edu": "ODC-By-1.0",
15
+ "HuggingFaceFW/finepdfs-edu": "ODC-By-1.0 + Common Crawl Terms of Use",
16
+ "HuggingFaceTB/cosmopedia": "Apache-2.0",
17
+ "HuggingFaceFW/finewiki": "CC-BY-SA-4.0",
18
+ "omarkamali/wikipedia-monthly": "CC-BY-SA-4.0",
19
+ "HuggingFaceTB/finemath": "ODC-By-1.0",
20
+ "open-web-math/open-web-math": "ODC-By (card body; cardData.license empty)",
21
+ "HuggingFaceFW/finephrase": "ODC-By-1.0",
22
+ "SimpleStories/SimpleStories": "MIT",
23
+ "common-pile/arxiv_abstracts": "CC0 (arXiv metadata)",
24
+ }
25
+
26
+
27
+ def card(man, repo):
28
+ tot = man["total_tokens"]
29
+ vt = man.get("val_tokens", 0)
30
+ rows = []
31
+ for k, v in sorted(man.get("stage", {}).items()):
32
+ rows.append(f"| {k} | {v.get('tokens', 0):,} | {v.get('docs_seen', 0):,} | "
33
+ f"{v.get('complete', False)} |")
34
+ src_table = "\n".join(rows) or "| (none) | | | |"
35
+ return f"""---
36
+ license: cc-by-sa-4.0
37
+ pretty_name: ounce100m pretraining mix v1
38
+ tags: [pretraining, tokenized, english, ounce100m]
39
+ ---
40
+
41
+ # {repo.split('/')[1]}
42
+
43
+ Tokenised, deduplicated English pre-training mix built for project **ounce100m** — a 90–110M-parameter
44
+ decoder-only base model trained from scratch on ~1B tokens on 2x Tesla T4. This dataset is the exact
45
+ input to that run, published so the result can be reproduced.
46
+
47
+ - **Training tokens:** {tot:,} across {len(man['shards'])} shards
48
+ - **Held-out validation tokens:** {vt:,} in {len(man.get('val_shards', []))} shards (`val/`), sampled
49
+ every {man.get('val_holdout', {}).get('stride', '?')}th merged document — across sources, never whole
50
+ documents, and excluded from the training shards
51
+ - **Tokenizer:** `{man['tokenizer']['id']}` — vocab {man['tokenizer']['vocab_size']:,},
52
+ sha256 `{man['tokenizer']['sha256'][:16]}…`
53
+ - **Built:** {man.get('built_utc', 'see manifest')}
54
+
55
+ ## Format
56
+
57
+ Raw binary, no wrapper library. Each shard:
58
+
59
+ ```
60
+ header : struct '<IIII' = vocab, n_docs, n_tokens, flags (16 bytes)
61
+ offsets : uint32 x (n_docs + 1) cumulative token positions, [0] = 0
62
+ ids : uint16 x n_tokens SmolLM2 token ids
63
+ ```
64
+
65
+ Document *i* of a shard is `ids[offsets[i]:offsets[i+1]]`. Concatenating the shards in filename order
66
+ gives the training stream; a training position is exactly `(shard_index, token_offset)`, which is what
67
+ lets the run resume at the identical data position after an interruption.
68
+
69
+ ```python
70
+ import struct, array
71
+ with open(path, "rb") as f:
72
+ vocab, n_docs, n_tokens, flags = struct.unpack("<IIII", f.read(16))
73
+ offs = array.array("I"); offs.fromfile(f, n_docs + 1)
74
+ ids = array.array("H"); ids.fromfile(f, n_tokens)
75
+ ```
76
+
77
+ ## Composition
78
+
79
+ | source | tokens staged | docs seen | complete |
80
+ |---|---|---|---|
81
+ {src_table}
82
+
83
+ Per-shard contents, source mix and sha256 digests are in `manifest.json`. Selection rules and the
84
+ reasoning behind each source are in the project's `docs/02-mix-plan.md`; licence attribution and the
85
+ transformations applied to each source are in `ATTRIBUTION.md`.
86
+
87
+ ## Provenance and hygiene
88
+
89
+ - Built by streaming public datasets on the Hugging Face Hub, quality/English-gated, deduplicated by
90
+ exact document hash and by source url/id.
91
+ - **No benchmark content.** Overlap was computed mechanically against eval *train/validation/dev*
92
+ material only; test splits were never opened during construction. Counts are in `audit.json`.
93
+ - No raw web crawl: web-derived sources here are classifier-scored or curated subsets.
94
+ - Trained model: see the `ounce100m-*` model repos in the same namespace.
95
+ """
96
+
97
+
98
+ ATTRIBUTION = """# Attribution
99
+
100
+ This is a derived dataset. Each source keeps its own licence; the mix as a whole is offered under
101
+ CC-BY-SA-4.0 because it contains CC-BY-SA-4.0 material (finewiki, wikipedia-monthly), whose share-alike
102
+ clause applies to derivatives.
103
+
104
+ | source | licence | transformation applied here |
105
+ |---|---|---|
106
+ """
107
+
108
+
109
+ def attribution(man):
110
+ lines = []
111
+ for k, v in sorted(man.get("stage", {}).items()):
112
+ repo = v.get("repo") or k.split("__")[0]
113
+ lic = LICENCES.get(repo, "see card")
114
+ lines.append(f"| `{k}` ({repo}) | {lic} | streamed, whitespace-normalised, "
115
+ f"length/quality gated, exact- and url-deduplicated, tokenised to uint16 |")
116
+ return ATTRIBUTION + "\n".join(lines) + "\n"
117
+
118
+
119
+ def main():
120
+ ap = argparse.ArgumentParser()
121
+ ap.add_argument("--root", default="/kaggle/working/mixroot")
122
+ ap.add_argument("--repo", default="Cion-lab/ounce100m-mix-v1")
123
+ ap.add_argument("--dry-run", action="store_true")
124
+ a = ap.parse_args()
125
+
126
+ import ounce100m_credentials
127
+ ounce100m_credentials.install(verify=True)
128
+ tok = os.environ["HF_TOKEN"]
129
+ from huggingface_hub import HfApi, create_repo
130
+
131
+ man = json.load(open(os.path.join(a.root, "manifest.json")))
132
+ if not man.get("shards"):
133
+ print("FATAL: refusing to publish a mix with zero shards", flush=True)
134
+ sys.exit(5)
135
+ man["built_utc"] = __import__("time").strftime("%Y-%m-%dT%H:%M:%SZ", __import__("time").gmtime())
136
+ with open(os.path.join(a.root, "manifest.json"), "w") as f:
137
+ json.dump(man, f, indent=1)
138
+ open(os.path.join(a.root, "README.md"), "w").write(card(man, a.repo))
139
+ open(os.path.join(a.root, "ATTRIBUTION.md"), "w").write(attribution(man))
140
+
141
+ api = HfApi(token=tok)
142
+ owner, name = a.repo.split("/", 1)
143
+ if not a.dry_run:
144
+ create_repo(repo_id=a.repo, repo_type="dataset", private=False, exist_ok=True, token=tok)
145
+ # .gitattributes must exist before the first .bin, or a >10MB shard is rejected as
146
+ # "should be tracked by LFS" rather than uploaded as LFS.
147
+ api.upload_file(path_or_fileobj=b"*.bin filter=lfs diff=lfs merge=lfs -text\n",
148
+ path_in_repo=".gitattributes", repo_id=a.repo, repo_type="dataset",
149
+ commit_message="track shard binaries with LFS")
150
+
151
+ to_upload = []
152
+ for grp in ("shards", "val_shards"):
153
+ for rec in man.get(grp, []):
154
+ to_upload.append((rec["file"], os.path.join(a.root, rec["file"])))
155
+ for extra in ("manifest.json", "README.md", "ATTRIBUTION.md", "audit.json",
156
+ "build_state.json"):
157
+ p = os.path.join(a.root, extra)
158
+ if os.path.exists(p):
159
+ to_upload.append((extra, p))
160
+ print(f"uploading {len(to_upload)} files to {a.repo}", flush=True)
161
+ if a.dry_run:
162
+ for np, lp in to_upload:
163
+ print(" would upload", np, os.path.getsize(lp), flush=True)
164
+ return
165
+
166
+ for np, lp in to_upload:
167
+ with open(lp, "rb") as fh:
168
+ payload = fh.read()
169
+ api.upload_file(path_or_fileobj=payload, path_in_repo=np, repo_id=a.repo,
170
+ repo_type="dataset", commit_message=f"add {np}")
171
+ print(f" + {np} ({len(payload) / 1024**2:.1f} MB)", flush=True)
172
+
173
+ # Independent verification from the Hub, not from the client that just wrote it.
174
+ remote = set(api.list_repo_files(repo_id=a.repo, repo_type="dataset"))
175
+ want = {np for np, _ in to_upload} | {".gitattributes"}
176
+ R = {"repo": a.repo, "files_expected": len(want), "files_present": len(want & remote),
177
+ "missing": sorted(want - remote)}
178
+ first = man["shards"][0]["file"]
179
+ url = f"https://huggingface.co/datasets/{a.repo}/resolve/main/{first}"
180
+ try:
181
+ with urllib.request.urlopen(url, timeout=300) as r:
182
+ body = r.read()
183
+ hv = struct.unpack("<IIII", body[:16])
184
+ R["anonymous_shard_read"] = {"bytes": len(body), "header": list(hv),
185
+ "n_tokens_matches_manifest": hv[2] == man["shards"][0]["tokens"],
186
+ "sha256_matches_manifest": hashlib.sha256(body).hexdigest()
187
+ == man["shards"][0]["sha256"]}
188
+ except Exception as e:
189
+ R["anonymous_shard_read"] = {"error": f"{type(e).__name__}: {str(e)[:160]}"}
190
+ R["published_verified"] = bool(not R["missing"] and R["anonymous_shard_read"].get(
191
+ "sha256_matches_manifest"))
192
+ print("PUBLISH_JSON_BEGIN")
193
+ print(json.dumps(R, indent=1, default=str))
194
+ print("PUBLISH_JSON_END")
195
+
196
+
197
+ if __name__ == "__main__":
198
+ main()