Download probes/p2_source_inventory.py from Cion-lab/ounce100m-code: direct link, hf CLI and curl.
- Browser
- Download file 7.97 kB
-
https://huggingface.co/Cion-lab/ounce100m-code/resolve/main/probes/p2_source_inventory.py
- Command line
-
hf download hf://Cion-lab/ounce100m-code/probes/p2_source_inventory.py
-
curl -L -o p2_source_inventory.py https://huggingface.co/Cion-lab/ounce100m-code/resolve/main/probes/p2_source_inventory.py
7.97 kB
| # Phase 2 step 1: inventory the mix's sources with our own measurements, not their cards. | |
| # | |
| # Why before building. docs/02-mix-plan.md §4 records that two candidate sources disagree with | |
| # themselves (cosmopedia-v2 card 17.8 B vs ~31.7 B implied by stored columns; finephrase's | |
| # `token_count` describes the *source* document, not the rewrite). The mix's token arithmetic therefore | |
| # cannot be inherited -- it needs tokens/document measured with the tokenizer we actually froze, from the | |
| # rows we will actually read. This job establishes, per source: | |
| # * the repo+config resolves at all (a 404 here is far cheaper now than mid-build), | |
| # * the text column exists and under that name, | |
| # * mean characters and mean tokens per document under the SmolLM2 tokenizer, | |
| # * therefore how many documents each target token count costs, | |
| # * and whether the stream is English in practice. | |
| # | |
| # Quota-free by rule (§3.7): this is data work, so it runs on a Kaggle CPU instance. | |
| # Credential comes from the private store (D-006); nothing here prints a token. | |
| import json | |
| import os | |
| import time | |
| R = {} | |
| START = time.monotonic() | |
| BUDGET_S = 1200 | |
| ROWS_PER_SOURCE = 1500 | |
| # (repo, config, split, text column we expect, target tokens in the mix, why it is in the mix) | |
| SOURCES = [ | |
| ("HuggingFaceFW/fineweb-edu", "sample/10BT", "train", "text", 300e6, "scored general English"), | |
| ("HuggingFaceFW/finepdfs-edu", "eng_Latn", "train", "text", 150e6, "long-form educational PDFs"), | |
| ("HuggingFaceTB/cosmopedia", "stanford", "train", "text", 60e6, "textbook expository"), | |
| ("HuggingFaceTB/cosmopedia", "openstax", "train", "text", 30e6, "textbook science"), | |
| ("HuggingFaceTB/cosmopedia", "khanacademy", "train", "text", 10e6, "textbook worked steps"), | |
| ("HuggingFaceTB/cosmopedia", "auto_math_text", "train", "text", 30e6, "synthetic math text"), | |
| ("HuggingFaceTB/cosmopedia", "wikihow", "train", "text", 30e6, "procedural how-to"), | |
| ("HuggingFaceFW/finewiki", "data/enwiki", "train", "rewritten", 80e6, "rewritten wiki prose; NOTE column name is a guess"), | |
| ("omarkamali/wikipedia-monthly", "20250702.en", "train", "text", 60e6, "current encyclopedic"), | |
| ("HuggingFaceTB/finemath", "finemath4plus", "train", "text", 130e6, "math, decontaminated"), | |
| ("open-web-math/open-web-math", "default", "train", "text", 70e6, "forum/webbook math"), | |
| ("HuggingFaceFW/finephrase", "tutorial", "train", "completion", 60e6, "stepwise tutorial register; column is a guess"), | |
| ("HuggingFaceFW/finephrase", "faq", "train", "completion", 30e6, "FAQ register"), | |
| ("HuggingFaceFW/finephrase", "table", "train", "completion", 30e6, "tabular->prose"), | |
| ("HuggingFaceCode/stack-v3-train", "python", "train", "content", 60e6, "code; column and config names both guesses"), | |
| ("SimpleStories/SimpleStories", "default", "train", "text", 40e6, "long-range simple narrative"), | |
| ("common-pile/arxiv_abstracts", "default", "train", "raw_content", 20e6, "CC0 scientific abstracts"), | |
| ("common-pile/libretexts", "default", "train", "raw_content", 10e6, "OER textbooks"), | |
| ] | |
| def guard(name, fn): | |
| try: | |
| R[name] = fn() | |
| except Exception as e: | |
| R[name] = {"error": f"{type(e).__name__}: {e}"[:260]} | |
| def remaining(): | |
| return BUDGET_S - (time.monotonic() - START) | |
| # ---- the tokenizer everything downstream must agree with | |
| import ounce100m_credentials # noqa: E402 | |
| guard("credentials", lambda: ounce100m_credentials.install(verify=True)) | |
| from tokenizers import Tokenizer # noqa: E402 | |
| from huggingface_hub import hf_hub_download # noqa: E402 | |
| def load_tok(): | |
| path = hf_hub_download("HuggingFaceTB/SmolLM2-135M", "tokenizer.json") | |
| tk = Tokenizer.from_file(path) | |
| tk.no_truncation() | |
| vocab = tk.get_vocab_size() | |
| probe = ("The patient was given an intravenous dose because the oral route could not achieve " | |
| "sufficient bioavailability, and the nurse monitored the infusion rate.") | |
| ids = tk.encode(probe, add_special_tokens=False).ids | |
| return {"path": path, "vocab_size": vocab, "roundtrip_ok": tk.decode(ids).split()[0] == "The", | |
| "probe_tokens": len(ids), "probe_chars_per_token": round(len(probe) / len(ids), 2), | |
| "target_vocab_49152": vocab == 49152} | |
| guard("tokenizer", load_tok) | |
| _tok_path = os.environ.get("SMOL_TOK") or (R.get("tokenizer", {}) or {}).get("path") | |
| def tk_of(): | |
| t = _TOK[0] | |
| return t | |
| _TOK = [None] | |
| if _tok_path: | |
| _TOK[0] = Tokenizer.from_file(_tok_path) | |
| _TOK[0].no_truncation() | |
| def pick_text(row, expected): | |
| """Return (column_used, text). Cards lie; discover the real column and say which one we used.""" | |
| if expected in row and isinstance(row.get(expected), str): | |
| return expected, row[expected] | |
| for k, v in row.items(): | |
| if isinstance(v, str) and len(v) > 120: | |
| return k, v | |
| for k, v in row.items(): | |
| if isinstance(v, str) and v: | |
| return k, v | |
| return None, "" | |
| def inventory(repo, config, split, expect_col, target_tokens, role): | |
| import datasets | |
| ds = datasets.load_dataset(repo, config, split=split, streaming=True) | |
| n, chars, tok_n, cols, samples, lang_flags = 0, 0, 0, None, [], 0 | |
| col_used = {} | |
| t0 = time.monotonic() | |
| texts_for_tok = [] | |
| for row in ds: | |
| if n >= ROWS_PER_SOURCE or time.monotonic() - t0 > 180: | |
| break | |
| if cols is None: | |
| cols = sorted(row.keys()) | |
| c, txt = pick_text(row, expect_col) | |
| col_used[c] = col_used.get(c, 0) + 1 | |
| n += 1 | |
| chars += len(txt) | |
| if len(texts_for_tok) < 400: | |
| texts_for_tok.append(txt[:8000]) | |
| low = txt[:400].lower() | |
| if sum(w in low for w in (" the ", " and ", " of ", " to ", " that ")) < 2: | |
| lang_flags += 1 | |
| el = time.monotonic() - t0 | |
| toks = sum(len(_TOK[0].encode(t, add_special_tokens=False).ids) for t in texts_for_tok) | |
| tpd = toks / max(1, len(texts_for_tok)) | |
| cpd = chars / max(1, n) | |
| return { | |
| "config": config, "split": split, "role": role, "rows_sampled": n, "seconds": round(el, 1), | |
| "columns_seen": cols, "expected_text_col": expect_col, | |
| "col_used": sorted(col_used, key=lambda k: -col_used[k])[:3], | |
| "mean_chars_per_doc_all": round(cpd, 0), | |
| "mean_tokens_per_doc_sampled": round(tpd, 1), | |
| "chars_per_token": round(sum(len(t) for t in texts_for_tok) / max(1, toks), 2), | |
| "docs_needed_for_target": int(target_tokens / max(1.0, tpd)), | |
| "target_tokens": int(target_tokens), | |
| "possibly_non_english_rows": lang_flags, | |
| "est_GB_of_target": round(target_tokens / max(1.0, tpd) * (cpd + 900) / 1024**3, 2), | |
| "example_head": texts_for_tok[0][:120] if texts_for_tok else None, | |
| } | |
| inv = {} | |
| for repo, config, split, expect_col, target, role in SOURCES: | |
| key = f"{repo}::{config}" | |
| if remaining() < 60: | |
| inv[key] = {"skipped": "budget"} | |
| continue | |
| if _TOK[0] is None: | |
| inv[key] = {"skipped": "no tokenizer"} | |
| continue | |
| try: | |
| inv[key] = inventory(repo, config, split, expect_col, target, role) | |
| except Exception as e: | |
| inv[key] = {"error": f"{type(e).__name__}: {e}"[:240]} | |
| R["inventory"] = inv | |
| # ---- what we can conclude about the mix arithmetic | |
| ok = {k: v for k, v in inv.items() if "mean_tokens_per_doc_sampled" in v} | |
| R["mix_rollup"] = { | |
| "sources_ok": len(ok), "sources_failed_or_skipped": len(inv) - len(ok), | |
| "total_target_tokens": int(sum(v["target_tokens"] for v in ok.values())), | |
| "total_docs_needed": int(sum(v["docs_needed_for_target"] for v in ok.values())), | |
| "total_est_GB": round(sum(v["est_GB_of_target"] for v in ok.values()), 2), | |
| "failed": sorted(k for k, v in inv.items() if "mean_tokens_per_doc_sampled" not in v), | |
| } | |
| R["_seconds_used"] = round(time.monotonic() - START, 1) | |
| R["_budget_remaining"] = round(remaining(), 1) | |
| print("PROBE_JSON_BEGIN") | |
| print(json.dumps(R, indent=1, default=str)) | |
| print("PROBE_JSON_END") | |