ounce100m-code / probes /p2_source_inventory.py
Cion-lab's picture
Phase 2: source inventory probe + d576/49152 param candidates
fc350c9 verified
Raw History Blame Contribute Delete
7.97 kB
# Phase 2 step 1: inventory the mix's sources with our own measurements, not their cards.
#
# Why before building. docs/02-mix-plan.md §4 records that two candidate sources disagree with
# themselves (cosmopedia-v2 card 17.8 B vs ~31.7 B implied by stored columns; finephrase's
# `token_count` describes the *source* document, not the rewrite). The mix's token arithmetic therefore
# cannot be inherited -- it needs tokens/document measured with the tokenizer we actually froze, from the
# rows we will actually read. This job establishes, per source:
# * the repo+config resolves at all (a 404 here is far cheaper now than mid-build),
# * the text column exists and under that name,
# * mean characters and mean tokens per document under the SmolLM2 tokenizer,
# * therefore how many documents each target token count costs,
# * and whether the stream is English in practice.
#
# Quota-free by rule (§3.7): this is data work, so it runs on a Kaggle CPU instance.
# Credential comes from the private store (D-006); nothing here prints a token.
import json
import os
import time
R = {}
START = time.monotonic()
BUDGET_S = 1200
ROWS_PER_SOURCE = 1500
# (repo, config, split, text column we expect, target tokens in the mix, why it is in the mix)
SOURCES = [
("HuggingFaceFW/fineweb-edu", "sample/10BT", "train", "text", 300e6, "scored general English"),
("HuggingFaceFW/finepdfs-edu", "eng_Latn", "train", "text", 150e6, "long-form educational PDFs"),
("HuggingFaceTB/cosmopedia", "stanford", "train", "text", 60e6, "textbook expository"),
("HuggingFaceTB/cosmopedia", "openstax", "train", "text", 30e6, "textbook science"),
("HuggingFaceTB/cosmopedia", "khanacademy", "train", "text", 10e6, "textbook worked steps"),
("HuggingFaceTB/cosmopedia", "auto_math_text", "train", "text", 30e6, "synthetic math text"),
("HuggingFaceTB/cosmopedia", "wikihow", "train", "text", 30e6, "procedural how-to"),
("HuggingFaceFW/finewiki", "data/enwiki", "train", "rewritten", 80e6, "rewritten wiki prose; NOTE column name is a guess"),
("omarkamali/wikipedia-monthly", "20250702.en", "train", "text", 60e6, "current encyclopedic"),
("HuggingFaceTB/finemath", "finemath4plus", "train", "text", 130e6, "math, decontaminated"),
("open-web-math/open-web-math", "default", "train", "text", 70e6, "forum/webbook math"),
("HuggingFaceFW/finephrase", "tutorial", "train", "completion", 60e6, "stepwise tutorial register; column is a guess"),
("HuggingFaceFW/finephrase", "faq", "train", "completion", 30e6, "FAQ register"),
("HuggingFaceFW/finephrase", "table", "train", "completion", 30e6, "tabular->prose"),
("HuggingFaceCode/stack-v3-train", "python", "train", "content", 60e6, "code; column and config names both guesses"),
("SimpleStories/SimpleStories", "default", "train", "text", 40e6, "long-range simple narrative"),
("common-pile/arxiv_abstracts", "default", "train", "raw_content", 20e6, "CC0 scientific abstracts"),
("common-pile/libretexts", "default", "train", "raw_content", 10e6, "OER textbooks"),
]
def guard(name, fn):
try:
R[name] = fn()
except Exception as e:
R[name] = {"error": f"{type(e).__name__}: {e}"[:260]}
def remaining():
return BUDGET_S - (time.monotonic() - START)
# ---- the tokenizer everything downstream must agree with
import ounce100m_credentials # noqa: E402
guard("credentials", lambda: ounce100m_credentials.install(verify=True))
from tokenizers import Tokenizer # noqa: E402
from huggingface_hub import hf_hub_download # noqa: E402
def load_tok():
path = hf_hub_download("HuggingFaceTB/SmolLM2-135M", "tokenizer.json")
tk = Tokenizer.from_file(path)
tk.no_truncation()
vocab = tk.get_vocab_size()
probe = ("The patient was given an intravenous dose because the oral route could not achieve "
"sufficient bioavailability, and the nurse monitored the infusion rate.")
ids = tk.encode(probe, add_special_tokens=False).ids
return {"path": path, "vocab_size": vocab, "roundtrip_ok": tk.decode(ids).split()[0] == "The",
"probe_tokens": len(ids), "probe_chars_per_token": round(len(probe) / len(ids), 2),
"target_vocab_49152": vocab == 49152}
guard("tokenizer", load_tok)
_tok_path = os.environ.get("SMOL_TOK") or (R.get("tokenizer", {}) or {}).get("path")
def tk_of():
t = _TOK[0]
return t
_TOK = [None]
if _tok_path:
_TOK[0] = Tokenizer.from_file(_tok_path)
_TOK[0].no_truncation()
def pick_text(row, expected):
"""Return (column_used, text). Cards lie; discover the real column and say which one we used."""
if expected in row and isinstance(row.get(expected), str):
return expected, row[expected]
for k, v in row.items():
if isinstance(v, str) and len(v) > 120:
return k, v
for k, v in row.items():
if isinstance(v, str) and v:
return k, v
return None, ""
def inventory(repo, config, split, expect_col, target_tokens, role):
import datasets
ds = datasets.load_dataset(repo, config, split=split, streaming=True)
n, chars, tok_n, cols, samples, lang_flags = 0, 0, 0, None, [], 0
col_used = {}
t0 = time.monotonic()
texts_for_tok = []
for row in ds:
if n >= ROWS_PER_SOURCE or time.monotonic() - t0 > 180:
break
if cols is None:
cols = sorted(row.keys())
c, txt = pick_text(row, expect_col)
col_used[c] = col_used.get(c, 0) + 1
n += 1
chars += len(txt)
if len(texts_for_tok) < 400:
texts_for_tok.append(txt[:8000])
low = txt[:400].lower()
if sum(w in low for w in (" the ", " and ", " of ", " to ", " that ")) < 2:
lang_flags += 1
el = time.monotonic() - t0
toks = sum(len(_TOK[0].encode(t, add_special_tokens=False).ids) for t in texts_for_tok)
tpd = toks / max(1, len(texts_for_tok))
cpd = chars / max(1, n)
return {
"config": config, "split": split, "role": role, "rows_sampled": n, "seconds": round(el, 1),
"columns_seen": cols, "expected_text_col": expect_col,
"col_used": sorted(col_used, key=lambda k: -col_used[k])[:3],
"mean_chars_per_doc_all": round(cpd, 0),
"mean_tokens_per_doc_sampled": round(tpd, 1),
"chars_per_token": round(sum(len(t) for t in texts_for_tok) / max(1, toks), 2),
"docs_needed_for_target": int(target_tokens / max(1.0, tpd)),
"target_tokens": int(target_tokens),
"possibly_non_english_rows": lang_flags,
"est_GB_of_target": round(target_tokens / max(1.0, tpd) * (cpd + 900) / 1024**3, 2),
"example_head": texts_for_tok[0][:120] if texts_for_tok else None,
}
inv = {}
for repo, config, split, expect_col, target, role in SOURCES:
key = f"{repo}::{config}"
if remaining() < 60:
inv[key] = {"skipped": "budget"}
continue
if _TOK[0] is None:
inv[key] = {"skipped": "no tokenizer"}
continue
try:
inv[key] = inventory(repo, config, split, expect_col, target, role)
except Exception as e:
inv[key] = {"error": f"{type(e).__name__}: {e}"[:240]}
R["inventory"] = inv
# ---- what we can conclude about the mix arithmetic
ok = {k: v for k, v in inv.items() if "mean_tokens_per_doc_sampled" in v}
R["mix_rollup"] = {
"sources_ok": len(ok), "sources_failed_or_skipped": len(inv) - len(ok),
"total_target_tokens": int(sum(v["target_tokens"] for v in ok.values())),
"total_docs_needed": int(sum(v["docs_needed_for_target"] for v in ok.values())),
"total_est_GB": round(sum(v["est_GB_of_target"] for v in ok.values()), 2),
"failed": sorted(k for k, v in inv.items() if "mean_tokens_per_doc_sampled" not in v),
}
R["_seconds_used"] = round(time.monotonic() - START, 1)
R["_budget_remaining"] = round(remaining(), 1)
print("PROBE_JSON_BEGIN")
print(json.dumps(R, indent=1, default=str))
print("PROBE_JSON_END")