# Phase 2 step 1: inventory the mix's sources with our own measurements, not their cards. # # Why before building. docs/02-mix-plan.md §4 records that two candidate sources disagree with # themselves (cosmopedia-v2 card 17.8 B vs ~31.7 B implied by stored columns; finephrase's # `token_count` describes the *source* document, not the rewrite). The mix's token arithmetic therefore # cannot be inherited -- it needs tokens/document measured with the tokenizer we actually froze, from the # rows we will actually read. This job establishes, per source: # * the repo+config resolves at all (a 404 here is far cheaper now than mid-build), # * the text column exists and under that name, # * mean characters and mean tokens per document under the SmolLM2 tokenizer, # * therefore how many documents each target token count costs, # * and whether the stream is English in practice. # # Quota-free by rule (§3.7): this is data work, so it runs on a Kaggle CPU instance. # Credential comes from the private store (D-006); nothing here prints a token. import json import os import time R = {} START = time.monotonic() BUDGET_S = 1200 ROWS_PER_SOURCE = 1500 # (repo, config, split, text column we expect, target tokens in the mix, why it is in the mix) SOURCES = [ ("HuggingFaceFW/fineweb-edu", "sample/10BT", "train", "text", 300e6, "scored general English"), ("HuggingFaceFW/finepdfs-edu", "eng_Latn", "train", "text", 150e6, "long-form educational PDFs"), ("HuggingFaceTB/cosmopedia", "stanford", "train", "text", 60e6, "textbook expository"), ("HuggingFaceTB/cosmopedia", "openstax", "train", "text", 30e6, "textbook science"), ("HuggingFaceTB/cosmopedia", "khanacademy", "train", "text", 10e6, "textbook worked steps"), ("HuggingFaceTB/cosmopedia", "auto_math_text", "train", "text", 30e6, "synthetic math text"), ("HuggingFaceTB/cosmopedia", "wikihow", "train", "text", 30e6, "procedural how-to"), ("HuggingFaceFW/finewiki", "data/enwiki", "train", "rewritten", 80e6, "rewritten wiki prose; NOTE column name is a guess"), ("omarkamali/wikipedia-monthly", "20250702.en", "train", "text", 60e6, "current encyclopedic"), ("HuggingFaceTB/finemath", "finemath4plus", "train", "text", 130e6, "math, decontaminated"), ("open-web-math/open-web-math", "default", "train", "text", 70e6, "forum/webbook math"), ("HuggingFaceFW/finephrase", "tutorial", "train", "completion", 60e6, "stepwise tutorial register; column is a guess"), ("HuggingFaceFW/finephrase", "faq", "train", "completion", 30e6, "FAQ register"), ("HuggingFaceFW/finephrase", "table", "train", "completion", 30e6, "tabular->prose"), ("HuggingFaceCode/stack-v3-train", "python", "train", "content", 60e6, "code; column and config names both guesses"), ("SimpleStories/SimpleStories", "default", "train", "text", 40e6, "long-range simple narrative"), ("common-pile/arxiv_abstracts", "default", "train", "raw_content", 20e6, "CC0 scientific abstracts"), ("common-pile/libretexts", "default", "train", "raw_content", 10e6, "OER textbooks"), ] def guard(name, fn): try: R[name] = fn() except Exception as e: R[name] = {"error": f"{type(e).__name__}: {e}"[:260]} def remaining(): return BUDGET_S - (time.monotonic() - START) # ---- the tokenizer everything downstream must agree with import ounce100m_credentials # noqa: E402 guard("credentials", lambda: ounce100m_credentials.install(verify=True)) from tokenizers import Tokenizer # noqa: E402 from huggingface_hub import hf_hub_download # noqa: E402 def load_tok(): path = hf_hub_download("HuggingFaceTB/SmolLM2-135M", "tokenizer.json") tk = Tokenizer.from_file(path) tk.no_truncation() vocab = tk.get_vocab_size() probe = ("The patient was given an intravenous dose because the oral route could not achieve " "sufficient bioavailability, and the nurse monitored the infusion rate.") ids = tk.encode(probe, add_special_tokens=False).ids return {"path": path, "vocab_size": vocab, "roundtrip_ok": tk.decode(ids).split()[0] == "The", "probe_tokens": len(ids), "probe_chars_per_token": round(len(probe) / len(ids), 2), "target_vocab_49152": vocab == 49152} guard("tokenizer", load_tok) _tok_path = os.environ.get("SMOL_TOK") or (R.get("tokenizer", {}) or {}).get("path") def tk_of(): t = _TOK[0] return t _TOK = [None] if _tok_path: _TOK[0] = Tokenizer.from_file(_tok_path) _TOK[0].no_truncation() def pick_text(row, expected): """Return (column_used, text). Cards lie; discover the real column and say which one we used.""" if expected in row and isinstance(row.get(expected), str): return expected, row[expected] for k, v in row.items(): if isinstance(v, str) and len(v) > 120: return k, v for k, v in row.items(): if isinstance(v, str) and v: return k, v return None, "" def inventory(repo, config, split, expect_col, target_tokens, role): import datasets ds = datasets.load_dataset(repo, config, split=split, streaming=True) n, chars, tok_n, cols, samples, lang_flags = 0, 0, 0, None, [], 0 col_used = {} t0 = time.monotonic() texts_for_tok = [] for row in ds: if n >= ROWS_PER_SOURCE or time.monotonic() - t0 > 180: break if cols is None: cols = sorted(row.keys()) c, txt = pick_text(row, expect_col) col_used[c] = col_used.get(c, 0) + 1 n += 1 chars += len(txt) if len(texts_for_tok) < 400: texts_for_tok.append(txt[:8000]) low = txt[:400].lower() if sum(w in low for w in (" the ", " and ", " of ", " to ", " that ")) < 2: lang_flags += 1 el = time.monotonic() - t0 toks = sum(len(_TOK[0].encode(t, add_special_tokens=False).ids) for t in texts_for_tok) tpd = toks / max(1, len(texts_for_tok)) cpd = chars / max(1, n) return { "config": config, "split": split, "role": role, "rows_sampled": n, "seconds": round(el, 1), "columns_seen": cols, "expected_text_col": expect_col, "col_used": sorted(col_used, key=lambda k: -col_used[k])[:3], "mean_chars_per_doc_all": round(cpd, 0), "mean_tokens_per_doc_sampled": round(tpd, 1), "chars_per_token": round(sum(len(t) for t in texts_for_tok) / max(1, toks), 2), "docs_needed_for_target": int(target_tokens / max(1.0, tpd)), "target_tokens": int(target_tokens), "possibly_non_english_rows": lang_flags, "est_GB_of_target": round(target_tokens / max(1.0, tpd) * (cpd + 900) / 1024**3, 2), "example_head": texts_for_tok[0][:120] if texts_for_tok else None, } inv = {} for repo, config, split, expect_col, target, role in SOURCES: key = f"{repo}::{config}" if remaining() < 60: inv[key] = {"skipped": "budget"} continue if _TOK[0] is None: inv[key] = {"skipped": "no tokenizer"} continue try: inv[key] = inventory(repo, config, split, expect_col, target, role) except Exception as e: inv[key] = {"error": f"{type(e).__name__}: {e}"[:240]} R["inventory"] = inv # ---- what we can conclude about the mix arithmetic ok = {k: v for k, v in inv.items() if "mean_tokens_per_doc_sampled" in v} R["mix_rollup"] = { "sources_ok": len(ok), "sources_failed_or_skipped": len(inv) - len(ok), "total_target_tokens": int(sum(v["target_tokens"] for v in ok.values())), "total_docs_needed": int(sum(v["docs_needed_for_target"] for v in ok.values())), "total_est_GB": round(sum(v["est_GB_of_target"] for v in ok.values()), 2), "failed": sorted(k for k, v in inv.items() if "mean_tokens_per_doc_sampled" not in v), } R["_seconds_used"] = round(time.monotonic() - START, 1) R["_budget_remaining"] = round(remaining(), 1) print("PROBE_JSON_BEGIN") print(json.dumps(R, indent=1, default=str)) print("PROBE_JSON_END")