Phase 2: source inventory probe + d576/49152 param candidates
Browse files- config/param_count.py +18 -0
- probes/p2_source_inventory.py +180 -0
config/param_count.py
CHANGED
|
@@ -41,6 +41,24 @@ CANDIDATES = {
|
|
| 41 |
"G_1024_h_8l_gqa2_2816": dict(PROBE_SHAPE, hidden_size=1024, num_hidden_layers=8,
|
| 42 |
num_attention_heads=8, num_key_value_heads=2,
|
| 43 |
intermediate_size=2816),
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 44 |
}
|
| 45 |
|
| 46 |
|
|
|
|
| 41 |
"G_1024_h_8l_gqa2_2816": dict(PROBE_SHAPE, hidden_size=1024, num_hidden_layers=8,
|
| 42 |
num_attention_heads=8, num_key_value_heads=2,
|
| 43 |
intermediate_size=2816),
|
| 44 |
+
# The family the design actually converges on (docs/01-plan.md section 2): deep-thin at d576 with
|
| 45 |
+
# the SmolLM2 tokenizer's 49,152 vocab, which is what makes the embedding tax affordable. Depth is
|
| 46 |
+
# the knob that moves the total, so size the whole band rather than guessing one value.
|
| 47 |
+
"H_576_20l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=20, num_attention_heads=9,
|
| 48 |
+
num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
|
| 49 |
+
tie_word_embeddings=True),
|
| 50 |
+
"I_576_22l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=22, num_attention_heads=9,
|
| 51 |
+
num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
|
| 52 |
+
tie_word_embeddings=True),
|
| 53 |
+
"J_576_24l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=24, num_attention_heads=9,
|
| 54 |
+
num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
|
| 55 |
+
tie_word_embeddings=True),
|
| 56 |
+
"K_576_26l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=26, num_attention_heads=9,
|
| 57 |
+
num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
|
| 58 |
+
tie_word_embeddings=True),
|
| 59 |
+
"L_640_20l_gqa4_smol": dict(hidden_size=640, num_hidden_layers=20, num_attention_heads=10,
|
| 60 |
+
num_key_value_heads=4, intermediate_size=1728, vocab_size=49152,
|
| 61 |
+
tie_word_embeddings=True),
|
| 62 |
}
|
| 63 |
|
| 64 |
|
probes/p2_source_inventory.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Phase 2 step 1: inventory the mix's sources with our own measurements, not their cards.
|
| 2 |
+
#
|
| 3 |
+
# Why before building. docs/02-mix-plan.md §4 records that two candidate sources disagree with
|
| 4 |
+
# themselves (cosmopedia-v2 card 17.8 B vs ~31.7 B implied by stored columns; finephrase's
|
| 5 |
+
# `token_count` describes the *source* document, not the rewrite). The mix's token arithmetic therefore
|
| 6 |
+
# cannot be inherited -- it needs tokens/document measured with the tokenizer we actually froze, from the
|
| 7 |
+
# rows we will actually read. This job establishes, per source:
|
| 8 |
+
# * the repo+config resolves at all (a 404 here is far cheaper now than mid-build),
|
| 9 |
+
# * the text column exists and under that name,
|
| 10 |
+
# * mean characters and mean tokens per document under the SmolLM2 tokenizer,
|
| 11 |
+
# * therefore how many documents each target token count costs,
|
| 12 |
+
# * and whether the stream is English in practice.
|
| 13 |
+
#
|
| 14 |
+
# Quota-free by rule (§3.7): this is data work, so it runs on a Kaggle CPU instance.
|
| 15 |
+
# Credential comes from the private store (D-006); nothing here prints a token.
|
| 16 |
+
|
| 17 |
+
import json
|
| 18 |
+
import os
|
| 19 |
+
import time
|
| 20 |
+
|
| 21 |
+
R = {}
|
| 22 |
+
START = time.monotonic()
|
| 23 |
+
BUDGET_S = 1200
|
| 24 |
+
ROWS_PER_SOURCE = 1500
|
| 25 |
+
|
| 26 |
+
# (repo, config, split, text column we expect, target tokens in the mix, why it is in the mix)
|
| 27 |
+
SOURCES = [
|
| 28 |
+
("HuggingFaceFW/fineweb-edu", "sample/10BT", "train", "text", 300e6, "scored general English"),
|
| 29 |
+
("HuggingFaceFW/finepdfs-edu", "eng_Latn", "train", "text", 150e6, "long-form educational PDFs"),
|
| 30 |
+
("HuggingFaceTB/cosmopedia", "stanford", "train", "text", 60e6, "textbook expository"),
|
| 31 |
+
("HuggingFaceTB/cosmopedia", "openstax", "train", "text", 30e6, "textbook science"),
|
| 32 |
+
("HuggingFaceTB/cosmopedia", "khanacademy", "train", "text", 10e6, "textbook worked steps"),
|
| 33 |
+
("HuggingFaceTB/cosmopedia", "auto_math_text", "train", "text", 30e6, "synthetic math text"),
|
| 34 |
+
("HuggingFaceTB/cosmopedia", "wikihow", "train", "text", 30e6, "procedural how-to"),
|
| 35 |
+
("HuggingFaceFW/finewiki", "data/enwiki", "train", "rewritten", 80e6, "rewritten wiki prose; NOTE column name is a guess"),
|
| 36 |
+
("omarkamali/wikipedia-monthly", "20250702.en", "train", "text", 60e6, "current encyclopedic"),
|
| 37 |
+
("HuggingFaceTB/finemath", "finemath4plus", "train", "text", 130e6, "math, decontaminated"),
|
| 38 |
+
("open-web-math/open-web-math", "default", "train", "text", 70e6, "forum/webbook math"),
|
| 39 |
+
("HuggingFaceFW/finephrase", "tutorial", "train", "completion", 60e6, "stepwise tutorial register; column is a guess"),
|
| 40 |
+
("HuggingFaceFW/finephrase", "faq", "train", "completion", 30e6, "FAQ register"),
|
| 41 |
+
("HuggingFaceFW/finephrase", "table", "train", "completion", 30e6, "tabular->prose"),
|
| 42 |
+
("HuggingFaceCode/stack-v3-train", "python", "train", "content", 60e6, "code; column and config names both guesses"),
|
| 43 |
+
("SimpleStories/SimpleStories", "default", "train", "text", 40e6, "long-range simple narrative"),
|
| 44 |
+
("common-pile/arxiv_abstracts", "default", "train", "raw_content", 20e6, "CC0 scientific abstracts"),
|
| 45 |
+
("common-pile/libretexts", "default", "train", "raw_content", 10e6, "OER textbooks"),
|
| 46 |
+
]
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def guard(name, fn):
|
| 50 |
+
try:
|
| 51 |
+
R[name] = fn()
|
| 52 |
+
except Exception as e:
|
| 53 |
+
R[name] = {"error": f"{type(e).__name__}: {e}"[:260]}
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def remaining():
|
| 57 |
+
return BUDGET_S - (time.monotonic() - START)
|
| 58 |
+
|
| 59 |
+
|
| 60 |
+
# ---- the tokenizer everything downstream must agree with
|
| 61 |
+
import ounce100m_credentials # noqa: E402
|
| 62 |
+
|
| 63 |
+
guard("credentials", lambda: ounce100m_credentials.install(verify=True))
|
| 64 |
+
|
| 65 |
+
from tokenizers import Tokenizer # noqa: E402
|
| 66 |
+
from huggingface_hub import hf_hub_download # noqa: E402
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def load_tok():
|
| 70 |
+
path = hf_hub_download("HuggingFaceTB/SmolLM2-135M", "tokenizer.json")
|
| 71 |
+
tk = Tokenizer.from_file(path)
|
| 72 |
+
tk.no_truncation()
|
| 73 |
+
vocab = tk.get_vocab_size()
|
| 74 |
+
probe = ("The patient was given an intravenous dose because the oral route could not achieve "
|
| 75 |
+
"sufficient bioavailability, and the nurse monitored the infusion rate.")
|
| 76 |
+
ids = tk.encode(probe, add_special_tokens=False).ids
|
| 77 |
+
return {"path": path, "vocab_size": vocab, "roundtrip_ok": tk.decode(ids).split()[0] == "The",
|
| 78 |
+
"probe_tokens": len(ids), "probe_chars_per_token": round(len(probe) / len(ids), 2),
|
| 79 |
+
"target_vocab_49152": vocab == 49152}
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
guard("tokenizer", load_tok)
|
| 83 |
+
_tok_path = os.environ.get("SMOL_TOK") or (R.get("tokenizer", {}) or {}).get("path")
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
def tk_of():
|
| 87 |
+
t = _TOK[0]
|
| 88 |
+
return t
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
_TOK = [None]
|
| 92 |
+
if _tok_path:
|
| 93 |
+
_TOK[0] = Tokenizer.from_file(_tok_path)
|
| 94 |
+
_TOK[0].no_truncation()
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def pick_text(row, expected):
|
| 98 |
+
"""Return (column_used, text). Cards lie; discover the real column and say which one we used."""
|
| 99 |
+
if expected in row and isinstance(row.get(expected), str):
|
| 100 |
+
return expected, row[expected]
|
| 101 |
+
for k, v in row.items():
|
| 102 |
+
if isinstance(v, str) and len(v) > 120:
|
| 103 |
+
return k, v
|
| 104 |
+
for k, v in row.items():
|
| 105 |
+
if isinstance(v, str) and v:
|
| 106 |
+
return k, v
|
| 107 |
+
return None, ""
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
def inventory(repo, config, split, expect_col, target_tokens, role):
|
| 111 |
+
import datasets
|
| 112 |
+
|
| 113 |
+
ds = datasets.load_dataset(repo, config, split=split, streaming=True)
|
| 114 |
+
n, chars, tok_n, cols, samples, lang_flags = 0, 0, 0, None, [], 0
|
| 115 |
+
col_used = {}
|
| 116 |
+
t0 = time.monotonic()
|
| 117 |
+
texts_for_tok = []
|
| 118 |
+
for row in ds:
|
| 119 |
+
if n >= ROWS_PER_SOURCE or time.monotonic() - t0 > 180:
|
| 120 |
+
break
|
| 121 |
+
if cols is None:
|
| 122 |
+
cols = sorted(row.keys())
|
| 123 |
+
c, txt = pick_text(row, expect_col)
|
| 124 |
+
col_used[c] = col_used.get(c, 0) + 1
|
| 125 |
+
n += 1
|
| 126 |
+
chars += len(txt)
|
| 127 |
+
if len(texts_for_tok) < 400:
|
| 128 |
+
texts_for_tok.append(txt[:8000])
|
| 129 |
+
low = txt[:400].lower()
|
| 130 |
+
if sum(w in low for w in (" the ", " and ", " of ", " to ", " that ")) < 2:
|
| 131 |
+
lang_flags += 1
|
| 132 |
+
el = time.monotonic() - t0
|
| 133 |
+
toks = sum(len(_TOK[0].encode(t, add_special_tokens=False).ids) for t in texts_for_tok)
|
| 134 |
+
tpd = toks / max(1, len(texts_for_tok))
|
| 135 |
+
cpd = chars / max(1, n)
|
| 136 |
+
return {
|
| 137 |
+
"config": config, "split": split, "role": role, "rows_sampled": n, "seconds": round(el, 1),
|
| 138 |
+
"columns_seen": cols, "expected_text_col": expect_col,
|
| 139 |
+
"col_used": sorted(col_used, key=lambda k: -col_used[k])[:3],
|
| 140 |
+
"mean_chars_per_doc_all": round(cpd, 0),
|
| 141 |
+
"mean_tokens_per_doc_sampled": round(tpd, 1),
|
| 142 |
+
"chars_per_token": round(sum(len(t) for t in texts_for_tok) / max(1, toks), 2),
|
| 143 |
+
"docs_needed_for_target": int(target_tokens / max(1.0, tpd)),
|
| 144 |
+
"target_tokens": int(target_tokens),
|
| 145 |
+
"possibly_non_english_rows": lang_flags,
|
| 146 |
+
"est_GB_of_target": round(target_tokens / max(1.0, tpd) * (cpd + 900) / 1024**3, 2),
|
| 147 |
+
"example_head": texts_for_tok[0][:120] if texts_for_tok else None,
|
| 148 |
+
}
|
| 149 |
+
|
| 150 |
+
|
| 151 |
+
inv = {}
|
| 152 |
+
for repo, config, split, expect_col, target, role in SOURCES:
|
| 153 |
+
key = f"{repo}::{config}"
|
| 154 |
+
if remaining() < 60:
|
| 155 |
+
inv[key] = {"skipped": "budget"}
|
| 156 |
+
continue
|
| 157 |
+
if _TOK[0] is None:
|
| 158 |
+
inv[key] = {"skipped": "no tokenizer"}
|
| 159 |
+
continue
|
| 160 |
+
try:
|
| 161 |
+
inv[key] = inventory(repo, config, split, expect_col, target, role)
|
| 162 |
+
except Exception as e:
|
| 163 |
+
inv[key] = {"error": f"{type(e).__name__}: {e}"[:240]}
|
| 164 |
+
R["inventory"] = inv
|
| 165 |
+
|
| 166 |
+
# ---- what we can conclude about the mix arithmetic
|
| 167 |
+
ok = {k: v for k, v in inv.items() if "mean_tokens_per_doc_sampled" in v}
|
| 168 |
+
R["mix_rollup"] = {
|
| 169 |
+
"sources_ok": len(ok), "sources_failed_or_skipped": len(inv) - len(ok),
|
| 170 |
+
"total_target_tokens": int(sum(v["target_tokens"] for v in ok.values())),
|
| 171 |
+
"total_docs_needed": int(sum(v["docs_needed_for_target"] for v in ok.values())),
|
| 172 |
+
"total_est_GB": round(sum(v["est_GB_of_target"] for v in ok.values()), 2),
|
| 173 |
+
"failed": sorted(k for k, v in inv.items() if "mean_tokens_per_doc_sampled" not in v),
|
| 174 |
+
}
|
| 175 |
+
|
| 176 |
+
R["_seconds_used"] = round(time.monotonic() - START, 1)
|
| 177 |
+
R["_budget_remaining"] = round(remaining(), 1)
|
| 178 |
+
print("PROBE_JSON_BEGIN")
|
| 179 |
+
print(json.dumps(R, indent=1, default=str))
|
| 180 |
+
print("PROBE_JSON_END")
|