Cion-lab commited on
Commit
fc350c9
·
verified ·
1 Parent(s): 0a2e67e

Phase 2: source inventory probe + d576/49152 param candidates

Browse files
config/param_count.py CHANGED
@@ -41,6 +41,24 @@ CANDIDATES = {
41
  "G_1024_h_8l_gqa2_2816": dict(PROBE_SHAPE, hidden_size=1024, num_hidden_layers=8,
42
  num_attention_heads=8, num_key_value_heads=2,
43
  intermediate_size=2816),
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
44
  }
45
 
46
 
 
41
  "G_1024_h_8l_gqa2_2816": dict(PROBE_SHAPE, hidden_size=1024, num_hidden_layers=8,
42
  num_attention_heads=8, num_key_value_heads=2,
43
  intermediate_size=2816),
44
+ # The family the design actually converges on (docs/01-plan.md section 2): deep-thin at d576 with
45
+ # the SmolLM2 tokenizer's 49,152 vocab, which is what makes the embedding tax affordable. Depth is
46
+ # the knob that moves the total, so size the whole band rather than guessing one value.
47
+ "H_576_20l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=20, num_attention_heads=9,
48
+ num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
49
+ tie_word_embeddings=True),
50
+ "I_576_22l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=22, num_attention_heads=9,
51
+ num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
52
+ tie_word_embeddings=True),
53
+ "J_576_24l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=24, num_attention_heads=9,
54
+ num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
55
+ tie_word_embeddings=True),
56
+ "K_576_26l_gqa3_smol": dict(hidden_size=576, num_hidden_layers=26, num_attention_heads=9,
57
+ num_key_value_heads=3, intermediate_size=1536, vocab_size=49152,
58
+ tie_word_embeddings=True),
59
+ "L_640_20l_gqa4_smol": dict(hidden_size=640, num_hidden_layers=20, num_attention_heads=10,
60
+ num_key_value_heads=4, intermediate_size=1728, vocab_size=49152,
61
+ tie_word_embeddings=True),
62
  }
63
 
64
 
probes/p2_source_inventory.py ADDED
@@ -0,0 +1,180 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Phase 2 step 1: inventory the mix's sources with our own measurements, not their cards.
2
+ #
3
+ # Why before building. docs/02-mix-plan.md §4 records that two candidate sources disagree with
4
+ # themselves (cosmopedia-v2 card 17.8 B vs ~31.7 B implied by stored columns; finephrase's
5
+ # `token_count` describes the *source* document, not the rewrite). The mix's token arithmetic therefore
6
+ # cannot be inherited -- it needs tokens/document measured with the tokenizer we actually froze, from the
7
+ # rows we will actually read. This job establishes, per source:
8
+ # * the repo+config resolves at all (a 404 here is far cheaper now than mid-build),
9
+ # * the text column exists and under that name,
10
+ # * mean characters and mean tokens per document under the SmolLM2 tokenizer,
11
+ # * therefore how many documents each target token count costs,
12
+ # * and whether the stream is English in practice.
13
+ #
14
+ # Quota-free by rule (§3.7): this is data work, so it runs on a Kaggle CPU instance.
15
+ # Credential comes from the private store (D-006); nothing here prints a token.
16
+
17
+ import json
18
+ import os
19
+ import time
20
+
21
+ R = {}
22
+ START = time.monotonic()
23
+ BUDGET_S = 1200
24
+ ROWS_PER_SOURCE = 1500
25
+
26
+ # (repo, config, split, text column we expect, target tokens in the mix, why it is in the mix)
27
+ SOURCES = [
28
+ ("HuggingFaceFW/fineweb-edu", "sample/10BT", "train", "text", 300e6, "scored general English"),
29
+ ("HuggingFaceFW/finepdfs-edu", "eng_Latn", "train", "text", 150e6, "long-form educational PDFs"),
30
+ ("HuggingFaceTB/cosmopedia", "stanford", "train", "text", 60e6, "textbook expository"),
31
+ ("HuggingFaceTB/cosmopedia", "openstax", "train", "text", 30e6, "textbook science"),
32
+ ("HuggingFaceTB/cosmopedia", "khanacademy", "train", "text", 10e6, "textbook worked steps"),
33
+ ("HuggingFaceTB/cosmopedia", "auto_math_text", "train", "text", 30e6, "synthetic math text"),
34
+ ("HuggingFaceTB/cosmopedia", "wikihow", "train", "text", 30e6, "procedural how-to"),
35
+ ("HuggingFaceFW/finewiki", "data/enwiki", "train", "rewritten", 80e6, "rewritten wiki prose; NOTE column name is a guess"),
36
+ ("omarkamali/wikipedia-monthly", "20250702.en", "train", "text", 60e6, "current encyclopedic"),
37
+ ("HuggingFaceTB/finemath", "finemath4plus", "train", "text", 130e6, "math, decontaminated"),
38
+ ("open-web-math/open-web-math", "default", "train", "text", 70e6, "forum/webbook math"),
39
+ ("HuggingFaceFW/finephrase", "tutorial", "train", "completion", 60e6, "stepwise tutorial register; column is a guess"),
40
+ ("HuggingFaceFW/finephrase", "faq", "train", "completion", 30e6, "FAQ register"),
41
+ ("HuggingFaceFW/finephrase", "table", "train", "completion", 30e6, "tabular->prose"),
42
+ ("HuggingFaceCode/stack-v3-train", "python", "train", "content", 60e6, "code; column and config names both guesses"),
43
+ ("SimpleStories/SimpleStories", "default", "train", "text", 40e6, "long-range simple narrative"),
44
+ ("common-pile/arxiv_abstracts", "default", "train", "raw_content", 20e6, "CC0 scientific abstracts"),
45
+ ("common-pile/libretexts", "default", "train", "raw_content", 10e6, "OER textbooks"),
46
+ ]
47
+
48
+
49
+ def guard(name, fn):
50
+ try:
51
+ R[name] = fn()
52
+ except Exception as e:
53
+ R[name] = {"error": f"{type(e).__name__}: {e}"[:260]}
54
+
55
+
56
+ def remaining():
57
+ return BUDGET_S - (time.monotonic() - START)
58
+
59
+
60
+ # ---- the tokenizer everything downstream must agree with
61
+ import ounce100m_credentials # noqa: E402
62
+
63
+ guard("credentials", lambda: ounce100m_credentials.install(verify=True))
64
+
65
+ from tokenizers import Tokenizer # noqa: E402
66
+ from huggingface_hub import hf_hub_download # noqa: E402
67
+
68
+
69
+ def load_tok():
70
+ path = hf_hub_download("HuggingFaceTB/SmolLM2-135M", "tokenizer.json")
71
+ tk = Tokenizer.from_file(path)
72
+ tk.no_truncation()
73
+ vocab = tk.get_vocab_size()
74
+ probe = ("The patient was given an intravenous dose because the oral route could not achieve "
75
+ "sufficient bioavailability, and the nurse monitored the infusion rate.")
76
+ ids = tk.encode(probe, add_special_tokens=False).ids
77
+ return {"path": path, "vocab_size": vocab, "roundtrip_ok": tk.decode(ids).split()[0] == "The",
78
+ "probe_tokens": len(ids), "probe_chars_per_token": round(len(probe) / len(ids), 2),
79
+ "target_vocab_49152": vocab == 49152}
80
+
81
+
82
+ guard("tokenizer", load_tok)
83
+ _tok_path = os.environ.get("SMOL_TOK") or (R.get("tokenizer", {}) or {}).get("path")
84
+
85
+
86
+ def tk_of():
87
+ t = _TOK[0]
88
+ return t
89
+
90
+
91
+ _TOK = [None]
92
+ if _tok_path:
93
+ _TOK[0] = Tokenizer.from_file(_tok_path)
94
+ _TOK[0].no_truncation()
95
+
96
+
97
+ def pick_text(row, expected):
98
+ """Return (column_used, text). Cards lie; discover the real column and say which one we used."""
99
+ if expected in row and isinstance(row.get(expected), str):
100
+ return expected, row[expected]
101
+ for k, v in row.items():
102
+ if isinstance(v, str) and len(v) > 120:
103
+ return k, v
104
+ for k, v in row.items():
105
+ if isinstance(v, str) and v:
106
+ return k, v
107
+ return None, ""
108
+
109
+
110
+ def inventory(repo, config, split, expect_col, target_tokens, role):
111
+ import datasets
112
+
113
+ ds = datasets.load_dataset(repo, config, split=split, streaming=True)
114
+ n, chars, tok_n, cols, samples, lang_flags = 0, 0, 0, None, [], 0
115
+ col_used = {}
116
+ t0 = time.monotonic()
117
+ texts_for_tok = []
118
+ for row in ds:
119
+ if n >= ROWS_PER_SOURCE or time.monotonic() - t0 > 180:
120
+ break
121
+ if cols is None:
122
+ cols = sorted(row.keys())
123
+ c, txt = pick_text(row, expect_col)
124
+ col_used[c] = col_used.get(c, 0) + 1
125
+ n += 1
126
+ chars += len(txt)
127
+ if len(texts_for_tok) < 400:
128
+ texts_for_tok.append(txt[:8000])
129
+ low = txt[:400].lower()
130
+ if sum(w in low for w in (" the ", " and ", " of ", " to ", " that ")) < 2:
131
+ lang_flags += 1
132
+ el = time.monotonic() - t0
133
+ toks = sum(len(_TOK[0].encode(t, add_special_tokens=False).ids) for t in texts_for_tok)
134
+ tpd = toks / max(1, len(texts_for_tok))
135
+ cpd = chars / max(1, n)
136
+ return {
137
+ "config": config, "split": split, "role": role, "rows_sampled": n, "seconds": round(el, 1),
138
+ "columns_seen": cols, "expected_text_col": expect_col,
139
+ "col_used": sorted(col_used, key=lambda k: -col_used[k])[:3],
140
+ "mean_chars_per_doc_all": round(cpd, 0),
141
+ "mean_tokens_per_doc_sampled": round(tpd, 1),
142
+ "chars_per_token": round(sum(len(t) for t in texts_for_tok) / max(1, toks), 2),
143
+ "docs_needed_for_target": int(target_tokens / max(1.0, tpd)),
144
+ "target_tokens": int(target_tokens),
145
+ "possibly_non_english_rows": lang_flags,
146
+ "est_GB_of_target": round(target_tokens / max(1.0, tpd) * (cpd + 900) / 1024**3, 2),
147
+ "example_head": texts_for_tok[0][:120] if texts_for_tok else None,
148
+ }
149
+
150
+
151
+ inv = {}
152
+ for repo, config, split, expect_col, target, role in SOURCES:
153
+ key = f"{repo}::{config}"
154
+ if remaining() < 60:
155
+ inv[key] = {"skipped": "budget"}
156
+ continue
157
+ if _TOK[0] is None:
158
+ inv[key] = {"skipped": "no tokenizer"}
159
+ continue
160
+ try:
161
+ inv[key] = inventory(repo, config, split, expect_col, target, role)
162
+ except Exception as e:
163
+ inv[key] = {"error": f"{type(e).__name__}: {e}"[:240]}
164
+ R["inventory"] = inv
165
+
166
+ # ---- what we can conclude about the mix arithmetic
167
+ ok = {k: v for k, v in inv.items() if "mean_tokens_per_doc_sampled" in v}
168
+ R["mix_rollup"] = {
169
+ "sources_ok": len(ok), "sources_failed_or_skipped": len(inv) - len(ok),
170
+ "total_target_tokens": int(sum(v["target_tokens"] for v in ok.values())),
171
+ "total_docs_needed": int(sum(v["docs_needed_for_target"] for v in ok.values())),
172
+ "total_est_GB": round(sum(v["est_GB_of_target"] for v in ok.values()), 2),
173
+ "failed": sorted(k for k, v in inv.items() if "mean_tokens_per_doc_sampled" not in v),
174
+ }
175
+
176
+ R["_seconds_used"] = round(time.monotonic() - START, 1)
177
+ R["_budget_remaining"] = round(remaining(), 1)
178
+ print("PROBE_JSON_BEGIN")
179
+ print(json.dumps(R, indent=1, default=str))
180
+ print("PROBE_JSON_END")