#!/usr/bin/env python3 """Build the SN3 *true-objective* eval mix. SN3 duel score is mu_hat = sum_i w_i * mu_i over 10 fixed corpora with FIXED sampling weights (from upstream unarbos/teutonic npy_sources.py, reproduced exactly by live dashboard n_sequences/25000). Corpora are PUBLIC pretokenized uint32 .npy shards on Hippius, tokenizer silx-ai/Quasar-10B, seq_len 2048, so the real eval distribution can be fetched directly via HTTP range reads. """ from __future__ import annotations import hashlib import json import pathlib import sys import time import urllib.error import urllib.request import numpy as np OUT = pathlib.Path("/var/lib/mining/work/trainmix") SEQ_LEN = 2048 ITEM = 4 VOCAB_SIZE = 248320 SEQS_PER_CORPUS = 6000 CHUNKS = 8 SEED = 20260727 SOURCES = { "finewebedu": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/finewebedu/manifest.json", 0.27), "automathtext-v2-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/automathtext-v2-quasar-10b/manifest.json", 0.22), "openthoughts3-1.2m-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/openthoughts3-1.2m-quasar-10b/manifest.json", 0.10), "ultradata-math-l3-quasar-10b": ("https://us-east-1.hippius.com/teutonic-sn3/dataset/ultradata-math-l3-quasar-10b/manifest.json", 0.09), "pes2o-v3": ("https://s3.hippius.com/teutonic-sn3/dataset/pes2o-v3/manifest.json", 0.08), "openmathreasoning-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/openmathreasoning-quasar-10b/manifest.json", 0.06), "nemotron-cc-math-v1-4plus-mind-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/nemotron-cc-math-v1-4plus-mind-quasar-10b/manifest.json", 0.05), "cosmopedia-wikihow-stories-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/cosmopedia-wikihow-stories-quasar-10b/manifest.json", 0.05), "nemotron-specialized-v1.2-quasar-10b": ("https://eu-central-1.hippius.com/teutonic-sn3/dataset/nemotron-specialized-v1.2-quasar-10b/manifest.json", 0.04), "dendrite-synth-run": ("https://us-east-1.hippius.com/dendrite-teutonic/dataset/dendrite-synth-run/manifest.json", 0.04), } def log(msg): print("[%s] %s" % (time.strftime("%H:%M:%S"), msg), flush=True) def fetch(url, rng=None, tries=4): last = None for attempt in range(tries): req = urllib.request.Request(url) if rng is not None: req.add_header("Range", "bytes=%d-%d" % rng) try: with urllib.request.urlopen(req, timeout=180) as resp: return resp.read() except Exception as exc: last = exc time.sleep(2 * (attempt + 1)) raise RuntimeError("fetch failed %s rng=%s: %s" % (url, rng, last)) def bucket_root(manifest_url, corpus): tail = "dataset/%s/manifest.json" % corpus if not manifest_url.endswith(tail): return manifest_url[: manifest_url.rfind("dataset/")] return manifest_url[: -len(tail)] def npy_header_len(head): if head[:6] != b"\x93NUMPY": raise RuntimeError("not a .npy stream") if head[6] == 1: return 10 + int.from_bytes(head[8:10], "little") return 12 + int.from_bytes(head[8:12], "little") def build_corpus(corpus, manifest_url, weight, rs): log("--- %s (w=%s)" % (corpus, weight)) man = json.loads(fetch(manifest_url)) shards = [s for s in (man.get("shards") or man.get("files") or []) if str(s.get("key", "")).endswith(".npy")] if not shards: raise RuntimeError("%s: manifest has no .npy shards" % corpus) root = bucket_root(manifest_url, corpus) per_chunk = SEQS_PER_CORPUS // CHUNKS pick = min(CHUNKS, len(shards)) idxs = rs.choice(len(shards), size=pick, replace=False) plan = [(int(idxs[j % pick]), per_chunk) for j in range(CHUNKS)] blocks, prov = [], [] for shard_i, want in plan: entry = shards[shard_i] key = entry["key"] url = root + key head = fetch(url, (0, 255)) hlen = npy_header_len(head) size_bytes = int(entry.get("size_bytes") or 0) or (hlen + int(entry.get("n_tokens", 0)) * ITEM) total_seqs = (size_bytes - hlen) // (SEQ_LEN * ITEM) if total_seqs < want: want = total_seqs if want <= 0: continue start_seq = int(rs.integers(0, max(0, total_seqs - want) + 1)) b0 = hlen + start_seq * SEQ_LEN * ITEM b1 = b0 + want * SEQ_LEN * ITEM - 1 raw = fetch(url, (b0, b1)) if len(raw) != (b1 - b0 + 1): raise RuntimeError("%s: short read %d != %d on %s" % (corpus, len(raw), b1 - b0 + 1, key)) blocks.append(np.frombuffer(raw, dtype="= VOCAB_SIZE).any(axis=1) n_drop = int(n_raw - keep.sum()) tok = np.ascontiguousarray(tok[keep]) out = OUT / ("%s.npy" % corpus) np.save(out, tok) log(" -> %s shape=%s dropped_oob=%d" % (out.name, tok.shape, n_drop)) return { "corpus": corpus, "weight": weight, "manifest_url": manifest_url, "manifest_tokenizer": man.get("tokenizer"), "manifest_dtype": man.get("dtype"), "manifest_total_tokens": man.get("total_tokens"), "manifest_total_shards": man.get("total_shards") or len(shards), "n_seqs": int(tok.shape[0]), "n_seqs_dropped_oob": n_drop, "seq_len": SEQ_LEN, "file": out.name, "file_sha256": hashlib.sha256(out.read_bytes()).hexdigest(), "chunks": prov, } def main(): OUT.mkdir(parents=True, exist_ok=True) rs = np.random.default_rng(SEED) entries, failed = [], {} for corpus, (url, weight) in SOURCES.items(): try: entries.append(build_corpus(corpus, url, weight, rs)) except Exception as exc: log(" !! FAILED %s: %s" % (corpus, exc)) failed[corpus] = str(exc) index = { "name": "sn3-trainmix-v1", "purpose": "TRAINING shards for full 10-corpus SN3 objective; disjoint-by-seed from truemix-v1 eval set", "built_at": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), "seq_len": SEQ_LEN, "vocab_guard": VOCAB_SIZE, "seed": SEED, "weights_source": "upstream unarbos/teutonic npy_sources.py DEFAULT_SOURCE_WEIGHT_MAP", "weights_verified_against": "live dashboard eval-0330 n_sequences/25000 exact match", "protocol_delta": 0.0015, "tokenizer": "silx-ai/Quasar-10B", "corpora": entries, "failed": failed, "weight_total": round(sum(e["weight"] for e in entries), 6), } (OUT / "index.json").write_text(json.dumps(index, indent=2)) log("index.json: %d ok, %d failed, weight_total=%s" % (len(entries), len(failed), index["weight_total"])) return 0 if not failed else 1 if __name__ == "__main__": sys.exit(main())