"""CORTEX tokenizer — loading and validating the CORTEX vocabulary. The CORTEX tokenizer is reused unchanged from the distributed repository (``Frankenstein-Labs/Cortex-ai``, MIT). This module never rebuilds or mutates it; it loads the shipped ``tokenizer.json`` and exposes helpers used by the pipeline. """ from __future__ import annotations import json from pathlib import Path from tokenizers import Tokenizer __all__ = ["load_tokenizer", "describe_tokenizer", "assert_vocab_matches_config"] DEFAULT_TOKENIZER = Path(__file__).resolve().parent.parent / "tokenizer.json" def load_tokenizer(path: str | Path | None = None) -> Tokenizer: """Load the CORTEX BPE tokenizer. Defaults to the ``tokenizer.json`` shipped at the repository root, so the pipeline and the distributed checkpoint use one and the same vocabulary file. """ path = Path(path) if path else DEFAULT_TOKENIZER if not path.exists(): raise FileNotFoundError( f"tokenizer not found at {path}. See tokenizer/README.md for how it is obtained." ) return Tokenizer.from_file(str(path)) def describe_tokenizer(path: str | Path | None = None) -> dict: """Return measured facts about the tokenizer file.""" path = Path(path) if path else DEFAULT_TOKENIZER raw = json.loads(path.read_text(encoding="utf-8")) model = raw.get("model", {}) added = raw.get("added_tokens", []) vocab_size = len(model.get("vocab", {})) added_ids = {t["id"] for t in added} base_ids = set(range(vocab_size)) return { "tokenizer_class": raw.get("tokenizer_class", "PreTrainedTokenizerFast"), "model_type": model.get("type"), "base_vocab_size": vocab_size, "merges": len(model.get("merges", [])), "added_tokens": len(added), "unique_ids": len(base_ids | added_ids), "overlapping_ids": len(base_ids & added_ids), "max_added_id": max(added_ids) if added_ids else None, } def assert_vocab_matches_config(tokenizer: Tokenizer, vocab_size: int) -> None: """Fail loudly if the tokenizer and the model disagree on vocabulary size. The CORTEX tokenizer exposes 128000 BPE entries plus 1283 added tokens, but ids 0, 1 and 2 appear in both sets. The number of *unique* ids is what must equal ``vocab_size``, so raw counters must never be compared directly. """ unique = tokenizer.get_vocab_size(with_added_tokens=True) if unique != vocab_size: raise ValueError( f"tokenizer exposes {unique} unique ids but the model config declares " f"vocab_size={vocab_size}" )