Cortex-ai / tokenizer /cortex_tokenizer.py
openhands
openhands
feat(cortex): add CORTEX training pipeline and model audit
6fbe100
Raw History Blame Contribute Delete
2.63 kB
"""CORTEX tokenizer — loading and validating the CORTEX vocabulary.
The CORTEX tokenizer is reused unchanged from the distributed repository
(``Frankenstein-Labs/Cortex-ai``, MIT). This module never rebuilds or mutates it; it
loads the shipped ``tokenizer.json`` and exposes helpers used by the pipeline.
"""
from __future__ import annotations
import json
from pathlib import Path
from tokenizers import Tokenizer
__all__ = ["load_tokenizer", "describe_tokenizer", "assert_vocab_matches_config"]
DEFAULT_TOKENIZER = Path(__file__).resolve().parent.parent / "tokenizer.json"
def load_tokenizer(path: str | Path | None = None) -> Tokenizer:
"""Load the CORTEX BPE tokenizer.
Defaults to the ``tokenizer.json`` shipped at the repository root, so the pipeline
and the distributed checkpoint use one and the same vocabulary file.
"""
path = Path(path) if path else DEFAULT_TOKENIZER
if not path.exists():
raise FileNotFoundError(
f"tokenizer not found at {path}. See tokenizer/README.md for how it is obtained."
)
return Tokenizer.from_file(str(path))
def describe_tokenizer(path: str | Path | None = None) -> dict:
"""Return measured facts about the tokenizer file."""
path = Path(path) if path else DEFAULT_TOKENIZER
raw = json.loads(path.read_text(encoding="utf-8"))
model = raw.get("model", {})
added = raw.get("added_tokens", [])
vocab_size = len(model.get("vocab", {}))
added_ids = {t["id"] for t in added}
base_ids = set(range(vocab_size))
return {
"tokenizer_class": raw.get("tokenizer_class", "PreTrainedTokenizerFast"),
"model_type": model.get("type"),
"base_vocab_size": vocab_size,
"merges": len(model.get("merges", [])),
"added_tokens": len(added),
"unique_ids": len(base_ids | added_ids),
"overlapping_ids": len(base_ids & added_ids),
"max_added_id": max(added_ids) if added_ids else None,
}
def assert_vocab_matches_config(tokenizer: Tokenizer, vocab_size: int) -> None:
"""Fail loudly if the tokenizer and the model disagree on vocabulary size.
The CORTEX tokenizer exposes 128000 BPE entries plus 1283 added tokens, but ids 0,
1 and 2 appear in both sets. The number of *unique* ids is what must equal
``vocab_size``, so raw counters must never be compared directly.
"""
unique = tokenizer.get_vocab_size(with_added_tokens=True)
if unique != vocab_size:
raise ValueError(
f"tokenizer exposes {unique} unique ids but the model config declares "
f"vocab_size={vocab_size}"
)