File size: 2,629 Bytes
6fbe100
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
"""CORTEX tokenizer — loading and validating the CORTEX vocabulary.

The CORTEX tokenizer is reused unchanged from the distributed repository
(``Frankenstein-Labs/Cortex-ai``, MIT). This module never rebuilds or mutates it; it
loads the shipped ``tokenizer.json`` and exposes helpers used by the pipeline.
"""

from __future__ import annotations

import json
from pathlib import Path

from tokenizers import Tokenizer

__all__ = ["load_tokenizer", "describe_tokenizer", "assert_vocab_matches_config"]

DEFAULT_TOKENIZER = Path(__file__).resolve().parent.parent / "tokenizer.json"


def load_tokenizer(path: str | Path | None = None) -> Tokenizer:
    """Load the CORTEX BPE tokenizer.

    Defaults to the ``tokenizer.json`` shipped at the repository root, so the pipeline
    and the distributed checkpoint use one and the same vocabulary file.
    """
    path = Path(path) if path else DEFAULT_TOKENIZER
    if not path.exists():
        raise FileNotFoundError(
            f"tokenizer not found at {path}. See tokenizer/README.md for how it is obtained."
        )
    return Tokenizer.from_file(str(path))


def describe_tokenizer(path: str | Path | None = None) -> dict:
    """Return measured facts about the tokenizer file."""
    path = Path(path) if path else DEFAULT_TOKENIZER
    raw = json.loads(path.read_text(encoding="utf-8"))
    model = raw.get("model", {})
    added = raw.get("added_tokens", [])
    vocab_size = len(model.get("vocab", {}))
    added_ids = {t["id"] for t in added}
    base_ids = set(range(vocab_size))
    return {
        "tokenizer_class": raw.get("tokenizer_class", "PreTrainedTokenizerFast"),
        "model_type": model.get("type"),
        "base_vocab_size": vocab_size,
        "merges": len(model.get("merges", [])),
        "added_tokens": len(added),
        "unique_ids": len(base_ids | added_ids),
        "overlapping_ids": len(base_ids & added_ids),
        "max_added_id": max(added_ids) if added_ids else None,
    }


def assert_vocab_matches_config(tokenizer: Tokenizer, vocab_size: int) -> None:
    """Fail loudly if the tokenizer and the model disagree on vocabulary size.

    The CORTEX tokenizer exposes 128000 BPE entries plus 1283 added tokens, but ids 0,
    1 and 2 appear in both sets. The number of *unique* ids is what must equal
    ``vocab_size``, so raw counters must never be compared directly.
    """
    unique = tokenizer.get_vocab_size(with_added_tokens=True)
    if unique != vocab_size:
        raise ValueError(
            f"tokenizer exposes {unique} unique ids but the model config declares "
            f"vocab_size={vocab_size}"
        )