File size: 12,019 Bytes
408fd1a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
"""Quickstart for the Hindko tokenizer (SentencePiece Unigram, 32,768 ids), release 1.0.0.

    pip install tokenizers transformers huggingface_hub
    export HF_TOKEN=hf_...            # the repository is private: a read token is required
    python examples/quickstart.py     # or: python examples/quickstart.py path/to/local/folder

What it shows
  1. loading with transformers (AutoTokenizer) and with tokenizers (the canonical encoder);
  2. encode / decode round trip (nothing is added automatically: no BOS/EOS);
  3. the ChatML chat template;
  4. normalize(): the corpus "data form" for messy input, and why it matters (token counts
     for raw vs normalized text, including Arabic-keyboard letters).

normalize() below is a self-contained adaptation of the corpus pipeline's hp.normalize 1.0.1
(rules R01-R14). It is written to produce the same output as that module; it does not import it.
"""
from __future__ import annotations

import os
import re
import sys
import unicodedata
from pathlib import Path

REPO_ID = "junaid008/hindko-tokenizer"

# ============================================================================ normalize()
# Canonical "data form" (hp.normalize 1.0.1). Deterministic, idempotent, NFC in and out.
# It folds only encoding noise and never merges letters that differ in Hindko/Urdu.
# Not reversible: after it, decode() returns normalize(x), not x. Do not use it where
# whitespace must be preserved exactly (code, tables, verbatim quotes).

NORMALIZATION_VERSION = "1.0.1"


def _cls(cps) -> str:
    cps = sorted(set(cps))
    out, i = [], 0
    while i < len(cps):
        j = i
        while j + 1 < len(cps) and cps[j + 1] == cps[j] + 1:
            j += 1
        a, b = cps[i], cps[j]
        out.append(re.escape(chr(a)) if a == b else re.escape(chr(a)) + "-" + re.escape(chr(b)))
        i = j + 1
    return "".join(out)


def _nfc(t: str) -> str:
    return t if unicodedata.is_normalized("NFC", t) else unicodedata.normalize("NFC", t)


ZWNJ, ZWJ, KASHIDA, SMALL_V = chr(0x200C), chr(0x200D), chr(0x0640), chr(0x065A)
ALLAH_LIGATURE = chr(0xFDF2)
ALLAH_WORD = "".join(map(chr, (0x0627, 0x0644, 0x0644, 0x06C1)))  # corpus spelling, HEH GOAL

ARABIC_BLOCKS = ((0x0600, 0x06FF), (0x0750, 0x077F), (0x0870, 0x089F), (0x08A0, 0x08FF),
                 (0xFB50, 0xFDFF), (0xFE70, 0xFEFF))
ARABIC_SCRIPT = [cp for lo, hi in ARABIC_BLOCKS for cp in range(lo, hi + 1)]
_WORD_CPS = [cp for lo, hi in ARABIC_BLOCKS[:4] for cp in range(lo, hi + 1)
             if unicodedata.category(chr(cp)) in ("Lo", "Lm", "Mn", "Mc")]
WORD_RE = re.compile("[" + _cls(_WORD_CPS + [0x200C]) + "]+")

# letters of Arabic orthography proper; letters only Urdu/Hindko use (evidence of Urdu orthography)
ARABIC_PROPER = frozenset(list(range(0x0621, 0x063B)) + list(range(0x0641, 0x064B)) + [0x0671, 0x066E, 0x066F])
URDU_EVIDENCE = frozenset([0x067E, 0x0679, 0x0686, 0x0688, 0x0691, 0x0698, 0x06A9, 0x06AF, 0x06BA, 0x06BE,
                           0x06C1, 0x06C2, 0x06C3, 0x06CC, 0x06D2, 0x06D3, 0x0768] + list(range(0x08BE, 0x08C3)))
URDU_EVIDENCE_MARKS = frozenset([0x065A])
HIGH_HAMZA_YEH = 0x0678
FOLD_ALLOWED = ARABIC_PROPER | URDU_EVIDENCE | {HIGH_HAMZA_YEH}
LETTER_FOLD = {0x064A: 0x06CC, 0x0649: 0x06CC, 0x0643: 0x06A9, 0x0629: 0x06C3, 0x0678: 0x0626}
_LETTER_FOLD_TABLE = {k: chr(v) for k, v in LETTER_FOLD.items()}
_FOLDABLE_RE = re.compile("[" + _cls(LETTER_FOLD) + "]")

TONE_BASE = {0x067E: 0x08BE, 0x062A: 0x08BF, 0x0679: 0x08C0, 0x0686: 0x08C1, 0x06A9: 0x08C2}
_TONE_MAP = {chr(k): chr(v) for k, v in TONE_BASE.items()}
TONE_RE = re.compile("([" + _cls(TONE_BASE) + "])([" + _cls(list(range(0x064B, 0x0653)) + [0x0670]) + "]*)"
                     + re.escape(SMALL_V))

NEWLINE_RE = re.compile("\r\n|[" + _cls([0x0D, 0x0B, 0x0C, 0x85, 0x2028, 0x2029]) + "]")
CONTROL_RE = re.compile("[" + _cls(list(range(0x00, 0x09)) + list(range(0x0B, 0x20)) + list(range(0x7F, 0xA0))) + "]")

PRESENTATION = [cp for lo, hi in ((0xFB50, 0xFDFF), (0xFE70, 0xFEFF)) for cp in range(lo, hi + 1)]
PRESENTATION_KEEP = frozenset(list(range(0xFD3E, 0xFD50)) + [0xFDCF, 0xFDFA, 0xFDFB, 0xFDFC, 0xFDFD, 0xFDFE, 0xFDFF])
PRESENTATION_RE = re.compile("[" + _cls(PRESENTATION) + "]")
ALLAH_RE = re.compile("[" + _cls([0x0627, 0xFE8D, 0xFE8E]) + "]?" + re.escape(ALLAH_LIGATURE))


def _presentation_target(cp: int) -> str:
    c = chr(cp)
    if cp in PRESENTATION_KEEP or cp == 0xFDF2:
        return c
    d = unicodedata.normalize("NFKC", c)
    if d == c:
        return c
    core = d.lstrip(" " + KASHIDA)
    if core and all(unicodedata.category(x) == "Mn" for x in core):
        return core
    if " " in d:
        return c
    return unicodedata.normalize("NFC", d)


_PRESENTATION_MAP = {chr(cp): _presentation_target(cp) for cp in PRESENTATION}

INVISIBLE = ([0x00AD, 0x034F, 0x061C, 0x180E, 0x200B, 0x200E, 0x200F, 0x2060, 0x2061, 0x2062, 0x2063, 0x2064,
              0xFEFF] + list(range(0x202A, 0x202F)) + list(range(0x2066, 0x2070)))
INVISIBLE_RE = re.compile("[" + _cls(INVISIBLE) + "]")
_ZWJ_NEIGH = _cls(ARABIC_SCRIPT + [0x200C, 0x200D]) + r"\s"
ZWJ_RE = re.compile("(?<![^" + _ZWJ_NEIGH + "])" + ZWJ + "|" + ZWJ + "(?![^" + _ZWJ_NEIGH + "])")
SPACES = [0x09] + [cp for cp in range(0x80, 0x3001) if unicodedata.category(chr(cp)) == "Zs"]
SPACES_RE = re.compile("[" + _cls(SPACES) + "]")
MULTISPACE_RE = re.compile(" {2,}")
LINE_EDGE_SPACE_RE = re.compile("^ +| +$", re.M)
MULTI_NL_RE = re.compile("\n{3,}")
ARABIC_INDIC_DIGITS = {chr(0x0660 + i): chr(0x06F0 + i) for i in range(10)}
ARABIC_INDIC_RE = re.compile("[" + _cls(range(0x0660, 0x066A)) + "]")


def _presentation(t: str) -> str:
    if not PRESENTATION_RE.search(t):
        return t
    t = ALLAH_RE.sub(ALLAH_WORD, t)
    return PRESENTATION_RE.sub(lambda m: _PRESENTATION_MAP[m.group()], t)


def _zwnj(t: str) -> tuple[str, int]:
    if ZWNJ not in t:
        return t, 0
    out, removed, i, n = [], 0, 0, len(t)
    while i < n:
        c = t[i]
        if c != ZWNJ:
            out.append(c)
            i += 1
            continue
        j = i
        while j < n and t[j] == ZWNJ:
            j += 1
        prev_ok = bool(out) and unicodedata.category(out[-1])[0] in "LM"
        next_ok = j < n and unicodedata.category(t[j])[0] in "LM"
        if prev_ok and next_ok:
            out.append(ZWNJ)
            removed += (j - i) - 1
        else:
            removed += j - i
        i = j
    return "".join(out), removed


def _spaces(t: str) -> str:
    t = SPACES_RE.sub(" ", t)
    t = MULTISPACE_RE.sub(" ", t)
    t = LINE_EDGE_SPACE_RE.sub("", t)
    return MULTI_NL_RE.sub("\n\n", t)


def _word_is_urdu_orthography(word: str) -> bool:
    evidence = False
    for c in word:
        o = ord(c)
        if unicodedata.category(c)[0] == "M":
            evidence = evidence or o in URDU_EVIDENCE_MARKS
            continue
        if o == 0x200C:
            continue
        if o not in FOLD_ALLOWED:
            return False
        if o in URDU_EVIDENCE or o == HIGH_HAMZA_YEH:
            evidence = True
    return evidence


def _arabic_letters(t: str) -> str:
    if not _FOLDABLE_RE.search(t):
        return t

    def f(m):
        w = m.group()
        if _FOLDABLE_RE.search(w) and _word_is_urdu_orthography(w):
            return w.translate(_LETTER_FOLD_TABLE)
        return w
    return WORD_RE.sub(f, t)


def normalize(text: str) -> str:
    """Return the canonical data form of `text` (the form the tokenizer was trained and evaluated on)."""
    t = _nfc(text)                                           # R01
    t = NEWLINE_RE.sub("\n", t)                              # R02
    t = CONTROL_RE.sub("", t)                                # R03
    t = _presentation(t)                                     # R04
    t = t.replace(KASHIDA, "")                               # R05
    t = INVISIBLE_RE.sub("", t)                              # R06
    t = _nfc(t)                                              # R10 before R07/R08
    for _ in range(t.count(ZWJ) + t.count(ZWNJ) + 1):        # R07 R08 R09 R10 until stable
        t, k1 = ZWJ_RE.subn("", t) if ZWJ in t else (t, 0)
        t, k2 = _zwnj(t)
        t = _nfc(_spaces(t))
        if not (k1 or k2):
            break
    t = _arabic_letters(t)                                   # R11 (only inside provably Urdu/Hindko words)
    if SMALL_V in t:                                         # R12 Hindko tone letters
        t = TONE_RE.sub(lambda m: _TONE_MAP[m.group(1)] + m.group(2), t)
    t = ARABIC_INDIC_RE.sub(lambda m: ARABIC_INDIC_DIGITS[m.group()], t)  # R13
    return _nfc(t)                                           # R14


def fold_arabic_yeh_kaf(text: str) -> str:
    """Optional, stronger fix for Arabic-keyboard input: fold EVERY ي->ی and ك->ک.

    Use only when the input is known to be Hindko or Urdu (it also rewrites Arabic quotations).
    ه (Arabic HEH) is deliberately not folded: Arabic-keyboard users type it for both ہ and ھ."""
    return text.replace("ي", "ی").replace("ك", "ک")


# ============================================================================ demo
def load(path_or_repo: str):
    """Load with transformers if available, else with tokenizers. Works for a local folder or the hub."""
    token = os.environ.get("HF_TOKEN")
    try:
        from transformers import AutoTokenizer
        return "transformers", AutoTokenizer.from_pretrained(path_or_repo, token=token)
    except ImportError:
        from tokenizers import Tokenizer
        if Path(path_or_repo).is_dir():
            return "tokenizers", Tokenizer.from_file(str(Path(path_or_repo) / "tokenizer.json"))
        return "tokenizers", Tokenizer.from_pretrained(path_or_repo, token=token)


def main() -> None:
    src = sys.argv[1] if len(sys.argv) > 1 else (
        str(Path(__file__).resolve().parents[1]) if (Path(__file__).resolve().parents[1] / "tokenizer.json").exists()
        else REPO_ID)
    kind, tok = load(src)
    enc = (lambda s: tok(s)["input_ids"]) if kind == "transformers" else (lambda s: tok.encode(s).ids)
    dec = ((lambda ids: tok.decode(ids, skip_special_tokens=False, clean_up_tokenization_spaces=False))
           if kind == "transformers" else (lambda ids: tok.decode(ids, skip_special_tokens=False)))
    pieces = (lambda ids: tok.convert_ids_to_tokens(ids)) if kind == "transformers" else (
        lambda ids: [tok.id_to_token(i) for i in ids])
    print(f"loaded from {src} with {kind}")

    # 1. encode / decode (a sentence from the held-out test split; ~ "Very few people know how and where drama was born.")
    text = "ایہہ بہت کہٹ لوک جانڑدین کہ ڈرامے دی پیدائش کسراں تے کتھے ہوئی آئی۔"
    ids = enc(text)
    print(f"\n{len(text.split())} words -> {len(ids)} tokens")
    print(" | ".join(pieces(ids)))
    assert dec(ids) == text, "round trip failed"
    print("round trip exact: True  (no BOS/EOS added; add <|bos|>=1 / <|endoftext|>=0 yourself)")

    # 2. chat template (transformers only)
    if kind == "transformers":
        msgs = [{"role": "user", "content": "سلام! تساں کیہہ حال اے؟"}]
        rendered = tok.apply_chat_template(msgs, tokenize=False, add_generation_prompt=True)
        print("\nchat template:", repr(rendered))

    # 3. messy input: normalize() first
    messy = "ایہہ  بہت​ کہٹ لوک\r\nجانڑدین کہ ڈرامے دی پیدائش"
    print(f"\nmessy input: raw {len(enc(messy))} tokens, normalize() {len(enc(normalize(messy)))} tokens")

    # 4. Arabic-keyboard letters (ي ك ه typed for ی ک ہ): the tokenizer's clearest weak spot
    ak = text.replace("ی", "ي").replace("ک", "ك").replace("ہ", "ه")
    n_clean, n_ak = len(enc(text)), len(enc(ak))
    n_norm, n_fold = len(enc(normalize(ak))), len(enc(fold_arabic_yeh_kaf(normalize(ak))))
    print(f"Arabic-keyboard typing: clean {n_clean}, as typed {n_ak}, normalize() {n_norm}, "
          f"normalize()+YEH/KAF fold {n_fold} tokens")


if __name__ == "__main__":
    main()