Spaces:
Sleeping
Sleeping
Download app/normalize.py from ihhereanth/ENContextV1: direct link, hf CLI and curl.
- Browser
- Download file 1.01 kB
-
https://huggingface.co/spaces/ihhereanth/ENContextV1/resolve/main/app/normalize.py
- Command line
-
hf download hf://spaces/ihhereanth/ENContextV1/app/normalize.py
-
curl -L -o normalize.py https://huggingface.co/spaces/ihhereanth/ENContextV1/resolve/main/app/normalize.py
1.01 kB
| import unicodedata, re | |
| from pythainlp.util import normalize as th_normalize | |
| ZERO_WIDTH = dict.fromkeys(map(ord, '\u200b\u200c\u200d\ufeff\u00ad'), None) | |
| def normalize_th(s: str) -> str: | |
| """Canonical form — ต้องใช้ตัวเดียวกันนี้ทั้งใน pipeline และ runtime""" | |
| if not s: | |
| return '' | |
| s = unicodedata.normalize('NFC', s) # 1. Unicode NFC | |
| s = s.translate(ZERO_WIDTH) # 2. ลบ zero-width | |
| s = th_normalize(s) # 3. จัดลำดับวรรณยุกต์/สระซ้ำ (PyThaiNLP) | |
| s = re.sub(r'\s+', '', s) # 4. ลบช่องว่างภายใน | |
| s = s.strip() | |
| return s | |
| # ทดสอบ: สระ/วรรณยุกต์สลับลำดับต้องยุบเป็นรูปเดียวกัน | |
| assert normalize_th('เเมว') == normalize_th('แมว') # เ+เ vs แ | |
| print("normalizer OK") |