ENContextV1 / app /normalize.py
ihhereanth's picture
Create normalize.py
f15eeff verified
Raw History Blame Contribute Delete
1.01 kB
import unicodedata, re
from pythainlp.util import normalize as th_normalize
ZERO_WIDTH = dict.fromkeys(map(ord, '\u200b\u200c\u200d\ufeff\u00ad'), None)
def normalize_th(s: str) -> str:
"""Canonical form — ต้องใช้ตัวเดียวกันนี้ทั้งใน pipeline และ runtime"""
if not s:
return ''
s = unicodedata.normalize('NFC', s) # 1. Unicode NFC
s = s.translate(ZERO_WIDTH) # 2. ลบ zero-width
s = th_normalize(s) # 3. จัดลำดับวรรณยุกต์/สระซ้ำ (PyThaiNLP)
s = re.sub(r'\s+', '', s) # 4. ลบช่องว่างภายใน
s = s.strip()
return s
# ทดสอบ: สระ/วรรณยุกต์สลับลำดับต้องยุบเป็นรูปเดียวกัน
assert normalize_th('เเมว') == normalize_th('แมว') # เ+เ vs แ
print("normalizer OK")