import unicodedata, re from pythainlp.util import normalize as th_normalize ZERO_WIDTH = dict.fromkeys(map(ord, '\u200b\u200c\u200d\ufeff\u00ad'), None) def normalize_th(s: str) -> str: """Canonical form — ต้องใช้ตัวเดียวกันนี้ทั้งใน pipeline และ runtime""" if not s: return '' s = unicodedata.normalize('NFC', s) # 1. Unicode NFC s = s.translate(ZERO_WIDTH) # 2. ลบ zero-width s = th_normalize(s) # 3. จัดลำดับวรรณยุกต์/สระซ้ำ (PyThaiNLP) s = re.sub(r'\s+', '', s) # 4. ลบช่องว่างภายใน s = s.strip() return s # ทดสอบ: สระ/วรรณยุกต์สลับลำดับต้องยุบเป็นรูปเดียวกัน assert normalize_th('เเมว') == normalize_th('แมว') # เ+เ vs แ print("normalizer OK")