"""Adversarial text normalisation, matching the production OpenTextShield API. Trimmed standalone copy of EnhancedPreprocessor.normalize_unicode from https://github.com/TelecomsXChangeAPi/OpenTextShield (src/api_interface/services/enhanced_preprocessing.py), so the demo Space classifies obfuscated text the same way the deployed API does. """ import re import unicodedata # Zero-width / invisible formatting characters used in obfuscation attacks. INVISIBLE_CHARS = frozenset({ "​", "‌", "‍", "‎", "‏", "⁠", "⁡", "⁢", "⁣", "⁤", "", "­", "᠎", "͏", "؜", }) # Cyrillic and Greek letters that render like Latin ones. Folded only where # they are plausibly a spoof, never in real Russian, Ukrainian or Greek text. CONFUSABLES = { "а": "a", "с": "c", "ԁ": "d", "е": "e", "һ": "h", "і": "i", "ј": "j", "ӏ": "l", "о": "o", "р": "p", "ԛ": "q", "ѕ": "s", "ԝ": "w", "х": "x", "у": "y", "А": "A", "В": "B", "С": "C", "Е": "E", "Н": "H", "І": "I", "Ӏ": "I", "Ј": "J", "К": "K", "М": "M", "О": "O", "Р": "P", "Ԛ": "Q", "Ѕ": "S", "Т": "T", "Ԝ": "W", "Х": "X", "Ү": "Y", "α": "a", "ι": "i", "κ": "k", "ν": "v", "ο": "o", "ρ": "p", "τ": "t", "υ": "u", "χ": "x", "Α": "A", "Β": "B", "Ε": "E", "Ζ": "Z", "Η": "H", "Ι": "I", "Κ": "K", "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P", "Τ": "T", "Υ": "Y", "Χ": "X", } def _script(ch: str) -> str: try: return unicodedata.name(ch).split(" ", 1)[0] except ValueError: return "" def _fold_compat_alnum(ch: str) -> str: if ch.isascii() or unicodedata.category(ch)[0] not in "LN": return ch folded = unicodedata.normalize("NFKC", ch) return folded if len(folded) == 1 and folded.isascii() and folded.isalnum() else ch def _fold_spoofed_word(word: str, mostly_latin: bool) -> str: foreign = [ch for ch in word if ch.isalpha() and _script(ch) in ("CYRILLIC", "GREEK")] if not foreign or any(ch not in CONFUSABLES for ch in foreign): return word has_latin = any(ch.isalpha() and _script(ch) == "LATIN" for ch in word) if has_latin or mostly_latin: return "".join(CONFUSABLES.get(ch, ch) for ch in word) return word def normalize_unicode(text: str) -> str: """Undo text obfuscation without damaging real non-Latin text.""" text = unicodedata.normalize("NFC", text) text = "".join(ch for ch in text if ch not in INVISIBLE_CHARS) text = "".join(_fold_compat_alnum(ch) for ch in text) letters = [ch for ch in text if ch.isalpha()] latin = sum(1 for ch in letters if _script(ch) == "LATIN") mostly_latin = bool(letters) and latin * 2 > len(letters) if mostly_latin: text = "".join( unicodedata.normalize("NFKC", ch) if "!" <= ch <= "~" or ch == " " else ch for ch in text ) return re.sub(r"\S+", lambda m: _fold_spoofed_word(m.group(), mostly_latin), text)