""" normalize.py ============ Hinglish-specific text normalization for comment moderation. Handles: - Common Hinglish slur variant normalization - Leetspeak decoding - Phonetic variant mapping - Obfuscated profanity detection Usage: python -c "from preprocessing.normalize import normalize_hinglish; print(normalize_hinglish('ch00t1y4'))" """ import re from typing import Dict, List, Optional # --------------------------------------------------------------------------- # Leetspeak / character substitution mapping # --------------------------------------------------------------------------- LEET_MAP: Dict[str, str] = { "0": "o", "1": "i", "3": "e", "4": "a", "5": "s", "7": "t", "8": "b", "@": "a", "$": "s", "!": "i", "|": "l", } # Compiled regex for leetspeak characters RE_LEET = re.compile(r"[013457@$!|]") def decode_leetspeak(text: str) -> str: """ Convert leetspeak characters to their alphabetic equivalents. Examples: ch00t1y4 → chootiya h4ck3r → hacker """ return RE_LEET.sub(lambda m: LEET_MAP.get(m.group(), m.group()), text) # --------------------------------------------------------------------------- # Hinglish profanity variant normalization # --------------------------------------------------------------------------- # Maps common spelling variants to canonical forms. # This helps the model by reducing surface variation. # All entries are lowercase. HINGLISH_VARIANTS: Dict[str, List[str]] = { # --- Profanity canonical forms --- "madarchod": [ "madarchodd", "madarchd", "mc", "m.c.", "m c", "madarchoot", "madarchodu", "maderchod", "maderchoot", "maadarchod", "madarchoodd", "madr chod", ], "bhenchod": [ "benchod", "bhenchod", "bc", "b.c.", "b c", "behenchod", "bhnchod", "bhenchoot", "behen chod", "bhen chod", "bhenchodu", ], "chutiya": [ "chootiya", "chutia", "chutiye", "chutiyaa", "chootia", "chutiyo", "chutiyee", "chutiyapa", "ch00tiya", "chut1ya", ], "gaandu": [ "gandu", "gaand", "gaandd", "gand", "gaanduu", "g4ndu", "ganduu", ], "randi": [ "randii", "rundi", "randiya", "randiyo", "r4ndi", "randwe", ], "harami": [ "haraami", "haramii", "haram1", "haramkhor", "haraamii", ], "kutte": [ "kuttee", "kutta", "kuttte", "kuttey", "kutt3", "kutiya", ], "sala": [ "saala", "sale", "saaale", "saale", "s4la", "s4le", ], "bhosdike": [ "bsdk", "bhosdiwale", "bhosdika", "bhosdki", "bh0sdike", "bhosdi", ], "lodu": [ "laude", "laudu", "lauda", "l0du", "lavde", "lawde", "laudey", ], # --- Threat-related --- "maar dunga": [ "maar daaluga", "maar dalunga", "maarunga", "maar deta", "maar khayega", ], # --- Insult variants --- "pagal": [ "paagal", "pagall", "p4gal", "pagl", ], "bewakoof": [ "bevkoof", "bewkoof", "bewaqoof", "bevakoof", "b3wakoof", ], "gadha": [ "gadhe", "gadhaa", "g4dha", ], } # Build reverse lookup: variant → canonical _VARIANT_TO_CANONICAL: Dict[str, str] = {} for canonical, variants in HINGLISH_VARIANTS.items(): for variant in variants: _VARIANT_TO_CANONICAL[variant] = canonical # Sort by length (longest first) to match multi-word variants first _SORTED_VARIANTS = sorted(_VARIANT_TO_CANONICAL.keys(), key=len, reverse=True) # Build a regex pattern for word-boundary matching _VARIANT_PATTERN = re.compile( r"\b(" + "|".join(re.escape(v) for v in _SORTED_VARIANTS) + r")\b", re.IGNORECASE, ) def normalize_hinglish_slurs(text: str) -> str: """ Replace variant spellings of Hinglish profanity/slurs with their canonical forms to reduce surface variation for the model. Args: text: Lowercased input text. Returns: Text with normalized slur spellings. """ def _replace(match): variant = match.group(0).lower() return _VARIANT_TO_CANONICAL.get(variant, variant) return _VARIANT_PATTERN.sub(_replace, text) # --------------------------------------------------------------------------- # Obfuscation pattern detection # --------------------------------------------------------------------------- # Common obfuscation: inserting dots, spaces, or special chars within words # e.g., "f.u.c.k" or "f u c k" or "f*ck" RE_DOTTED_WORD = re.compile(r"\b(\w)(?:[.\-_*#](\w)){2,}\b") def normalize_obfuscation(text: str) -> str: """ Remove common obfuscation patterns like dots/stars between chars. Examples: f.u.c.k → fuck s.h.i.t → shit """ # Remove single-char separators between word characters text = re.sub(r"(?<=\w)[.\-_*#](?=\w)", "", text) return text # --------------------------------------------------------------------------- # Main normalization function # --------------------------------------------------------------------------- def normalize_hinglish(text: Optional[str]) -> str: """ Full Hinglish normalization pipeline. 1. Decode leetspeak 2. Remove obfuscation patterns 3. Normalize slur variants to canonical forms Args: text: Input text (should already be lowercased). Returns: Normalized text. """ if not text or not isinstance(text, str): return "" text = text.lower() text = decode_leetspeak(text) text = normalize_obfuscation(text) text = normalize_hinglish_slurs(text) return text # --------------------------------------------------------------------------- # CLI # --------------------------------------------------------------------------- def main(): """Quick demo of normalization.""" test_cases = [ "ch00t1y4", "Tu bsdk pagal hai", "m.c. sale kutte", "Bhai app slow hai", "f.u.c.k you", "bhenchoot tujhe maar dalunga", "Website bahut acha hai", "Tu bewkoof hai kya", ] print("=" * 60) print("Hinglish Normalization Demo") print("=" * 60) for text in test_cases: normalized = normalize_hinglish(text) print(f" {text:40s} → {normalized}") print("=" * 60) if __name__ == "__main__": main()