Spaces:
Running on Zero
Running on Zero
Download preprocessing/normalize.py from heatherlead/comment_moderation: direct link, hf CLI and curl.
- Browser
- Download file 6.39 kB
-
https://huggingface.co/spaces/heatherlead/comment_moderation/resolve/main/preprocessing/normalize.py
- Command line
-
hf download hf://spaces/heatherlead/comment_moderation/preprocessing/normalize.py
-
curl -L -o normalize.py https://huggingface.co/spaces/heatherlead/comment_moderation/resolve/main/preprocessing/normalize.py
6.39 kB
| """ | |
| normalize.py | |
| ============ | |
| Hinglish-specific text normalization for comment moderation. | |
| Handles: | |
| - Common Hinglish slur variant normalization | |
| - Leetspeak decoding | |
| - Phonetic variant mapping | |
| - Obfuscated profanity detection | |
| Usage: | |
| python -c "from preprocessing.normalize import normalize_hinglish; print(normalize_hinglish('ch00t1y4'))" | |
| """ | |
| import re | |
| from typing import Dict, List, Optional | |
| # --------------------------------------------------------------------------- | |
| # Leetspeak / character substitution mapping | |
| # --------------------------------------------------------------------------- | |
| LEET_MAP: Dict[str, str] = { | |
| "0": "o", | |
| "1": "i", | |
| "3": "e", | |
| "4": "a", | |
| "5": "s", | |
| "7": "t", | |
| "8": "b", | |
| "@": "a", | |
| "$": "s", | |
| "!": "i", | |
| "|": "l", | |
| } | |
| # Compiled regex for leetspeak characters | |
| RE_LEET = re.compile(r"[013457@$!|]") | |
| def decode_leetspeak(text: str) -> str: | |
| """ | |
| Convert leetspeak characters to their alphabetic equivalents. | |
| Examples: | |
| ch00t1y4 → chootiya | |
| h4ck3r → hacker | |
| """ | |
| return RE_LEET.sub(lambda m: LEET_MAP.get(m.group(), m.group()), text) | |
| # --------------------------------------------------------------------------- | |
| # Hinglish profanity variant normalization | |
| # --------------------------------------------------------------------------- | |
| # Maps common spelling variants to canonical forms. | |
| # This helps the model by reducing surface variation. | |
| # All entries are lowercase. | |
| HINGLISH_VARIANTS: Dict[str, List[str]] = { | |
| # --- Profanity canonical forms --- | |
| "madarchod": [ | |
| "madarchodd", "madarchd", "mc", "m.c.", "m c", | |
| "madarchoot", "madarchodu", "maderchod", "maderchoot", | |
| "maadarchod", "madarchoodd", "madr chod", | |
| ], | |
| "bhenchod": [ | |
| "benchod", "bhenchod", "bc", "b.c.", "b c", | |
| "behenchod", "bhnchod", "bhenchoot", "behen chod", | |
| "bhen chod", "bhenchodu", | |
| ], | |
| "chutiya": [ | |
| "chootiya", "chutia", "chutiye", "chutiyaa", | |
| "chootia", "chutiyo", "chutiyee", "chutiyapa", | |
| "ch00tiya", "chut1ya", | |
| ], | |
| "gaandu": [ | |
| "gandu", "gaand", "gaandd", "gand", "gaanduu", | |
| "g4ndu", "ganduu", | |
| ], | |
| "randi": [ | |
| "randii", "rundi", "randiya", "randiyo", | |
| "r4ndi", "randwe", | |
| ], | |
| "harami": [ | |
| "haraami", "haramii", "haram1", "haramkhor", | |
| "haraamii", | |
| ], | |
| "kutte": [ | |
| "kuttee", "kutta", "kuttte", "kuttey", | |
| "kutt3", "kutiya", | |
| ], | |
| "sala": [ | |
| "saala", "sale", "saaale", "saale", | |
| "s4la", "s4le", | |
| ], | |
| "bhosdike": [ | |
| "bsdk", "bhosdiwale", "bhosdika", "bhosdki", | |
| "bh0sdike", "bhosdi", | |
| ], | |
| "lodu": [ | |
| "laude", "laudu", "lauda", "l0du", | |
| "lavde", "lawde", "laudey", | |
| ], | |
| # --- Threat-related --- | |
| "maar dunga": [ | |
| "maar daaluga", "maar dalunga", "maarunga", | |
| "maar deta", "maar khayega", | |
| ], | |
| # --- Insult variants --- | |
| "pagal": [ | |
| "paagal", "pagall", "p4gal", "pagl", | |
| ], | |
| "bewakoof": [ | |
| "bevkoof", "bewkoof", "bewaqoof", "bevakoof", | |
| "b3wakoof", | |
| ], | |
| "gadha": [ | |
| "gadhe", "gadhaa", "g4dha", | |
| ], | |
| } | |
| # Build reverse lookup: variant → canonical | |
| _VARIANT_TO_CANONICAL: Dict[str, str] = {} | |
| for canonical, variants in HINGLISH_VARIANTS.items(): | |
| for variant in variants: | |
| _VARIANT_TO_CANONICAL[variant] = canonical | |
| # Sort by length (longest first) to match multi-word variants first | |
| _SORTED_VARIANTS = sorted(_VARIANT_TO_CANONICAL.keys(), key=len, reverse=True) | |
| # Build a regex pattern for word-boundary matching | |
| _VARIANT_PATTERN = re.compile( | |
| r"\b(" + "|".join(re.escape(v) for v in _SORTED_VARIANTS) + r")\b", | |
| re.IGNORECASE, | |
| ) | |
| def normalize_hinglish_slurs(text: str) -> str: | |
| """ | |
| Replace variant spellings of Hinglish profanity/slurs with | |
| their canonical forms to reduce surface variation for the model. | |
| Args: | |
| text: Lowercased input text. | |
| Returns: | |
| Text with normalized slur spellings. | |
| """ | |
| def _replace(match): | |
| variant = match.group(0).lower() | |
| return _VARIANT_TO_CANONICAL.get(variant, variant) | |
| return _VARIANT_PATTERN.sub(_replace, text) | |
| # --------------------------------------------------------------------------- | |
| # Obfuscation pattern detection | |
| # --------------------------------------------------------------------------- | |
| # Common obfuscation: inserting dots, spaces, or special chars within words | |
| # e.g., "f.u.c.k" or "f u c k" or "f*ck" | |
| RE_DOTTED_WORD = re.compile(r"\b(\w)(?:[.\-_*#](\w)){2,}\b") | |
| def normalize_obfuscation(text: str) -> str: | |
| """ | |
| Remove common obfuscation patterns like dots/stars between chars. | |
| Examples: | |
| f.u.c.k → fuck | |
| s.h.i.t → shit | |
| """ | |
| # Remove single-char separators between word characters | |
| text = re.sub(r"(?<=\w)[.\-_*#](?=\w)", "", text) | |
| return text | |
| # --------------------------------------------------------------------------- | |
| # Main normalization function | |
| # --------------------------------------------------------------------------- | |
| def normalize_hinglish(text: Optional[str]) -> str: | |
| """ | |
| Full Hinglish normalization pipeline. | |
| 1. Decode leetspeak | |
| 2. Remove obfuscation patterns | |
| 3. Normalize slur variants to canonical forms | |
| Args: | |
| text: Input text (should already be lowercased). | |
| Returns: | |
| Normalized text. | |
| """ | |
| if not text or not isinstance(text, str): | |
| return "" | |
| text = text.lower() | |
| text = decode_leetspeak(text) | |
| text = normalize_obfuscation(text) | |
| text = normalize_hinglish_slurs(text) | |
| return text | |
| # --------------------------------------------------------------------------- | |
| # CLI | |
| # --------------------------------------------------------------------------- | |
| def main(): | |
| """Quick demo of normalization.""" | |
| test_cases = [ | |
| "ch00t1y4", | |
| "Tu bsdk pagal hai", | |
| "m.c. sale kutte", | |
| "Bhai app slow hai", | |
| "f.u.c.k you", | |
| "bhenchoot tujhe maar dalunga", | |
| "Website bahut acha hai", | |
| "Tu bewkoof hai kya", | |
| ] | |
| print("=" * 60) | |
| print("Hinglish Normalization Demo") | |
| print("=" * 60) | |
| for text in test_cases: | |
| normalized = normalize_hinglish(text) | |
| print(f" {text:40s} → {normalized}") | |
| print("=" * 60) | |
| if __name__ == "__main__": | |
| main() | |