heatherlead's picture
Add moderation backend with LFS tracking for large files
26c8f44
Raw History Blame Contribute Delete
6.39 kB
"""
normalize.py
============
Hinglish-specific text normalization for comment moderation.
Handles:
- Common Hinglish slur variant normalization
- Leetspeak decoding
- Phonetic variant mapping
- Obfuscated profanity detection
Usage:
python -c "from preprocessing.normalize import normalize_hinglish; print(normalize_hinglish('ch00t1y4'))"
"""
import re
from typing import Dict, List, Optional
# ---------------------------------------------------------------------------
# Leetspeak / character substitution mapping
# ---------------------------------------------------------------------------
LEET_MAP: Dict[str, str] = {
"0": "o",
"1": "i",
"3": "e",
"4": "a",
"5": "s",
"7": "t",
"8": "b",
"@": "a",
"$": "s",
"!": "i",
"|": "l",
}
# Compiled regex for leetspeak characters
RE_LEET = re.compile(r"[013457@$!|]")
def decode_leetspeak(text: str) -> str:
"""
Convert leetspeak characters to their alphabetic equivalents.
Examples:
ch00t1y4 → chootiya
h4ck3r → hacker
"""
return RE_LEET.sub(lambda m: LEET_MAP.get(m.group(), m.group()), text)
# ---------------------------------------------------------------------------
# Hinglish profanity variant normalization
# ---------------------------------------------------------------------------
# Maps common spelling variants to canonical forms.
# This helps the model by reducing surface variation.
# All entries are lowercase.
HINGLISH_VARIANTS: Dict[str, List[str]] = {
# --- Profanity canonical forms ---
"madarchod": [
"madarchodd", "madarchd", "mc", "m.c.", "m c",
"madarchoot", "madarchodu", "maderchod", "maderchoot",
"maadarchod", "madarchoodd", "madr chod",
],
"bhenchod": [
"benchod", "bhenchod", "bc", "b.c.", "b c",
"behenchod", "bhnchod", "bhenchoot", "behen chod",
"bhen chod", "bhenchodu",
],
"chutiya": [
"chootiya", "chutia", "chutiye", "chutiyaa",
"chootia", "chutiyo", "chutiyee", "chutiyapa",
"ch00tiya", "chut1ya",
],
"gaandu": [
"gandu", "gaand", "gaandd", "gand", "gaanduu",
"g4ndu", "ganduu",
],
"randi": [
"randii", "rundi", "randiya", "randiyo",
"r4ndi", "randwe",
],
"harami": [
"haraami", "haramii", "haram1", "haramkhor",
"haraamii",
],
"kutte": [
"kuttee", "kutta", "kuttte", "kuttey",
"kutt3", "kutiya",
],
"sala": [
"saala", "sale", "saaale", "saale",
"s4la", "s4le",
],
"bhosdike": [
"bsdk", "bhosdiwale", "bhosdika", "bhosdki",
"bh0sdike", "bhosdi",
],
"lodu": [
"laude", "laudu", "lauda", "l0du",
"lavde", "lawde", "laudey",
],
# --- Threat-related ---
"maar dunga": [
"maar daaluga", "maar dalunga", "maarunga",
"maar deta", "maar khayega",
],
# --- Insult variants ---
"pagal": [
"paagal", "pagall", "p4gal", "pagl",
],
"bewakoof": [
"bevkoof", "bewkoof", "bewaqoof", "bevakoof",
"b3wakoof",
],
"gadha": [
"gadhe", "gadhaa", "g4dha",
],
}
# Build reverse lookup: variant → canonical
_VARIANT_TO_CANONICAL: Dict[str, str] = {}
for canonical, variants in HINGLISH_VARIANTS.items():
for variant in variants:
_VARIANT_TO_CANONICAL[variant] = canonical
# Sort by length (longest first) to match multi-word variants first
_SORTED_VARIANTS = sorted(_VARIANT_TO_CANONICAL.keys(), key=len, reverse=True)
# Build a regex pattern for word-boundary matching
_VARIANT_PATTERN = re.compile(
r"\b(" + "|".join(re.escape(v) for v in _SORTED_VARIANTS) + r")\b",
re.IGNORECASE,
)
def normalize_hinglish_slurs(text: str) -> str:
"""
Replace variant spellings of Hinglish profanity/slurs with
their canonical forms to reduce surface variation for the model.
Args:
text: Lowercased input text.
Returns:
Text with normalized slur spellings.
"""
def _replace(match):
variant = match.group(0).lower()
return _VARIANT_TO_CANONICAL.get(variant, variant)
return _VARIANT_PATTERN.sub(_replace, text)
# ---------------------------------------------------------------------------
# Obfuscation pattern detection
# ---------------------------------------------------------------------------
# Common obfuscation: inserting dots, spaces, or special chars within words
# e.g., "f.u.c.k" or "f u c k" or "f*ck"
RE_DOTTED_WORD = re.compile(r"\b(\w)(?:[.\-_*#](\w)){2,}\b")
def normalize_obfuscation(text: str) -> str:
"""
Remove common obfuscation patterns like dots/stars between chars.
Examples:
f.u.c.k → fuck
s.h.i.t → shit
"""
# Remove single-char separators between word characters
text = re.sub(r"(?<=\w)[.\-_*#](?=\w)", "", text)
return text
# ---------------------------------------------------------------------------
# Main normalization function
# ---------------------------------------------------------------------------
def normalize_hinglish(text: Optional[str]) -> str:
"""
Full Hinglish normalization pipeline.
1. Decode leetspeak
2. Remove obfuscation patterns
3. Normalize slur variants to canonical forms
Args:
text: Input text (should already be lowercased).
Returns:
Normalized text.
"""
if not text or not isinstance(text, str):
return ""
text = text.lower()
text = decode_leetspeak(text)
text = normalize_obfuscation(text)
text = normalize_hinglish_slurs(text)
return text
# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------
def main():
"""Quick demo of normalization."""
test_cases = [
"ch00t1y4",
"Tu bsdk pagal hai",
"m.c. sale kutte",
"Bhai app slow hai",
"f.u.c.k you",
"bhenchoot tujhe maar dalunga",
"Website bahut acha hai",
"Tu bewkoof hai kya",
]
print("=" * 60)
print("Hinglish Normalization Demo")
print("=" * 60)
for text in test_cases:
normalized = normalize_hinglish(text)
print(f" {text:40s} → {normalized}")
print("=" * 60)
if __name__ == "__main__":
main()