File size: 6,390 Bytes
26c8f44
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
"""
normalize.py
============
Hinglish-specific text normalization for comment moderation.

Handles:
  - Common Hinglish slur variant normalization
  - Leetspeak decoding
  - Phonetic variant mapping
  - Obfuscated profanity detection

Usage:
    python -c "from preprocessing.normalize import normalize_hinglish; print(normalize_hinglish('ch00t1y4'))"
"""

import re
from typing import Dict, List, Optional


# ---------------------------------------------------------------------------
# Leetspeak / character substitution mapping
# ---------------------------------------------------------------------------

LEET_MAP: Dict[str, str] = {
    "0": "o",
    "1": "i",
    "3": "e",
    "4": "a",
    "5": "s",
    "7": "t",
    "8": "b",
    "@": "a",
    "$": "s",
    "!": "i",
    "|": "l",
}

# Compiled regex for leetspeak characters
RE_LEET = re.compile(r"[013457@$!|]")


def decode_leetspeak(text: str) -> str:
    """
    Convert leetspeak characters to their alphabetic equivalents.

    Examples:
        ch00t1y4 → chootiya
        h4ck3r  → hacker
    """
    return RE_LEET.sub(lambda m: LEET_MAP.get(m.group(), m.group()), text)


# ---------------------------------------------------------------------------
# Hinglish profanity variant normalization
# ---------------------------------------------------------------------------
# Maps common spelling variants to canonical forms.
# This helps the model by reducing surface variation.
# All entries are lowercase.

HINGLISH_VARIANTS: Dict[str, List[str]] = {
    # --- Profanity canonical forms ---
    "madarchod": [
        "madarchodd", "madarchd", "mc", "m.c.", "m c",
        "madarchoot", "madarchodu", "maderchod", "maderchoot",
        "maadarchod", "madarchoodd", "madr chod",
    ],
    "bhenchod": [
        "benchod", "bhenchod", "bc", "b.c.", "b c",
        "behenchod", "bhnchod", "bhenchoot", "behen chod",
        "bhen chod", "bhenchodu",
    ],
    "chutiya": [
        "chootiya", "chutia", "chutiye", "chutiyaa",
        "chootia", "chutiyo", "chutiyee", "chutiyapa",
        "ch00tiya", "chut1ya",
    ],
    "gaandu": [
        "gandu", "gaand", "gaandd", "gand", "gaanduu",
        "g4ndu", "ganduu",
    ],
    "randi": [
        "randii", "rundi", "randiya", "randiyo",
        "r4ndi", "randwe",
    ],
    "harami": [
        "haraami", "haramii", "haram1", "haramkhor",
        "haraamii",
    ],
    "kutte": [
        "kuttee", "kutta", "kuttte", "kuttey",
        "kutt3", "kutiya",
    ],
    "sala": [
        "saala", "sale", "saaale", "saale",
        "s4la", "s4le",
    ],
    "bhosdike": [
        "bsdk", "bhosdiwale", "bhosdika", "bhosdki",
        "bh0sdike", "bhosdi",
    ],
    "lodu": [
        "laude", "laudu", "lauda", "l0du",
        "lavde", "lawde", "laudey",
    ],

    # --- Threat-related ---
    "maar dunga": [
        "maar daaluga", "maar dalunga", "maarunga",
        "maar deta", "maar khayega",
    ],

    # --- Insult variants ---
    "pagal": [
        "paagal", "pagall", "p4gal", "pagl",
    ],
    "bewakoof": [
        "bevkoof", "bewkoof", "bewaqoof", "bevakoof",
        "b3wakoof",
    ],
    "gadha": [
        "gadhe", "gadhaa", "g4dha",
    ],
}

# Build reverse lookup: variant → canonical
_VARIANT_TO_CANONICAL: Dict[str, str] = {}
for canonical, variants in HINGLISH_VARIANTS.items():
    for variant in variants:
        _VARIANT_TO_CANONICAL[variant] = canonical

# Sort by length (longest first) to match multi-word variants first
_SORTED_VARIANTS = sorted(_VARIANT_TO_CANONICAL.keys(), key=len, reverse=True)

# Build a regex pattern for word-boundary matching
_VARIANT_PATTERN = re.compile(
    r"\b(" + "|".join(re.escape(v) for v in _SORTED_VARIANTS) + r")\b",
    re.IGNORECASE,
)


def normalize_hinglish_slurs(text: str) -> str:
    """
    Replace variant spellings of Hinglish profanity/slurs with
    their canonical forms to reduce surface variation for the model.

    Args:
        text: Lowercased input text.

    Returns:
        Text with normalized slur spellings.
    """
    def _replace(match):
        variant = match.group(0).lower()
        return _VARIANT_TO_CANONICAL.get(variant, variant)

    return _VARIANT_PATTERN.sub(_replace, text)


# ---------------------------------------------------------------------------
# Obfuscation pattern detection
# ---------------------------------------------------------------------------

# Common obfuscation: inserting dots, spaces, or special chars within words
# e.g., "f.u.c.k" or "f u c k" or "f*ck"
RE_DOTTED_WORD = re.compile(r"\b(\w)(?:[.\-_*#](\w)){2,}\b")


def normalize_obfuscation(text: str) -> str:
    """
    Remove common obfuscation patterns like dots/stars between chars.

    Examples:
        f.u.c.k → fuck
        s.h.i.t → shit
    """
    # Remove single-char separators between word characters
    text = re.sub(r"(?<=\w)[.\-_*#](?=\w)", "", text)
    return text


# ---------------------------------------------------------------------------
# Main normalization function
# ---------------------------------------------------------------------------

def normalize_hinglish(text: Optional[str]) -> str:
    """
    Full Hinglish normalization pipeline.

    1. Decode leetspeak
    2. Remove obfuscation patterns
    3. Normalize slur variants to canonical forms

    Args:
        text: Input text (should already be lowercased).

    Returns:
        Normalized text.
    """
    if not text or not isinstance(text, str):
        return ""

    text = text.lower()
    text = decode_leetspeak(text)
    text = normalize_obfuscation(text)
    text = normalize_hinglish_slurs(text)

    return text


# ---------------------------------------------------------------------------
# CLI
# ---------------------------------------------------------------------------

def main():
    """Quick demo of normalization."""
    test_cases = [
        "ch00t1y4",
        "Tu bsdk pagal hai",
        "m.c. sale kutte",
        "Bhai app slow hai",
        "f.u.c.k you",
        "bhenchoot tujhe maar dalunga",
        "Website bahut acha hai",
        "Tu bewkoof hai kya",
    ]

    print("=" * 60)
    print("Hinglish Normalization Demo")
    print("=" * 60)
    for text in test_cases:
        normalized = normalize_hinglish(text)
        print(f"  {text:40s} → {normalized}")
    print("=" * 60)


if __name__ == "__main__":
    main()