File size: 10,178 Bytes
e57521c
 
 
 
670d9e9
e57521c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
670d9e9
e57521c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
670d9e9
 
 
 
 
 
 
 
 
 
 
 
e57521c
 
 
 
 
 
 
 
 
 
 
670d9e9
 
 
e57521c
1ea1ef2
 
e57521c
 
670d9e9
e57521c
670d9e9
1ea1ef2
670d9e9
 
 
 
 
1ea1ef2
e57521c
 
1ea1ef2
e57521c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
670d9e9
e57521c
 
 
 
670d9e9
e57521c
 
 
 
 
 
 
 
 
 
 
670d9e9
 
 
 
 
 
 
 
 
 
e57521c
 
670d9e9
e57521c
 
670d9e9
e57521c
 
 
 
670d9e9
e57521c
 
670d9e9
e57521c
 
670d9e9
 
 
 
e57521c
 
 
 
 
 
 
 
 
 
 
 
 
 
670d9e9
e57521c
 
 
 
 
 
670d9e9
e57521c
 
 
 
 
 
 
670d9e9
e57521c
 
670d9e9
e57521c
 
 
 
670d9e9
 
 
 
 
e57521c
 
 
 
 
 
 
670d9e9
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
import re
import difflib
import torch
from transformers import MarianTokenizer, MarianMTModel
from spellchecker import SpellChecker

DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")

# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# TEXT CLEANING
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
_DIACRITICS_RE = re.compile(
    r"[\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]"
)

def clean_text(text: str) -> str:
    if not isinstance(text, str): return ""
    text = re.sub(r"(?:https?://|www\.)\S+", "", text)
    text = re.sub(r"\S+@\S+", "", text)
    text = re.sub(r"@\w+", "", text)
    text = re.sub(r"#", "", text)
    text = re.sub(r"\n+", " ", text)
    text = re.sub(r"[ุฅุฃุขุง]", "ุง", text)
    text = re.sub(r"ู‰", "ูŠ", text)
    text = re.sub(r"[ุคุฆ]", "ุก", text)
    text = re.sub(r"ฺฏ", "ูƒ", text)
    text = _DIACRITICS_RE.sub("", text)
    text = re.sub(r"(.)\1{3,}", r"\1\1", text)
    text = re.sub(r"[^\u0600-\u06FFa-zA-Z0-9\s\.\,\!\?\;\:\"\'\(\)\-]", "", text)
    return re.sub(r"\s+", " ", text).strip()

# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# LEXICONS
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
FRANCO_LEXICON = {
    "el", "al", "mn", "min", "w", "f", "b", "l", "m", "lw", "law",
    "aw", "wa", "fe", "fi", "fy", "lel", "lil", "fil", "fel", "bil", "bel",
    "3al", "3ala", "3shan", "3alshan", "wala", "wl2",
    "ana", "enta", "enti", "anty", "ehna", "entu", "hwa", "hya", "homa",
    "3ndi", "3ndk", "3ndh", "3ndhom",
    "msh", "mesh", "ma", "mafish", "mafesh", "mb", "mis", "mish",
    "mabyen", "mabyensh", "msh3awz", "msh3ayza",
    "dh", "deh", "di", "dol", "dool", "da", "dah",
    "awy", "bgd", "bejd", "keda", "kda", "lessa", "lsa", "kman", "kamaan",
    "gdan", "gdn", "yani", "ya3ni", "ya3ny", "ahlan", "taiban", "taban",
    "2awi", "2awy",
    "fein", "fen", "meen", "leih", "leh", "emta", "imta",
    "izzay", "ezzay", "ezay", "eih", "ayh", "eah",
    "yalla", "yala", "tab", "aho", "bs", "bas", "laken", "malesh",
    "wallah", "wlhy", "wlah", "sa7", "tamam", "tmam", "sfr",
    "tba2", "yb2a", "tab2a", "khalas", "kls", "zain", "wlahy",
    "haga", "hagat", "7aga", "7agat", "lazm", "lazmak", "lazmik", "lazem",
    "momken", "mumken", "ymkn", "wahid", "wa7ed", "wa7id", "ba3d", "b3d",
    "zift", "zft", "3ayz", "3ayza", "3ayez", "kwaysa", "kwayes", "kwys",
    "wahsh", "wahsha", "whs", "helw", "helwa", "gamed", "gamd", "gamda",
    "ta3ban", "ta3bana", "za3lan", "za3lana", "mabsut", "mabsout", "zay", "zai",
    "bshof", "bshuf", "bshouf", "bafdl", "bafadl", "ba3mel", "ba3ml",
    "bybin", "byen", "roh", "geh", "gai", "gebt", "ray7", "rai7",
    "rg3", "raga3", "rafe3", "tawer", "mshakel", "mushkela", "mwdu3", "mawdu3",
    "khidma", "khidme", "talb", "talab", "balgh", "blag", "khsar", "khasar",
    "shahn", "kan", "kanet", "knt", "byt", "byt3", "byt3ml",
    "ht", "han", "hyt", "mls", "mls3", "msh3",
    "kol", "kul", "gwa", "bra", "wl", "bl", "ml", "fl", "tl", "sl", "hl", "ql", "dl", "rl", "nl",
}

_ENGLISH_STOPWORDS = {
    "this", "is", "are", "am", "was", "were", "it", "in", "on", "at",
    "to", "for", "out", "of", "and", "the", "i", "a", "an", "my", "your",
    "we", "he", "she", "they", "us", "me", "him", "her", "them",
    "have", "has", "had", "do", "does", "did", "will", "would", "can",
    "could", "should", "not", "no", "yes", "but", "or", "so", "if",
    "then", "that", "which", "who", "what", "how", "when", "where", "why",
}

_FRANCO_START_DIGRAPHS = ("sh", "kh", "gh", "wl", "mn", "bn", "tn", "rh", "zh", "dh")
_FRANCO_DIGITS         = set("23456789")
_VOWELS                = set("aeiouAEIOU")

# Fuzzy targets restricted entirely to Arabizi
_FUZZY_TARGETS = frozenset(FRANCO_LEXICON)

def classify_word(word: str) -> str:
    if re.search(r"[\u0600-\u06FF]", word): return "arabic"
    core = re.sub(r"[^\w]", "", word).lower()
    if not core: return "arabic"
    
    if core in _ENGLISH_STOPWORDS: return "english"
    if any(c in core for c in _FRANCO_DIGITS): return "franco"
    if any(core.startswith(d) for d in _FRANCO_START_DIGRAPHS): return "franco"
    if core in FRANCO_LEXICON: return "franco"
    if len(core) <= 4 and sum(c in _VOWELS for c in core) <= 1: return "franco"
    
    if difflib.get_close_matches(core, _FUZZY_TARGETS, n=1, cutoff=0.80):
        return "franco"
        
    return "english"

_FRANCO_MAP = {
    "2": "ุก", "3": "ุน", "4": "ุด", "5": "ุฎ",
    "6": "ุท", "7": "ุญ", "8": "ู‚", "9": "ุต",
    "a": "ุง", "b": "ุจ", "c": "ูƒ", "d": "ุฏ",
    "e": "ูŠ", "f": "ู", "g": "ุฌ", "h": "ู‡",
    "i": "ูŠ", "j": "ุฌ", "k": "ูƒ", "l": "ู„",
    "m": "ู…", "n": "ู†", "o": "ูˆ", "p": "ุจ",
    "q": "ู‚", "r": "ุฑ", "s": "ุณ", "t": "ุช",
    "u": "ูˆ", "v": "ู", "w": "ูˆ", "x": "ุงูƒุณ",
    "y": "ูŠ", "z": "ุฒ",
}

def transliterate_franco(text: str) -> str:
    t = text.lower()
    t = t.replace("sh", "ุด").replace("ch", "ุชุด").replace("kh", "ุฎ")
    t = t.replace("th", "ุซ").replace("gh", "ุบ").replace("ph", "ู")
    t = t.replace("ou", "ูˆ").replace("ee", "ูŠ").replace("oo", "ูˆ")
    return "".join(_FRANCO_MAP.get(c, c) for c in t)

# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# SPELLCHECKER & TRANSLATOR
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
_EGY_MODEL    = "NAMAA-Space/masrawy-english-to-egyptian-arabic-translator-v2.9"
_egy_tokenizer = None
_egy_model     = None
en_spell       = SpellChecker(language='en')

def _load_namaa():
    global _egy_tokenizer, _egy_model
    if _egy_model is None:
        print("Loading NAMAA ENโ†’EGY translatorโ€ฆ")
        _egy_tokenizer = MarianTokenizer.from_pretrained(_EGY_MODEL)
        _egy_model     = MarianMTModel.from_pretrained(_EGY_MODEL).to(DEVICE)
        _egy_model.eval()

def translate_english_span(chunk: str) -> str:
    _load_namaa()
    
    # 1. Spellcheck English to prevent NAMAA failure on typos
    words = chunk.split()
    corrected_words = []
    for w in words:
        fixed = en_spell.correction(w)
        corrected_words.append(fixed if fixed else w)
    cleaned_chunk = " ".join(corrected_words)

    # 2. Safely translate or fallback to untouched Latin
    try:
        tokens = _egy_tokenizer(
            [cleaned_chunk], return_tensors="pt", padding=True, 
            truncation=True, max_length=64
        ).to(DEVICE)
        
        with torch.no_grad():
            out = _egy_model.generate(**tokens, num_beams=4, max_new_tokens=64)
        return _egy_tokenizer.decode(out[0], skip_special_tokens=True)
    except Exception:
        return cleaned_chunk

# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# CORE ROUTING PIPELINE
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
def _process_core(raw: str, mode: str) -> tuple[str, str]:
    # Strip tatweel/kashida FIRST so the prefix regex triggers correctly
    raw = raw.replace("ู€", "")
    
    # Detach Arabic prefix morphemes glued to Latin roots
    raw = re.sub(
        r"(ุงู„|ุจ|ุจูŠ|ู‡|ู|ูƒ|ู„|ู„ู„|ู…ุง|ู…ุด)([a-zA-Z0-9]+)",
        r"\1 \2", raw
    )

    if not re.search(r"[a-zA-Z0-9]", raw):
        return clean_text(raw), "Pure Arabic"

    words = raw.split()
    if not words:
        return "", "Pure Arabic"

    classified = [(w, classify_word(w)) for w in words]
    spans: list[tuple[str, list[str]]] = []
    
    for word, lang in classified:
        if spans and spans[-1][0] == lang:
            spans[-1][1].append(word)
        else:
            spans.append((lang, [word]))

    output_parts, langs_used = [], set()
    for lang, span_words in spans:
        chunk = " ".join(span_words)
        if lang == "arabic":
            output_parts.append(chunk)
        elif lang == "franco":
            output_parts.append(transliterate_franco(chunk))
            langs_used.add("franco")
        else:
            if mode == "full":
                output_parts.append(translate_english_span(chunk))
            else:
                output_parts.append(transliterate_franco(chunk))
            langs_used.add("english")

    cleaned = clean_text(" ".join(output_parts))
    
    if langs_used == {"franco", "english"}: route_tag = "Mixed Franco+English โ†’ Arabic (Deep Fusion)"
    elif "franco" in langs_used: route_tag = "Franco Transliterated โ†’ Arabic (Deep Fusion)"
    elif "english" in langs_used: route_tag = "English Translated โ†’ Arabic (Deep Fusion)"
    else: route_tag = "Pure Arabic (Deep Fusion)"

    return cleaned, route_tag

def preprocess(raw: str) -> tuple[str, str]:
    return _process_core(raw, mode="full")

def clean_text_v5(text: str, mode: str = "transliterate_only") -> str:
    if not isinstance(text, str): return ""
    return _process_core(text, mode=mode)[0]