import re import difflib import torch from transformers import MarianTokenizer, MarianMTModel from spellchecker import SpellChecker DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu") # ───────────────────────────────────────────────────────────────────────────── # TEXT CLEANING # ───────────────────────────────────────────────────────────────────────────── _DIACRITICS_RE = re.compile( r"[\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]" ) def clean_text(text: str) -> str: if not isinstance(text, str): return "" text = re.sub(r"(?:https?://|www\.)\S+", "", text) text = re.sub(r"\S+@\S+", "", text) text = re.sub(r"@\w+", "", text) text = re.sub(r"#", "", text) text = re.sub(r"\n+", " ", text) text = re.sub(r"[إأآا]", "ا", text) text = re.sub(r"ى", "ي", text) text = re.sub(r"[ؤئ]", "ء", text) text = re.sub(r"گ", "ك", text) text = _DIACRITICS_RE.sub("", text) text = re.sub(r"(.)\1{3,}", r"\1\1", text) text = re.sub(r"[^\u0600-\u06FFa-zA-Z0-9\s\.\,\!\?\;\:\"\'\(\)\-]", "", text) return re.sub(r"\s+", " ", text).strip() # ───────────────────────────────────────────────────────────────────────────── # LEXICONS # ───────────────────────────────────────────────────────────────────────────── FRANCO_LEXICON = { "el", "al", "mn", "min", "w", "f", "b", "l", "m", "lw", "law", "aw", "wa", "fe", "fi", "fy", "lel", "lil", "fil", "fel", "bil", "bel", "3al", "3ala", "3shan", "3alshan", "wala", "wl2", "ana", "enta", "enti", "anty", "ehna", "entu", "hwa", "hya", "homa", "3ndi", "3ndk", "3ndh", "3ndhom", "msh", "mesh", "ma", "mafish", "mafesh", "mb", "mis", "mish", "mabyen", "mabyensh", "msh3awz", "msh3ayza", "dh", "deh", "di", "dol", "dool", "da", "dah", "awy", "bgd", "bejd", "keda", "kda", "lessa", "lsa", "kman", "kamaan", "gdan", "gdn", "yani", "ya3ni", "ya3ny", "ahlan", "taiban", "taban", "2awi", "2awy", "fein", "fen", "meen", "leih", "leh", "emta", "imta", "izzay", "ezzay", "ezay", "eih", "ayh", "eah", "yalla", "yala", "tab", "aho", "bs", "bas", "laken", "malesh", "wallah", "wlhy", "wlah", "sa7", "tamam", "tmam", "sfr", "tba2", "yb2a", "tab2a", "khalas", "kls", "zain", "wlahy", "haga", "hagat", "7aga", "7agat", "lazm", "lazmak", "lazmik", "lazem", "momken", "mumken", "ymkn", "wahid", "wa7ed", "wa7id", "ba3d", "b3d", "zift", "zft", "3ayz", "3ayza", "3ayez", "kwaysa", "kwayes", "kwys", "wahsh", "wahsha", "whs", "helw", "helwa", "gamed", "gamd", "gamda", "ta3ban", "ta3bana", "za3lan", "za3lana", "mabsut", "mabsout", "zay", "zai", "bshof", "bshuf", "bshouf", "bafdl", "bafadl", "ba3mel", "ba3ml", "bybin", "byen", "roh", "geh", "gai", "gebt", "ray7", "rai7", "rg3", "raga3", "rafe3", "tawer", "mshakel", "mushkela", "mwdu3", "mawdu3", "khidma", "khidme", "talb", "talab", "balgh", "blag", "khsar", "khasar", "shahn", "kan", "kanet", "knt", "byt", "byt3", "byt3ml", "ht", "han", "hyt", "mls", "mls3", "msh3", "kol", "kul", "gwa", "bra", "wl", "bl", "ml", "fl", "tl", "sl", "hl", "ql", "dl", "rl", "nl", } _ENGLISH_STOPWORDS = { "this", "is", "are", "am", "was", "were", "it", "in", "on", "at", "to", "for", "out", "of", "and", "the", "i", "a", "an", "my", "your", "we", "he", "she", "they", "us", "me", "him", "her", "them", "have", "has", "had", "do", "does", "did", "will", "would", "can", "could", "should", "not", "no", "yes", "but", "or", "so", "if", "then", "that", "which", "who", "what", "how", "when", "where", "why", } _FRANCO_START_DIGRAPHS = ("sh", "kh", "gh", "wl", "mn", "bn", "tn", "rh", "zh", "dh") _FRANCO_DIGITS = set("23456789") _VOWELS = set("aeiouAEIOU") # Fuzzy targets restricted entirely to Arabizi _FUZZY_TARGETS = frozenset(FRANCO_LEXICON) def classify_word(word: str) -> str: if re.search(r"[\u0600-\u06FF]", word): return "arabic" core = re.sub(r"[^\w]", "", word).lower() if not core: return "arabic" if core in _ENGLISH_STOPWORDS: return "english" if any(c in core for c in _FRANCO_DIGITS): return "franco" if any(core.startswith(d) for d in _FRANCO_START_DIGRAPHS): return "franco" if core in FRANCO_LEXICON: return "franco" if len(core) <= 4 and sum(c in _VOWELS for c in core) <= 1: return "franco" if difflib.get_close_matches(core, _FUZZY_TARGETS, n=1, cutoff=0.80): return "franco" return "english" _FRANCO_MAP = { "2": "ء", "3": "ع", "4": "ش", "5": "خ", "6": "ط", "7": "ح", "8": "ق", "9": "ص", "a": "ا", "b": "ب", "c": "ك", "d": "د", "e": "ي", "f": "ف", "g": "ج", "h": "ه", "i": "ي", "j": "ج", "k": "ك", "l": "ل", "m": "م", "n": "ن", "o": "و", "p": "ب", "q": "ق", "r": "ر", "s": "س", "t": "ت", "u": "و", "v": "ف", "w": "و", "x": "اكس", "y": "ي", "z": "ز", } def transliterate_franco(text: str) -> str: t = text.lower() t = t.replace("sh", "ش").replace("ch", "تش").replace("kh", "خ") t = t.replace("th", "ث").replace("gh", "غ").replace("ph", "ف") t = t.replace("ou", "و").replace("ee", "ي").replace("oo", "و") return "".join(_FRANCO_MAP.get(c, c) for c in t) # ───────────────────────────────────────────────────────────────────────────── # SPELLCHECKER & TRANSLATOR # ───────────────────────────────────────────────────────────────────────────── _EGY_MODEL = "NAMAA-Space/masrawy-english-to-egyptian-arabic-translator-v2.9" _egy_tokenizer = None _egy_model = None en_spell = SpellChecker(language='en') def _load_namaa(): global _egy_tokenizer, _egy_model if _egy_model is None: print("Loading NAMAA EN→EGY translator…") _egy_tokenizer = MarianTokenizer.from_pretrained(_EGY_MODEL) _egy_model = MarianMTModel.from_pretrained(_EGY_MODEL).to(DEVICE) _egy_model.eval() def translate_english_span(chunk: str) -> str: _load_namaa() # 1. Spellcheck English to prevent NAMAA failure on typos words = chunk.split() corrected_words = [] for w in words: fixed = en_spell.correction(w) corrected_words.append(fixed if fixed else w) cleaned_chunk = " ".join(corrected_words) # 2. Safely translate or fallback to untouched Latin try: tokens = _egy_tokenizer( [cleaned_chunk], return_tensors="pt", padding=True, truncation=True, max_length=64 ).to(DEVICE) with torch.no_grad(): out = _egy_model.generate(**tokens, num_beams=4, max_new_tokens=64) return _egy_tokenizer.decode(out[0], skip_special_tokens=True) except Exception: return cleaned_chunk # ───────────────────────────────────────────────────────────────────────────── # CORE ROUTING PIPELINE # ───────────────────────────────────────────────────────────────────────────── def _process_core(raw: str, mode: str) -> tuple[str, str]: # Strip tatweel/kashida FIRST so the prefix regex triggers correctly raw = raw.replace("ـ", "") # Detach Arabic prefix morphemes glued to Latin roots raw = re.sub( r"(ال|ب|بي|ه|ف|ك|ل|لل|ما|مش)([a-zA-Z0-9]+)", r"\1 \2", raw ) if not re.search(r"[a-zA-Z0-9]", raw): return clean_text(raw), "Pure Arabic" words = raw.split() if not words: return "", "Pure Arabic" classified = [(w, classify_word(w)) for w in words] spans: list[tuple[str, list[str]]] = [] for word, lang in classified: if spans and spans[-1][0] == lang: spans[-1][1].append(word) else: spans.append((lang, [word])) output_parts, langs_used = [], set() for lang, span_words in spans: chunk = " ".join(span_words) if lang == "arabic": output_parts.append(chunk) elif lang == "franco": output_parts.append(transliterate_franco(chunk)) langs_used.add("franco") else: if mode == "full": output_parts.append(translate_english_span(chunk)) else: output_parts.append(transliterate_franco(chunk)) langs_used.add("english") cleaned = clean_text(" ".join(output_parts)) if langs_used == {"franco", "english"}: route_tag = "Mixed Franco+English → Arabic (Deep Fusion)" elif "franco" in langs_used: route_tag = "Franco Transliterated → Arabic (Deep Fusion)" elif "english" in langs_used: route_tag = "English Translated → Arabic (Deep Fusion)" else: route_tag = "Pure Arabic (Deep Fusion)" return cleaned, route_tag def preprocess(raw: str) -> tuple[str, str]: return _process_core(raw, mode="full") def clean_text_v5(text: str, mode: str = "transliterate_only") -> str: if not isinstance(text, str): return "" return _process_core(text, mode=mode)[0]