Spaces:
Running
Running
Download text_normalize.py from T0KII/taMASRIBERTs: direct link, hf CLI and curl.
- Browser
- Download file 10.2 kB
-
https://huggingface.co/spaces/T0KII/taMASRIBERTs/resolve/main/text_normalize.py
- Command line
-
hf download hf://spaces/T0KII/taMASRIBERTs/text_normalize.py
-
curl -L -o text_normalize.py https://huggingface.co/spaces/T0KII/taMASRIBERTs/resolve/main/text_normalize.py
10.2 kB
| import re | |
| import difflib | |
| import torch | |
| from transformers import MarianTokenizer, MarianMTModel | |
| from spellchecker import SpellChecker | |
| DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu") | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| # TEXT CLEANING | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| _DIACRITICS_RE = re.compile( | |
| r"[\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]" | |
| ) | |
| def clean_text(text: str) -> str: | |
| if not isinstance(text, str): return "" | |
| text = re.sub(r"(?:https?://|www\.)\S+", "", text) | |
| text = re.sub(r"\S+@\S+", "", text) | |
| text = re.sub(r"@\w+", "", text) | |
| text = re.sub(r"#", "", text) | |
| text = re.sub(r"\n+", " ", text) | |
| text = re.sub(r"[ุฅุฃุขุง]", "ุง", text) | |
| text = re.sub(r"ู", "ู", text) | |
| text = re.sub(r"[ุคุฆ]", "ุก", text) | |
| text = re.sub(r"ฺฏ", "ู", text) | |
| text = _DIACRITICS_RE.sub("", text) | |
| text = re.sub(r"(.)\1{3,}", r"\1\1", text) | |
| text = re.sub(r"[^\u0600-\u06FFa-zA-Z0-9\s\.\,\!\?\;\:\"\'\(\)\-]", "", text) | |
| return re.sub(r"\s+", " ", text).strip() | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| # LEXICONS | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| FRANCO_LEXICON = { | |
| "el", "al", "mn", "min", "w", "f", "b", "l", "m", "lw", "law", | |
| "aw", "wa", "fe", "fi", "fy", "lel", "lil", "fil", "fel", "bil", "bel", | |
| "3al", "3ala", "3shan", "3alshan", "wala", "wl2", | |
| "ana", "enta", "enti", "anty", "ehna", "entu", "hwa", "hya", "homa", | |
| "3ndi", "3ndk", "3ndh", "3ndhom", | |
| "msh", "mesh", "ma", "mafish", "mafesh", "mb", "mis", "mish", | |
| "mabyen", "mabyensh", "msh3awz", "msh3ayza", | |
| "dh", "deh", "di", "dol", "dool", "da", "dah", | |
| "awy", "bgd", "bejd", "keda", "kda", "lessa", "lsa", "kman", "kamaan", | |
| "gdan", "gdn", "yani", "ya3ni", "ya3ny", "ahlan", "taiban", "taban", | |
| "2awi", "2awy", | |
| "fein", "fen", "meen", "leih", "leh", "emta", "imta", | |
| "izzay", "ezzay", "ezay", "eih", "ayh", "eah", | |
| "yalla", "yala", "tab", "aho", "bs", "bas", "laken", "malesh", | |
| "wallah", "wlhy", "wlah", "sa7", "tamam", "tmam", "sfr", | |
| "tba2", "yb2a", "tab2a", "khalas", "kls", "zain", "wlahy", | |
| "haga", "hagat", "7aga", "7agat", "lazm", "lazmak", "lazmik", "lazem", | |
| "momken", "mumken", "ymkn", "wahid", "wa7ed", "wa7id", "ba3d", "b3d", | |
| "zift", "zft", "3ayz", "3ayza", "3ayez", "kwaysa", "kwayes", "kwys", | |
| "wahsh", "wahsha", "whs", "helw", "helwa", "gamed", "gamd", "gamda", | |
| "ta3ban", "ta3bana", "za3lan", "za3lana", "mabsut", "mabsout", "zay", "zai", | |
| "bshof", "bshuf", "bshouf", "bafdl", "bafadl", "ba3mel", "ba3ml", | |
| "bybin", "byen", "roh", "geh", "gai", "gebt", "ray7", "rai7", | |
| "rg3", "raga3", "rafe3", "tawer", "mshakel", "mushkela", "mwdu3", "mawdu3", | |
| "khidma", "khidme", "talb", "talab", "balgh", "blag", "khsar", "khasar", | |
| "shahn", "kan", "kanet", "knt", "byt", "byt3", "byt3ml", | |
| "ht", "han", "hyt", "mls", "mls3", "msh3", | |
| "kol", "kul", "gwa", "bra", "wl", "bl", "ml", "fl", "tl", "sl", "hl", "ql", "dl", "rl", "nl", | |
| } | |
| _ENGLISH_STOPWORDS = { | |
| "this", "is", "are", "am", "was", "were", "it", "in", "on", "at", | |
| "to", "for", "out", "of", "and", "the", "i", "a", "an", "my", "your", | |
| "we", "he", "she", "they", "us", "me", "him", "her", "them", | |
| "have", "has", "had", "do", "does", "did", "will", "would", "can", | |
| "could", "should", "not", "no", "yes", "but", "or", "so", "if", | |
| "then", "that", "which", "who", "what", "how", "when", "where", "why", | |
| } | |
| _FRANCO_START_DIGRAPHS = ("sh", "kh", "gh", "wl", "mn", "bn", "tn", "rh", "zh", "dh") | |
| _FRANCO_DIGITS = set("23456789") | |
| _VOWELS = set("aeiouAEIOU") | |
| # Fuzzy targets restricted entirely to Arabizi | |
| _FUZZY_TARGETS = frozenset(FRANCO_LEXICON) | |
| def classify_word(word: str) -> str: | |
| if re.search(r"[\u0600-\u06FF]", word): return "arabic" | |
| core = re.sub(r"[^\w]", "", word).lower() | |
| if not core: return "arabic" | |
| if core in _ENGLISH_STOPWORDS: return "english" | |
| if any(c in core for c in _FRANCO_DIGITS): return "franco" | |
| if any(core.startswith(d) for d in _FRANCO_START_DIGRAPHS): return "franco" | |
| if core in FRANCO_LEXICON: return "franco" | |
| if len(core) <= 4 and sum(c in _VOWELS for c in core) <= 1: return "franco" | |
| if difflib.get_close_matches(core, _FUZZY_TARGETS, n=1, cutoff=0.80): | |
| return "franco" | |
| return "english" | |
| _FRANCO_MAP = { | |
| "2": "ุก", "3": "ุน", "4": "ุด", "5": "ุฎ", | |
| "6": "ุท", "7": "ุญ", "8": "ู", "9": "ุต", | |
| "a": "ุง", "b": "ุจ", "c": "ู", "d": "ุฏ", | |
| "e": "ู", "f": "ู", "g": "ุฌ", "h": "ู", | |
| "i": "ู", "j": "ุฌ", "k": "ู", "l": "ู", | |
| "m": "ู ", "n": "ู", "o": "ู", "p": "ุจ", | |
| "q": "ู", "r": "ุฑ", "s": "ุณ", "t": "ุช", | |
| "u": "ู", "v": "ู", "w": "ู", "x": "ุงูุณ", | |
| "y": "ู", "z": "ุฒ", | |
| } | |
| def transliterate_franco(text: str) -> str: | |
| t = text.lower() | |
| t = t.replace("sh", "ุด").replace("ch", "ุชุด").replace("kh", "ุฎ") | |
| t = t.replace("th", "ุซ").replace("gh", "ุบ").replace("ph", "ู") | |
| t = t.replace("ou", "ู").replace("ee", "ู").replace("oo", "ู") | |
| return "".join(_FRANCO_MAP.get(c, c) for c in t) | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| # SPELLCHECKER & TRANSLATOR | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| _EGY_MODEL = "NAMAA-Space/masrawy-english-to-egyptian-arabic-translator-v2.9" | |
| _egy_tokenizer = None | |
| _egy_model = None | |
| en_spell = SpellChecker(language='en') | |
| def _load_namaa(): | |
| global _egy_tokenizer, _egy_model | |
| if _egy_model is None: | |
| print("Loading NAMAA ENโEGY translatorโฆ") | |
| _egy_tokenizer = MarianTokenizer.from_pretrained(_EGY_MODEL) | |
| _egy_model = MarianMTModel.from_pretrained(_EGY_MODEL).to(DEVICE) | |
| _egy_model.eval() | |
| def translate_english_span(chunk: str) -> str: | |
| _load_namaa() | |
| # 1. Spellcheck English to prevent NAMAA failure on typos | |
| words = chunk.split() | |
| corrected_words = [] | |
| for w in words: | |
| fixed = en_spell.correction(w) | |
| corrected_words.append(fixed if fixed else w) | |
| cleaned_chunk = " ".join(corrected_words) | |
| # 2. Safely translate or fallback to untouched Latin | |
| try: | |
| tokens = _egy_tokenizer( | |
| [cleaned_chunk], return_tensors="pt", padding=True, | |
| truncation=True, max_length=64 | |
| ).to(DEVICE) | |
| with torch.no_grad(): | |
| out = _egy_model.generate(**tokens, num_beams=4, max_new_tokens=64) | |
| return _egy_tokenizer.decode(out[0], skip_special_tokens=True) | |
| except Exception: | |
| return cleaned_chunk | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| # CORE ROUTING PIPELINE | |
| # โโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโโ | |
| def _process_core(raw: str, mode: str) -> tuple[str, str]: | |
| # Strip tatweel/kashida FIRST so the prefix regex triggers correctly | |
| raw = raw.replace("ู", "") | |
| # Detach Arabic prefix morphemes glued to Latin roots | |
| raw = re.sub( | |
| r"(ุงู|ุจ|ุจู|ู|ู|ู|ู|ูู|ู ุง|ู ุด)([a-zA-Z0-9]+)", | |
| r"\1 \2", raw | |
| ) | |
| if not re.search(r"[a-zA-Z0-9]", raw): | |
| return clean_text(raw), "Pure Arabic" | |
| words = raw.split() | |
| if not words: | |
| return "", "Pure Arabic" | |
| classified = [(w, classify_word(w)) for w in words] | |
| spans: list[tuple[str, list[str]]] = [] | |
| for word, lang in classified: | |
| if spans and spans[-1][0] == lang: | |
| spans[-1][1].append(word) | |
| else: | |
| spans.append((lang, [word])) | |
| output_parts, langs_used = [], set() | |
| for lang, span_words in spans: | |
| chunk = " ".join(span_words) | |
| if lang == "arabic": | |
| output_parts.append(chunk) | |
| elif lang == "franco": | |
| output_parts.append(transliterate_franco(chunk)) | |
| langs_used.add("franco") | |
| else: | |
| if mode == "full": | |
| output_parts.append(translate_english_span(chunk)) | |
| else: | |
| output_parts.append(transliterate_franco(chunk)) | |
| langs_used.add("english") | |
| cleaned = clean_text(" ".join(output_parts)) | |
| if langs_used == {"franco", "english"}: route_tag = "Mixed Franco+English โ Arabic (Deep Fusion)" | |
| elif "franco" in langs_used: route_tag = "Franco Transliterated โ Arabic (Deep Fusion)" | |
| elif "english" in langs_used: route_tag = "English Translated โ Arabic (Deep Fusion)" | |
| else: route_tag = "Pure Arabic (Deep Fusion)" | |
| return cleaned, route_tag | |
| def preprocess(raw: str) -> tuple[str, str]: | |
| return _process_core(raw, mode="full") | |
| def clean_text_v5(text: str, mode: str = "transliterate_only") -> str: | |
| if not isinstance(text, str): return "" | |
| return _process_core(text, mode=mode)[0] |