taMASRIBERTs / text_normalize.py
T0KII's picture
Update text_normalize.py
1ea1ef2 verified
Raw History Blame Contribute Delete
10.2 kB
import re
import difflib
import torch
from transformers import MarianTokenizer, MarianMTModel
from spellchecker import SpellChecker
DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# TEXT CLEANING
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
_DIACRITICS_RE = re.compile(
r"[\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]"
)
def clean_text(text: str) -> str:
if not isinstance(text, str): return ""
text = re.sub(r"(?:https?://|www\.)\S+", "", text)
text = re.sub(r"\S+@\S+", "", text)
text = re.sub(r"@\w+", "", text)
text = re.sub(r"#", "", text)
text = re.sub(r"\n+", " ", text)
text = re.sub(r"[ุฅุฃุขุง]", "ุง", text)
text = re.sub(r"ู‰", "ูŠ", text)
text = re.sub(r"[ุคุฆ]", "ุก", text)
text = re.sub(r"ฺฏ", "ูƒ", text)
text = _DIACRITICS_RE.sub("", text)
text = re.sub(r"(.)\1{3,}", r"\1\1", text)
text = re.sub(r"[^\u0600-\u06FFa-zA-Z0-9\s\.\,\!\?\;\:\"\'\(\)\-]", "", text)
return re.sub(r"\s+", " ", text).strip()
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# LEXICONS
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
FRANCO_LEXICON = {
"el", "al", "mn", "min", "w", "f", "b", "l", "m", "lw", "law",
"aw", "wa", "fe", "fi", "fy", "lel", "lil", "fil", "fel", "bil", "bel",
"3al", "3ala", "3shan", "3alshan", "wala", "wl2",
"ana", "enta", "enti", "anty", "ehna", "entu", "hwa", "hya", "homa",
"3ndi", "3ndk", "3ndh", "3ndhom",
"msh", "mesh", "ma", "mafish", "mafesh", "mb", "mis", "mish",
"mabyen", "mabyensh", "msh3awz", "msh3ayza",
"dh", "deh", "di", "dol", "dool", "da", "dah",
"awy", "bgd", "bejd", "keda", "kda", "lessa", "lsa", "kman", "kamaan",
"gdan", "gdn", "yani", "ya3ni", "ya3ny", "ahlan", "taiban", "taban",
"2awi", "2awy",
"fein", "fen", "meen", "leih", "leh", "emta", "imta",
"izzay", "ezzay", "ezay", "eih", "ayh", "eah",
"yalla", "yala", "tab", "aho", "bs", "bas", "laken", "malesh",
"wallah", "wlhy", "wlah", "sa7", "tamam", "tmam", "sfr",
"tba2", "yb2a", "tab2a", "khalas", "kls", "zain", "wlahy",
"haga", "hagat", "7aga", "7agat", "lazm", "lazmak", "lazmik", "lazem",
"momken", "mumken", "ymkn", "wahid", "wa7ed", "wa7id", "ba3d", "b3d",
"zift", "zft", "3ayz", "3ayza", "3ayez", "kwaysa", "kwayes", "kwys",
"wahsh", "wahsha", "whs", "helw", "helwa", "gamed", "gamd", "gamda",
"ta3ban", "ta3bana", "za3lan", "za3lana", "mabsut", "mabsout", "zay", "zai",
"bshof", "bshuf", "bshouf", "bafdl", "bafadl", "ba3mel", "ba3ml",
"bybin", "byen", "roh", "geh", "gai", "gebt", "ray7", "rai7",
"rg3", "raga3", "rafe3", "tawer", "mshakel", "mushkela", "mwdu3", "mawdu3",
"khidma", "khidme", "talb", "talab", "balgh", "blag", "khsar", "khasar",
"shahn", "kan", "kanet", "knt", "byt", "byt3", "byt3ml",
"ht", "han", "hyt", "mls", "mls3", "msh3",
"kol", "kul", "gwa", "bra", "wl", "bl", "ml", "fl", "tl", "sl", "hl", "ql", "dl", "rl", "nl",
}
_ENGLISH_STOPWORDS = {
"this", "is", "are", "am", "was", "were", "it", "in", "on", "at",
"to", "for", "out", "of", "and", "the", "i", "a", "an", "my", "your",
"we", "he", "she", "they", "us", "me", "him", "her", "them",
"have", "has", "had", "do", "does", "did", "will", "would", "can",
"could", "should", "not", "no", "yes", "but", "or", "so", "if",
"then", "that", "which", "who", "what", "how", "when", "where", "why",
}
_FRANCO_START_DIGRAPHS = ("sh", "kh", "gh", "wl", "mn", "bn", "tn", "rh", "zh", "dh")
_FRANCO_DIGITS = set("23456789")
_VOWELS = set("aeiouAEIOU")
# Fuzzy targets restricted entirely to Arabizi
_FUZZY_TARGETS = frozenset(FRANCO_LEXICON)
def classify_word(word: str) -> str:
if re.search(r"[\u0600-\u06FF]", word): return "arabic"
core = re.sub(r"[^\w]", "", word).lower()
if not core: return "arabic"
if core in _ENGLISH_STOPWORDS: return "english"
if any(c in core for c in _FRANCO_DIGITS): return "franco"
if any(core.startswith(d) for d in _FRANCO_START_DIGRAPHS): return "franco"
if core in FRANCO_LEXICON: return "franco"
if len(core) <= 4 and sum(c in _VOWELS for c in core) <= 1: return "franco"
if difflib.get_close_matches(core, _FUZZY_TARGETS, n=1, cutoff=0.80):
return "franco"
return "english"
_FRANCO_MAP = {
"2": "ุก", "3": "ุน", "4": "ุด", "5": "ุฎ",
"6": "ุท", "7": "ุญ", "8": "ู‚", "9": "ุต",
"a": "ุง", "b": "ุจ", "c": "ูƒ", "d": "ุฏ",
"e": "ูŠ", "f": "ู", "g": "ุฌ", "h": "ู‡",
"i": "ูŠ", "j": "ุฌ", "k": "ูƒ", "l": "ู„",
"m": "ู…", "n": "ู†", "o": "ูˆ", "p": "ุจ",
"q": "ู‚", "r": "ุฑ", "s": "ุณ", "t": "ุช",
"u": "ูˆ", "v": "ู", "w": "ูˆ", "x": "ุงูƒุณ",
"y": "ูŠ", "z": "ุฒ",
}
def transliterate_franco(text: str) -> str:
t = text.lower()
t = t.replace("sh", "ุด").replace("ch", "ุชุด").replace("kh", "ุฎ")
t = t.replace("th", "ุซ").replace("gh", "ุบ").replace("ph", "ู")
t = t.replace("ou", "ูˆ").replace("ee", "ูŠ").replace("oo", "ูˆ")
return "".join(_FRANCO_MAP.get(c, c) for c in t)
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# SPELLCHECKER & TRANSLATOR
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
_EGY_MODEL = "NAMAA-Space/masrawy-english-to-egyptian-arabic-translator-v2.9"
_egy_tokenizer = None
_egy_model = None
en_spell = SpellChecker(language='en')
def _load_namaa():
global _egy_tokenizer, _egy_model
if _egy_model is None:
print("Loading NAMAA ENโ†’EGY translatorโ€ฆ")
_egy_tokenizer = MarianTokenizer.from_pretrained(_EGY_MODEL)
_egy_model = MarianMTModel.from_pretrained(_EGY_MODEL).to(DEVICE)
_egy_model.eval()
def translate_english_span(chunk: str) -> str:
_load_namaa()
# 1. Spellcheck English to prevent NAMAA failure on typos
words = chunk.split()
corrected_words = []
for w in words:
fixed = en_spell.correction(w)
corrected_words.append(fixed if fixed else w)
cleaned_chunk = " ".join(corrected_words)
# 2. Safely translate or fallback to untouched Latin
try:
tokens = _egy_tokenizer(
[cleaned_chunk], return_tensors="pt", padding=True,
truncation=True, max_length=64
).to(DEVICE)
with torch.no_grad():
out = _egy_model.generate(**tokens, num_beams=4, max_new_tokens=64)
return _egy_tokenizer.decode(out[0], skip_special_tokens=True)
except Exception:
return cleaned_chunk
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
# CORE ROUTING PIPELINE
# โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€โ”€
def _process_core(raw: str, mode: str) -> tuple[str, str]:
# Strip tatweel/kashida FIRST so the prefix regex triggers correctly
raw = raw.replace("ู€", "")
# Detach Arabic prefix morphemes glued to Latin roots
raw = re.sub(
r"(ุงู„|ุจ|ุจูŠ|ู‡|ู|ูƒ|ู„|ู„ู„|ู…ุง|ู…ุด)([a-zA-Z0-9]+)",
r"\1 \2", raw
)
if not re.search(r"[a-zA-Z0-9]", raw):
return clean_text(raw), "Pure Arabic"
words = raw.split()
if not words:
return "", "Pure Arabic"
classified = [(w, classify_word(w)) for w in words]
spans: list[tuple[str, list[str]]] = []
for word, lang in classified:
if spans and spans[-1][0] == lang:
spans[-1][1].append(word)
else:
spans.append((lang, [word]))
output_parts, langs_used = [], set()
for lang, span_words in spans:
chunk = " ".join(span_words)
if lang == "arabic":
output_parts.append(chunk)
elif lang == "franco":
output_parts.append(transliterate_franco(chunk))
langs_used.add("franco")
else:
if mode == "full":
output_parts.append(translate_english_span(chunk))
else:
output_parts.append(transliterate_franco(chunk))
langs_used.add("english")
cleaned = clean_text(" ".join(output_parts))
if langs_used == {"franco", "english"}: route_tag = "Mixed Franco+English โ†’ Arabic (Deep Fusion)"
elif "franco" in langs_used: route_tag = "Franco Transliterated โ†’ Arabic (Deep Fusion)"
elif "english" in langs_used: route_tag = "English Translated โ†’ Arabic (Deep Fusion)"
else: route_tag = "Pure Arabic (Deep Fusion)"
return cleaned, route_tag
def preprocess(raw: str) -> tuple[str, str]:
return _process_core(raw, mode="full")
def clean_text_v5(text: str, mode: str = "transliterate_only") -> str:
if not isinstance(text, str): return ""
return _process_core(text, mode=mode)[0]