Spaces:
Running
Running
File size: 5,448 Bytes
597dbb9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 | """Phonetic-aware Arabic normalisation shared by indexing, retrieval, alignment and verification.
Three levels, from gentle to aggressive:
* ``normalize_strict`` diacritics / tatweel / Quranic marks removed, alef forms unified (hamza on waw / ya kept).
* ``normalize_lenient`` strict + ta marbuta -> ha and alef maqsura -> ya (used for Hadith, whose spelling varies).
* ``normalize_for_matching`` the phonetic skeleton used for retrieval and word alignment: additionally folds hamza
carriers and keeps Arabic letters only.
All levels also reconcile the Uthmani mushaf script with Modern Standard Arabic: alef wasla (ٱ) becomes alef, the
dagger alef / "alif khanjariyah" (ـٰ) and tatweel are removed, Persian ya / kaf and the ligature ﷲ are mapped to Arabic
letters, and zero-width / bidi control characters are dropped.
"""
from __future__ import annotations
import re
import unicodedata
from typing import List, Tuple
_ZERO_WIDTH = re.compile("[\u200b-\u200f\u202a-\u202e\u2066-\u2069\ufeff]")
_MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]")
_MATCH_MARKS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]")
_PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`﴿﴾]")
_SPACES = re.compile(r"\s+")
_ALEF_FORMS = re.compile(r"[أإآٱ]")
_PERSIAN = str.maketrans({"ی": "ي", "ې": "ي", "ک": "ك", "ە": "ه", "ہ": "ه", "ۀ": "ه"})
STOPWORDS = frozenset(
"""من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت
هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split()
)
def _prefold(text: str) -> str:
"""Script-level clean-up common to every normalisation level."""
text = unicodedata.normalize("NFC", text).replace("ﷲ", "الله")
return _ZERO_WIDTH.sub("", text).translate(_PERSIAN)
def normalize_strict(text) -> str:
if not text:
return ""
text = _MARKS.sub("", _prefold(text))
text = _ALEF_FORMS.sub("ا", text)
return _SPACES.sub(" ", _PUNCTUATION.sub(" ", text)).strip()
def normalize_lenient(text) -> str:
return normalize_strict(text).replace("ة", "ه").replace("ى", "ي")
def normalize_for_matching(text) -> str:
"""Phonetic skeleton: Arabic letters only, hamza carriers / ta marbuta / alef maqsura folded."""
if not text:
return ""
text = _MATCH_MARKS.sub("", _prefold(text))
for source, target in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")):
text = text.replace(source, target)
text = re.sub(r"[^\u0621-\u064A\s]", " ", text) # Arabic letters only: drops Arabic punctuation, digits, Latin
return _SPACES.sub(" ", text).strip()
def tokenize(text: str) -> List[str]:
return [token for token in text.split() if token]
def content_words(tokens) -> List[str]:
"""Drop stop-words and single-letter tokens."""
return [t for t in tokens if t not in STOPWORDS and len(t) > 1]
def char_ngrams(text: str, n: int = 4) -> set:
return {text[i : i + n] for i in range(len(text) - n + 1)}
# ---- word-level helpers used by the aligner -----------------------------------------------------------------------
_ARABIC_LETTER = re.compile(r"[\u0621-\u064A]")
def aligned_words(text: str) -> List[Tuple[str, str]]:
"""``[(original_word, phonetic_skeleton)]``; ayah markers like ``(12)`` and mark-only tokens are dropped."""
pairs = []
for word in text.split():
skeleton = normalize_for_matching(word)
if skeleton and _ARABIC_LETTER.search(skeleton):
pairs.append((word, skeleton.replace(" ", "")))
return pairs
_VOWELS = {"\u064E": "a", "\u064F": "u", "\u0650": "i", "\u064B": "A", "\u064C": "U", "\u064D": "I", "\u0651": "~"}
def vowel_signature(word: str) -> List[Tuple[str, str]]:
"""Per base letter, the set of short vowels / tanween / shadda written on it (sukun and Quranic marks ignored)."""
word = _prefold(word).replace("ٱ", "ا")
signature: List[Tuple[str, str]] = []
for char in word:
if char in _VOWELS:
if signature:
letter, marks = signature[-1]
signature[-1] = (letter, "".join(sorted(set(marks + _VOWELS[char]))))
elif "\u0621" <= char <= "\u064A":
signature.append((char, ""))
return signature
# ---- phonetic sequence matching --------------------------------------------------------------------------------
# Letters that are commonly confused in writing or dictation are folded into one class, so an altered or mis-spelled
# quotation can still be *found*; the verifier then reports the exact differences.
_PHONETIC_CLASSES = {"ص": "س", "ث": "س", "ذ": "ز", "ظ": "ز", "ض": "ز", "ط": "ت", "ك": "ق", "ح": "ه", "غ": "ع"}
_PHONETIC_TABLE = str.maketrans(_PHONETIC_CLASSES)
def phonetic_key(skeleton_word: str) -> str:
"""Sound-alike key of a phonetic-skeleton word (confusable consonants folded, doubled letters collapsed)."""
word = skeleton_word.translate(_PHONETIC_TABLE)
return "".join(ch for i, ch in enumerate(word) if i == 0 or ch != word[i - 1])
|