"""Arabic text normalisation and source retrieval over the Quran and Hadith corpora. The retriever is the single shared index used by both verification and correction. It was extracted from the IslamicEval 2025 research notebook (``research/IslamicEval_Unified.ipynb``) and keeps its retrieval logic: * Quran: word-vote retrieval of single ayahs (coverage / precision F1) plus IDF-weighted seeds for multi-ayah window search. * Hadith: IDF word-vote recall followed by a character 4-gram cosine re-rank. """ from __future__ import annotations import gzip import json import logging import math import os import re import unicodedata from collections import Counter, defaultdict from pathlib import Path from typing import Dict, Iterable, List, Optional, Sequence, Tuple logger = logging.getLogger(__name__) BASE_DIR = Path(__file__).resolve().parent DEFAULT_QURAN_PATH = BASE_DIR / "data" / "quran.json" DEFAULT_HADITH_PATH = BASE_DIR / "data" / "hadith.json" # -------------------------------------------------------------------------------------------------------------- # Arabic normalisation # -------------------------------------------------------------------------------------------------------------- _DIACRITICS = re.compile( r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]" ) _PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`]") _MULTI_SPACE = re.compile(r"\s+") _ALEF_VARIANTS = re.compile(r"[أإآٱ]") _MATCH_DIACRITICS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]") STOPWORDS = frozenset( """من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split() ) def normalize_strict(text: Optional[str]) -> str: """Remove diacritics, unify alef variants and strip punctuation (keeps ta marbuta / alef maqsura).""" if not text: return "" text = unicodedata.normalize("NFC", text) text = _DIACRITICS.sub("", text) text = _ALEF_VARIANTS.sub("ا", text) text = _PUNCTUATION.sub(" ", text) return _MULTI_SPACE.sub(" ", text).strip() def normalize_lenient(text: Optional[str]) -> str: """Strict normalisation plus ta-marbuta/ha and alef-maqsura/ya folding (used for Hadith, whose spelling varies).""" return normalize_strict(text).replace("ة", "ه").replace("ى", "ي") def normalize_for_matching(text: Optional[str]) -> str: """Aggressive normalisation used for window matching: keeps only Arabic letters and single spaces.""" if not text: return "" text = _MATCH_DIACRITICS.sub("", text) for src, dst in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")): text = text.replace(src, dst) text = re.sub(r"[^\u0600-\u06FF\s]", " ", text) return re.sub(r"\s+", " ", text).strip() def tokenize(text: str) -> List[str]: return [token for token in text.split() if token] def content_words(tokens: Iterable[str]) -> List[str]: """Drop stop-words and single-letter tokens.""" return [t for t in tokens if t not in STOPWORDS and len(t) > 1] def _char_ngrams(text: str, n: int = 4) -> set: return {text[i : i + n] for i in range(len(text) - n + 1)} # -------------------------------------------------------------------------------------------------------------- # Retriever # -------------------------------------------------------------------------------------------------------------- class CorpusError(RuntimeError): """Raised when a corpus file is missing or malformed.""" class SourceRetriever: """In-memory index over ``quran.json`` and ``hadith.json`` (builds in a few seconds).""" def __init__( self, quran_path: os.PathLike = DEFAULT_QURAN_PATH, hadith_path: os.PathLike = DEFAULT_HADITH_PATH, ) -> None: quran_path, hadith_path = Path(quran_path), Path(hadith_path) gzipped = hadith_path.with_name(hadith_path.name + ".gz") if not hadith_path.is_file() and gzipped.is_file(): # compressed copy used by the in-browser deployment hadith_path = gzipped for path in (quran_path, hadith_path): if not path.is_file(): raise CorpusError(f"Corpus file not found: {path}") self.__dict__.update(self._build(quran_path, hadith_path)) n_ayahs, n_hadith = len(self.quran), len(self.hadith) self.quran_idf = {w: math.log((n_ayahs + 1) / (len(d) + 1)) + 1 for w, d in self.q_content_index.items()} self.hadith_idf = {w: math.log((n_hadith + 1) / (len(d) + 1)) + 1 for w, d in self.h_content_index.items()} logger.info("Retriever ready: %d ayahs, %d hadith records", n_ayahs, n_hadith) # ---- construction --------------------------------------------------------------------------------------- @staticmethod def _read_json(path: Path) -> list: try: opener = gzip.open if path.suffix == ".gz" else open with opener(path, "rt", encoding="utf-8") as handle: data = json.load(handle) except (OSError, json.JSONDecodeError) as exc: raise CorpusError(f"Cannot read corpus file {path}: {exc}") from exc if not isinstance(data, list): raise CorpusError(f"Unexpected corpus format in {path}: expected a JSON list") return data def _build(self, quran_path: Path, hadith_path: Path) -> dict: quran: List[dict] = [] q_norm_match: List[str] = [] q_all_index: Dict[str, List[int]] = defaultdict(list) q_content_index: Dict[str, List[int]] = defaultdict(list) q_word_count: List[int] = [] by_surah: Dict[int, Dict[int, int]] = defaultdict(dict) for entry in self._read_json(quran_path): text = (entry.get("ayah_text") or "").strip() if not text: continue idx = len(quran) quran.append( { "surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""), "ayah_id": entry.get("ayah_id"), "text": text, } ) by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx q_norm_match.append(normalize_for_matching(text)) tokens = tokenize(normalize_strict(text)) q_word_count.append(len(tokens)) for word in set(tokens): q_all_index[word].append(idx) for word in set(content_words(q_norm_match[-1].split())): q_content_index[word].append(idx) hadith: List[dict] = [] h_content_index: Dict[str, List[int]] = defaultdict(list) for entry in self._read_json(hadith_path): if not entry: continue matn = (entry.get("Matn") or "").strip() or None full = (entry.get("hadithTxt") or "").strip() or None if not (matn or full): continue record = { "hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"), "matn": matn, "full": full, "norm_matn": normalize_for_matching(matn) if matn else None, "norm_full": normalize_for_matching(full) if full else None, } idx = len(hadith) hadith.append(record) words = set() for key in ("norm_matn", "norm_full"): if record[key]: words |= set(content_words(record[key].split())) for word in words: h_content_index[word].append(idx) return { "quran": quran, "quran_by_surah": dict(by_surah), "q_norm_match": q_norm_match, "q_all_index": dict(q_all_index), "q_content_index": dict(q_content_index), "q_word_count": q_word_count, "hadith": hadith, "h_content_index": dict(h_content_index), } # ---- Quran ---------------------------------------------------------------------------------------------- def search_quran_ayahs(self, query: str, top_k: int = 25, extra_seeds: int = 10) -> List[dict]: """Single-ayah candidates ranked by the F1 of query coverage and ayah precision. The F1 ranking favours short ayahs, so a fragment of a long ayah can be missed; the top IDF-voted ayahs (rare-word matches) are appended as extra candidates to recover those partial quotations.""" query_words = tokenize(normalize_strict(query)) if not query_words: return [] votes: Dict[int, int] = defaultdict(int) for word in query_words: for idx in self.q_all_index.get(word, ()): votes[idx] += 1 scored: List[Tuple[int, float]] = [] for idx, vote in votes.items(): coverage = vote / len(query_words) precision = vote / self.q_word_count[idx] if self.q_word_count[idx] else 0.0 f1 = 2 * coverage * precision / (coverage + precision) if coverage + precision > 0 else 0.0 scored.append((idx, f1)) scored.sort(key=lambda item: item[1], reverse=True) ranked = [idx for idx, _ in scored[:top_k]] scores = dict(scored) seeds = self.quran_seed_ayahs(content_words(normalize_for_matching(query).split()), extra_seeds) ranked += [idx for idx in seeds if idx not in set(ranked)] results = [] for idx in ranked: candidate = dict(self.quran[idx]) candidate.update(type="Quran", retrieval_score=scores.get(idx, 0.0)) results.append(candidate) return results def quran_seed_ayahs(self, query_words: Sequence[str], top_k: int = 25) -> List[int]: """IDF-voted ayah indices used to seed the multi-ayah window search.""" return [idx for idx, _ in self.vote(self.q_content_index, self.quran_idf, query_words, top_k)] # ---- Hadith --------------------------------------------------------------------------------------------- def search_hadith(self, query: str, top_k: int = 15, pool: int = 40) -> List[dict]: """IDF word-vote recall, then re-rank by character 4-gram cosine similarity.""" words = content_words(normalize_for_matching(query).split()) if not words: return [] query_grams = _char_ngrams(normalize_lenient(query)) results = [] for idx, _ in self.vote(self.h_content_index, self.hadith_idf, words, pool): record = self.hadith[idx] text = record["matn"] or record["full"] doc_grams = _char_ngrams(normalize_lenient(text)) cosine = ( len(query_grams & doc_grams) / ((len(query_grams) * len(doc_grams)) ** 0.5) if query_grams and doc_grams else 0.0 ) results.append( { "type": "Hadith", "idx": idx, "hadithID": record["hadithID"], "book": record["book"], "title": record["title"], "text": text, "has_matn": bool(record["matn"]), "retrieval_score": cosine, } ) results.sort(key=lambda c: c["retrieval_score"], reverse=True) return results[:top_k] # ---- shared --------------------------------------------------------------------------------------------- @staticmethod def vote(index: Dict[str, List[int]], idf: Dict[str, float], words: Iterable[str], top_k: int) -> List[Tuple[int, float]]: scores: Counter = Counter() for word in set(words): weight = idf.get(word, 0.0) if weight <= 0: continue for doc in index.get(word, ()): scores[doc] += weight return scores.most_common(top_k)