Islamic-Content-Verifier / retrieval.py
Ghada-99-Ragab's picture
Upload 8 files
1fe445a verified
Raw History Blame Contribute Delete
12.4 kB
"""Arabic text normalisation and source retrieval over the Quran and Hadith corpora.
The retriever is the single shared index used by both verification and correction. It was extracted from the
IslamicEval 2025 research notebook (``research/IslamicEval_Unified.ipynb``) and keeps its retrieval logic:
* Quran: word-vote retrieval of single ayahs (coverage / precision F1) plus IDF-weighted seeds for
multi-ayah window search.
* Hadith: IDF word-vote recall followed by a character 4-gram cosine re-rank.
"""
from __future__ import annotations
import gzip
import json
import logging
import math
import os
import re
import unicodedata
from collections import Counter, defaultdict
from pathlib import Path
from typing import Dict, Iterable, List, Optional, Sequence, Tuple
logger = logging.getLogger(__name__)
BASE_DIR = Path(__file__).resolve().parent
DEFAULT_QURAN_PATH = BASE_DIR / "data" / "quran.json"
DEFAULT_HADITH_PATH = BASE_DIR / "data" / "hadith.json"
# --------------------------------------------------------------------------------------------------------------
# Arabic normalisation
# --------------------------------------------------------------------------------------------------------------
_DIACRITICS = re.compile(
r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]"
)
_PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`]")
_MULTI_SPACE = re.compile(r"\s+")
_ALEF_VARIANTS = re.compile(r"[أإآٱ]")
_MATCH_DIACRITICS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]")
STOPWORDS = frozenset(
"""من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت
هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split()
)
def normalize_strict(text: Optional[str]) -> str:
"""Remove diacritics, unify alef variants and strip punctuation (keeps ta marbuta / alef maqsura)."""
if not text:
return ""
text = unicodedata.normalize("NFC", text)
text = _DIACRITICS.sub("", text)
text = _ALEF_VARIANTS.sub("ا", text)
text = _PUNCTUATION.sub(" ", text)
return _MULTI_SPACE.sub(" ", text).strip()
def normalize_lenient(text: Optional[str]) -> str:
"""Strict normalisation plus ta-marbuta/ha and alef-maqsura/ya folding (used for Hadith, whose spelling varies)."""
return normalize_strict(text).replace("ة", "ه").replace("ى", "ي")
def normalize_for_matching(text: Optional[str]) -> str:
"""Aggressive normalisation used for window matching: keeps only Arabic letters and single spaces."""
if not text:
return ""
text = _MATCH_DIACRITICS.sub("", text)
for src, dst in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")):
text = text.replace(src, dst)
text = re.sub(r"[^\u0600-\u06FF\s]", " ", text)
return re.sub(r"\s+", " ", text).strip()
def tokenize(text: str) -> List[str]:
return [token for token in text.split() if token]
def content_words(tokens: Iterable[str]) -> List[str]:
"""Drop stop-words and single-letter tokens."""
return [t for t in tokens if t not in STOPWORDS and len(t) > 1]
def _char_ngrams(text: str, n: int = 4) -> set:
return {text[i : i + n] for i in range(len(text) - n + 1)}
# --------------------------------------------------------------------------------------------------------------
# Retriever
# --------------------------------------------------------------------------------------------------------------
class CorpusError(RuntimeError):
"""Raised when a corpus file is missing or malformed."""
class SourceRetriever:
"""In-memory index over ``quran.json`` and ``hadith.json`` (builds in a few seconds)."""
def __init__(
self,
quran_path: os.PathLike = DEFAULT_QURAN_PATH,
hadith_path: os.PathLike = DEFAULT_HADITH_PATH,
) -> None:
quran_path, hadith_path = Path(quran_path), Path(hadith_path)
gzipped = hadith_path.with_name(hadith_path.name + ".gz")
if not hadith_path.is_file() and gzipped.is_file(): # compressed copy used by the in-browser deployment
hadith_path = gzipped
for path in (quran_path, hadith_path):
if not path.is_file():
raise CorpusError(f"Corpus file not found: {path}")
self.__dict__.update(self._build(quran_path, hadith_path))
n_ayahs, n_hadith = len(self.quran), len(self.hadith)
self.quran_idf = {w: math.log((n_ayahs + 1) / (len(d) + 1)) + 1 for w, d in self.q_content_index.items()}
self.hadith_idf = {w: math.log((n_hadith + 1) / (len(d) + 1)) + 1 for w, d in self.h_content_index.items()}
logger.info("Retriever ready: %d ayahs, %d hadith records", n_ayahs, n_hadith)
# ---- construction ---------------------------------------------------------------------------------------
@staticmethod
def _read_json(path: Path) -> list:
try:
opener = gzip.open if path.suffix == ".gz" else open
with opener(path, "rt", encoding="utf-8") as handle:
data = json.load(handle)
except (OSError, json.JSONDecodeError) as exc:
raise CorpusError(f"Cannot read corpus file {path}: {exc}") from exc
if not isinstance(data, list):
raise CorpusError(f"Unexpected corpus format in {path}: expected a JSON list")
return data
def _build(self, quran_path: Path, hadith_path: Path) -> dict:
quran: List[dict] = []
q_norm_match: List[str] = []
q_all_index: Dict[str, List[int]] = defaultdict(list)
q_content_index: Dict[str, List[int]] = defaultdict(list)
q_word_count: List[int] = []
by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
for entry in self._read_json(quran_path):
text = (entry.get("ayah_text") or "").strip()
if not text:
continue
idx = len(quran)
quran.append(
{
"surah_id": entry.get("surah_id"),
"surah_name": entry.get("surah_name", ""),
"ayah_id": entry.get("ayah_id"),
"text": text,
}
)
by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
q_norm_match.append(normalize_for_matching(text))
tokens = tokenize(normalize_strict(text))
q_word_count.append(len(tokens))
for word in set(tokens):
q_all_index[word].append(idx)
for word in set(content_words(q_norm_match[-1].split())):
q_content_index[word].append(idx)
hadith: List[dict] = []
h_content_index: Dict[str, List[int]] = defaultdict(list)
for entry in self._read_json(hadith_path):
if not entry:
continue
matn = (entry.get("Matn") or "").strip() or None
full = (entry.get("hadithTxt") or "").strip() or None
if not (matn or full):
continue
record = {
"hadithID": entry.get("hadithID"),
"book": entry.get("BookID"),
"title": entry.get("title"),
"matn": matn,
"full": full,
"norm_matn": normalize_for_matching(matn) if matn else None,
"norm_full": normalize_for_matching(full) if full else None,
}
idx = len(hadith)
hadith.append(record)
words = set()
for key in ("norm_matn", "norm_full"):
if record[key]:
words |= set(content_words(record[key].split()))
for word in words:
h_content_index[word].append(idx)
return {
"quran": quran,
"quran_by_surah": dict(by_surah),
"q_norm_match": q_norm_match,
"q_all_index": dict(q_all_index),
"q_content_index": dict(q_content_index),
"q_word_count": q_word_count,
"hadith": hadith,
"h_content_index": dict(h_content_index),
}
# ---- Quran ----------------------------------------------------------------------------------------------
def search_quran_ayahs(self, query: str, top_k: int = 25, extra_seeds: int = 10) -> List[dict]:
"""Single-ayah candidates ranked by the F1 of query coverage and ayah precision.
The F1 ranking favours short ayahs, so a fragment of a long ayah can be missed; the top IDF-voted ayahs
(rare-word matches) are appended as extra candidates to recover those partial quotations."""
query_words = tokenize(normalize_strict(query))
if not query_words:
return []
votes: Dict[int, int] = defaultdict(int)
for word in query_words:
for idx in self.q_all_index.get(word, ()):
votes[idx] += 1
scored: List[Tuple[int, float]] = []
for idx, vote in votes.items():
coverage = vote / len(query_words)
precision = vote / self.q_word_count[idx] if self.q_word_count[idx] else 0.0
f1 = 2 * coverage * precision / (coverage + precision) if coverage + precision > 0 else 0.0
scored.append((idx, f1))
scored.sort(key=lambda item: item[1], reverse=True)
ranked = [idx for idx, _ in scored[:top_k]]
scores = dict(scored)
seeds = self.quran_seed_ayahs(content_words(normalize_for_matching(query).split()), extra_seeds)
ranked += [idx for idx in seeds if idx not in set(ranked)]
results = []
for idx in ranked:
candidate = dict(self.quran[idx])
candidate.update(type="Quran", retrieval_score=scores.get(idx, 0.0))
results.append(candidate)
return results
def quran_seed_ayahs(self, query_words: Sequence[str], top_k: int = 25) -> List[int]:
"""IDF-voted ayah indices used to seed the multi-ayah window search."""
return [idx for idx, _ in self.vote(self.q_content_index, self.quran_idf, query_words, top_k)]
# ---- Hadith ---------------------------------------------------------------------------------------------
def search_hadith(self, query: str, top_k: int = 15, pool: int = 40) -> List[dict]:
"""IDF word-vote recall, then re-rank by character 4-gram cosine similarity."""
words = content_words(normalize_for_matching(query).split())
if not words:
return []
query_grams = _char_ngrams(normalize_lenient(query))
results = []
for idx, _ in self.vote(self.h_content_index, self.hadith_idf, words, pool):
record = self.hadith[idx]
text = record["matn"] or record["full"]
doc_grams = _char_ngrams(normalize_lenient(text))
cosine = (
len(query_grams & doc_grams) / ((len(query_grams) * len(doc_grams)) ** 0.5)
if query_grams and doc_grams
else 0.0
)
results.append(
{
"type": "Hadith",
"idx": idx,
"hadithID": record["hadithID"],
"book": record["book"],
"title": record["title"],
"text": text,
"has_matn": bool(record["matn"]),
"retrieval_score": cosine,
}
)
results.sort(key=lambda c: c["retrieval_score"], reverse=True)
return results[:top_k]
# ---- shared ---------------------------------------------------------------------------------------------
@staticmethod
def vote(index: Dict[str, List[int]], idf: Dict[str, float], words: Iterable[str], top_k: int) -> List[Tuple[int, float]]:
scores: Counter = Counter()
for word in set(words):
weight = idf.get(word, 0.0)
if weight <= 0:
continue
for doc in index.get(word, ()):
scores[doc] += weight
return scores.most_common(top_k)