Download retrieval.py from Ghada-99-Ragab/Islamic-Content-Verifier: direct link, hf CLI and curl.
- Browser
- Download file 12.4 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Islamic-Content-Verifier/resolve/main/retrieval.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Islamic-Content-Verifier/retrieval.py
-
curl -L -o retrieval.py https://huggingface.co/spaces/Ghada-99-Ragab/Islamic-Content-Verifier/resolve/main/retrieval.py
12.4 kB
| """Arabic text normalisation and source retrieval over the Quran and Hadith corpora. | |
| The retriever is the single shared index used by both verification and correction. It was extracted from the | |
| IslamicEval 2025 research notebook (``research/IslamicEval_Unified.ipynb``) and keeps its retrieval logic: | |
| * Quran: word-vote retrieval of single ayahs (coverage / precision F1) plus IDF-weighted seeds for | |
| multi-ayah window search. | |
| * Hadith: IDF word-vote recall followed by a character 4-gram cosine re-rank. | |
| """ | |
| from __future__ import annotations | |
| import gzip | |
| import json | |
| import logging | |
| import math | |
| import os | |
| import re | |
| import unicodedata | |
| from collections import Counter, defaultdict | |
| from pathlib import Path | |
| from typing import Dict, Iterable, List, Optional, Sequence, Tuple | |
| logger = logging.getLogger(__name__) | |
| BASE_DIR = Path(__file__).resolve().parent | |
| DEFAULT_QURAN_PATH = BASE_DIR / "data" / "quran.json" | |
| DEFAULT_HADITH_PATH = BASE_DIR / "data" / "hadith.json" | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Arabic normalisation | |
| # -------------------------------------------------------------------------------------------------------------- | |
| _DIACRITICS = re.compile( | |
| r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E4\u06E7\u06E8\u06EA-\u06ED\u0640]" | |
| ) | |
| _PUNCTUATION = re.compile(r"[،؛؟!،.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`]") | |
| _MULTI_SPACE = re.compile(r"\s+") | |
| _ALEF_VARIANTS = re.compile(r"[أإآٱ]") | |
| _MATCH_DIACRITICS = re.compile(r"[\u0610-\u061A\u064B-\u065F\u0670\u06D6-\u06DC\u06DF-\u06E8\u06EA-\u06ED\u0640]") | |
| STOPWORDS = frozenset( | |
| """من في على ان أن إن الى إلى عن مع ما لا لم لن قد و ثم أو او هو هي هم انت أنتم كان كانت يكون تكون قال قالت | |
| هذا هذه ذلك تلك الذي التي الذين اللاتي اللائي كل بعض غير عند بين حتى إذا اذا لو لكن بل يا أيها ايها""".split() | |
| ) | |
| def normalize_strict(text: Optional[str]) -> str: | |
| """Remove diacritics, unify alef variants and strip punctuation (keeps ta marbuta / alef maqsura).""" | |
| if not text: | |
| return "" | |
| text = unicodedata.normalize("NFC", text) | |
| text = _DIACRITICS.sub("", text) | |
| text = _ALEF_VARIANTS.sub("ا", text) | |
| text = _PUNCTUATION.sub(" ", text) | |
| return _MULTI_SPACE.sub(" ", text).strip() | |
| def normalize_lenient(text: Optional[str]) -> str: | |
| """Strict normalisation plus ta-marbuta/ha and alef-maqsura/ya folding (used for Hadith, whose spelling varies).""" | |
| return normalize_strict(text).replace("ة", "ه").replace("ى", "ي") | |
| def normalize_for_matching(text: Optional[str]) -> str: | |
| """Aggressive normalisation used for window matching: keeps only Arabic letters and single spaces.""" | |
| if not text: | |
| return "" | |
| text = _MATCH_DIACRITICS.sub("", text) | |
| for src, dst in (("أ", "ا"), ("إ", "ا"), ("آ", "ا"), ("ٱ", "ا"), ("ؤ", "و"), ("ئ", "ي"), ("ة", "ه"), ("ى", "ي")): | |
| text = text.replace(src, dst) | |
| text = re.sub(r"[^\u0600-\u06FF\s]", " ", text) | |
| return re.sub(r"\s+", " ", text).strip() | |
| def tokenize(text: str) -> List[str]: | |
| return [token for token in text.split() if token] | |
| def content_words(tokens: Iterable[str]) -> List[str]: | |
| """Drop stop-words and single-letter tokens.""" | |
| return [t for t in tokens if t not in STOPWORDS and len(t) > 1] | |
| def _char_ngrams(text: str, n: int = 4) -> set: | |
| return {text[i : i + n] for i in range(len(text) - n + 1)} | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Retriever | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class CorpusError(RuntimeError): | |
| """Raised when a corpus file is missing or malformed.""" | |
| class SourceRetriever: | |
| """In-memory index over ``quran.json`` and ``hadith.json`` (builds in a few seconds).""" | |
| def __init__( | |
| self, | |
| quran_path: os.PathLike = DEFAULT_QURAN_PATH, | |
| hadith_path: os.PathLike = DEFAULT_HADITH_PATH, | |
| ) -> None: | |
| quran_path, hadith_path = Path(quran_path), Path(hadith_path) | |
| gzipped = hadith_path.with_name(hadith_path.name + ".gz") | |
| if not hadith_path.is_file() and gzipped.is_file(): # compressed copy used by the in-browser deployment | |
| hadith_path = gzipped | |
| for path in (quran_path, hadith_path): | |
| if not path.is_file(): | |
| raise CorpusError(f"Corpus file not found: {path}") | |
| self.__dict__.update(self._build(quran_path, hadith_path)) | |
| n_ayahs, n_hadith = len(self.quran), len(self.hadith) | |
| self.quran_idf = {w: math.log((n_ayahs + 1) / (len(d) + 1)) + 1 for w, d in self.q_content_index.items()} | |
| self.hadith_idf = {w: math.log((n_hadith + 1) / (len(d) + 1)) + 1 for w, d in self.h_content_index.items()} | |
| logger.info("Retriever ready: %d ayahs, %d hadith records", n_ayahs, n_hadith) | |
| # ---- construction --------------------------------------------------------------------------------------- | |
| def _read_json(path: Path) -> list: | |
| try: | |
| opener = gzip.open if path.suffix == ".gz" else open | |
| with opener(path, "rt", encoding="utf-8") as handle: | |
| data = json.load(handle) | |
| except (OSError, json.JSONDecodeError) as exc: | |
| raise CorpusError(f"Cannot read corpus file {path}: {exc}") from exc | |
| if not isinstance(data, list): | |
| raise CorpusError(f"Unexpected corpus format in {path}: expected a JSON list") | |
| return data | |
| def _build(self, quran_path: Path, hadith_path: Path) -> dict: | |
| quran: List[dict] = [] | |
| q_norm_match: List[str] = [] | |
| q_all_index: Dict[str, List[int]] = defaultdict(list) | |
| q_content_index: Dict[str, List[int]] = defaultdict(list) | |
| q_word_count: List[int] = [] | |
| by_surah: Dict[int, Dict[int, int]] = defaultdict(dict) | |
| for entry in self._read_json(quran_path): | |
| text = (entry.get("ayah_text") or "").strip() | |
| if not text: | |
| continue | |
| idx = len(quran) | |
| quran.append( | |
| { | |
| "surah_id": entry.get("surah_id"), | |
| "surah_name": entry.get("surah_name", ""), | |
| "ayah_id": entry.get("ayah_id"), | |
| "text": text, | |
| } | |
| ) | |
| by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx | |
| q_norm_match.append(normalize_for_matching(text)) | |
| tokens = tokenize(normalize_strict(text)) | |
| q_word_count.append(len(tokens)) | |
| for word in set(tokens): | |
| q_all_index[word].append(idx) | |
| for word in set(content_words(q_norm_match[-1].split())): | |
| q_content_index[word].append(idx) | |
| hadith: List[dict] = [] | |
| h_content_index: Dict[str, List[int]] = defaultdict(list) | |
| for entry in self._read_json(hadith_path): | |
| if not entry: | |
| continue | |
| matn = (entry.get("Matn") or "").strip() or None | |
| full = (entry.get("hadithTxt") or "").strip() or None | |
| if not (matn or full): | |
| continue | |
| record = { | |
| "hadithID": entry.get("hadithID"), | |
| "book": entry.get("BookID"), | |
| "title": entry.get("title"), | |
| "matn": matn, | |
| "full": full, | |
| "norm_matn": normalize_for_matching(matn) if matn else None, | |
| "norm_full": normalize_for_matching(full) if full else None, | |
| } | |
| idx = len(hadith) | |
| hadith.append(record) | |
| words = set() | |
| for key in ("norm_matn", "norm_full"): | |
| if record[key]: | |
| words |= set(content_words(record[key].split())) | |
| for word in words: | |
| h_content_index[word].append(idx) | |
| return { | |
| "quran": quran, | |
| "quran_by_surah": dict(by_surah), | |
| "q_norm_match": q_norm_match, | |
| "q_all_index": dict(q_all_index), | |
| "q_content_index": dict(q_content_index), | |
| "q_word_count": q_word_count, | |
| "hadith": hadith, | |
| "h_content_index": dict(h_content_index), | |
| } | |
| # ---- Quran ---------------------------------------------------------------------------------------------- | |
| def search_quran_ayahs(self, query: str, top_k: int = 25, extra_seeds: int = 10) -> List[dict]: | |
| """Single-ayah candidates ranked by the F1 of query coverage and ayah precision. | |
| The F1 ranking favours short ayahs, so a fragment of a long ayah can be missed; the top IDF-voted ayahs | |
| (rare-word matches) are appended as extra candidates to recover those partial quotations.""" | |
| query_words = tokenize(normalize_strict(query)) | |
| if not query_words: | |
| return [] | |
| votes: Dict[int, int] = defaultdict(int) | |
| for word in query_words: | |
| for idx in self.q_all_index.get(word, ()): | |
| votes[idx] += 1 | |
| scored: List[Tuple[int, float]] = [] | |
| for idx, vote in votes.items(): | |
| coverage = vote / len(query_words) | |
| precision = vote / self.q_word_count[idx] if self.q_word_count[idx] else 0.0 | |
| f1 = 2 * coverage * precision / (coverage + precision) if coverage + precision > 0 else 0.0 | |
| scored.append((idx, f1)) | |
| scored.sort(key=lambda item: item[1], reverse=True) | |
| ranked = [idx for idx, _ in scored[:top_k]] | |
| scores = dict(scored) | |
| seeds = self.quran_seed_ayahs(content_words(normalize_for_matching(query).split()), extra_seeds) | |
| ranked += [idx for idx in seeds if idx not in set(ranked)] | |
| results = [] | |
| for idx in ranked: | |
| candidate = dict(self.quran[idx]) | |
| candidate.update(type="Quran", retrieval_score=scores.get(idx, 0.0)) | |
| results.append(candidate) | |
| return results | |
| def quran_seed_ayahs(self, query_words: Sequence[str], top_k: int = 25) -> List[int]: | |
| """IDF-voted ayah indices used to seed the multi-ayah window search.""" | |
| return [idx for idx, _ in self.vote(self.q_content_index, self.quran_idf, query_words, top_k)] | |
| # ---- Hadith --------------------------------------------------------------------------------------------- | |
| def search_hadith(self, query: str, top_k: int = 15, pool: int = 40) -> List[dict]: | |
| """IDF word-vote recall, then re-rank by character 4-gram cosine similarity.""" | |
| words = content_words(normalize_for_matching(query).split()) | |
| if not words: | |
| return [] | |
| query_grams = _char_ngrams(normalize_lenient(query)) | |
| results = [] | |
| for idx, _ in self.vote(self.h_content_index, self.hadith_idf, words, pool): | |
| record = self.hadith[idx] | |
| text = record["matn"] or record["full"] | |
| doc_grams = _char_ngrams(normalize_lenient(text)) | |
| cosine = ( | |
| len(query_grams & doc_grams) / ((len(query_grams) * len(doc_grams)) ** 0.5) | |
| if query_grams and doc_grams | |
| else 0.0 | |
| ) | |
| results.append( | |
| { | |
| "type": "Hadith", | |
| "idx": idx, | |
| "hadithID": record["hadithID"], | |
| "book": record["book"], | |
| "title": record["title"], | |
| "text": text, | |
| "has_matn": bool(record["matn"]), | |
| "retrieval_score": cosine, | |
| } | |
| ) | |
| results.sort(key=lambda c: c["retrieval_score"], reverse=True) | |
| return results[:top_k] | |
| # ---- shared --------------------------------------------------------------------------------------------- | |
| def vote(index: Dict[str, List[int]], idf: Dict[str, float], words: Iterable[str], top_k: int) -> List[Tuple[int, float]]: | |
| scores: Counter = Counter() | |
| for word in set(words): | |
| weight = idf.get(word, 0.0) | |
| if weight <= 0: | |
| continue | |
| for doc in index.get(word, ()): | |
| scores[doc] += weight | |
| return scores.most_common(top_k) | |