Ghada-99-Ragab's picture
Upload 8 files
1fe445a verified
Raw History Blame Contribute Delete
45.8 kB
"""Quotation detection, evidence-based verification and source-backed correction for Quran and Hadith.
This module consolidates the logic of the IslamicEval 2025 research notebook
(``research/IslamicEval_Unified.ipynb``) into one pipeline:
input text -> detection (1A) -> source retrieval -> verification (1B) -> evidence
-> strong evidence : source-backed correction (1C)
-> weak / ambiguous : human review
-> no source found : reported as unsupported, never "fixed"
Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated.
"""
from __future__ import annotations
import difflib
import logging
import re
import time
import unicodedata
from dataclasses import dataclass, field
from typing import Dict, List, Optional, Sequence
from retrieval import (
SourceRetriever,
content_words,
normalize_for_matching,
normalize_lenient,
normalize_strict,
tokenize,
)
try: # optional accelerator; identical formula (1 - distance / max_len)
from rapidfuzz.distance import Levenshtein as _RapidLevenshtein
except ImportError: # pragma: no cover
_RapidLevenshtein = None
logger = logging.getLogger(__name__)
MAX_INPUT_CHARS = 20_000
# --------------------------------------------------------------------------------------------------------------
# Configuration (all tunable numbers live here)
# --------------------------------------------------------------------------------------------------------------
@dataclass
class VerifierConfig:
"""Thresholds calibrated in Subtask 1B of the research notebook."""
quran_correct_threshold: float = 0.94
quran_uncertain_low: float = 0.45
quran_min_coverage: float = 0.40
hadith_correct_threshold: float = 0.79
hadith_uncertain_low: float = 0.30
hadith_min_coverage: float = 0.70
quran_top_k: int = 25
hadith_top_k: int = 15
hadith_retrieval_guard: float = 0.20
@dataclass
class CorrectorConfig:
"""Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the
case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact
fragment boundaries of a Hadith quotation cannot be reproduced reliably."""
max_window: int = 8
hadith_top_k: int = 40
quran_strong: float = 0.65
min_full_ratio: float = 0.40
quran_low: float = 0.55
hadith_strong: float = 1.01
hadith_low: float = 0.45
@dataclass
class PipelineConfig:
verifier: VerifierConfig = field(default_factory=VerifierConfig)
corrector: CorrectorConfig = field(default_factory=CorrectorConfig)
min_detection_conf: float = 0.60 # detector confidence below this -> human review
verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review
unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently
unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists)
# --------------------------------------------------------------------------------------------------------------
# Similarity signals (Subtask 1B)
# --------------------------------------------------------------------------------------------------------------
def _lcs_length(a: Sequence[str], b: Sequence[str]) -> int:
m, n = len(a), len(b)
if m == 0 or n == 0:
return 0
if m < n:
a, b, m, n = b, a, n, m
prev = [0] * (n + 1)
for i in range(m):
curr = [0] * (n + 1)
for j in range(n):
curr[j + 1] = prev[j] + 1 if a[i] == b[j] else max(curr[j], prev[j + 1])
prev = curr
return prev[n]
def _edit_similarity(a: str, b: str, max_len: int = 600) -> float:
"""Normalised Levenshtein similarity in [0, 1] on the first ``max_len`` characters."""
a, b = a[:max_len], b[:max_len]
if a == b:
return 1.0
if not a or not b:
return 0.0
if _RapidLevenshtein is not None:
return float(_RapidLevenshtein.normalized_similarity(a, b))
prev = list(range(len(b) + 1))
for i, ca in enumerate(a):
curr = [i + 1]
for j, cb in enumerate(b):
curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + (ca != cb)))
prev = curr
return 1.0 - prev[-1] / max(len(a), len(b))
def _light_normalize(text: str) -> str:
"""Keeps diacritics (so diacritic changes lower the score) but drops punctuation and tatweel."""
text = unicodedata.normalize("NFC", text)
text = re.sub(r"[،؛؟!.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`\u0640]", " ", text)
return re.sub(r"\s+", " ", text).strip()
def compute_signals(claim: str, candidate: str, content_type: str = "Ayah") -> Dict[str, float]:
"""Word-, character- and sequence-level similarity indicators between a quotation and a candidate source."""
is_quran = content_type == "Ayah"
normalize = normalize_strict if is_quran else normalize_lenient
norm_claim, norm_cand = normalize(claim), normalize(candidate)
claim_set, cand_set = set(tokenize(norm_claim)), set(tokenize(norm_cand))
if claim_set and cand_set:
shared = claim_set & cand_set
token_overlap = len(shared) / len(claim_set | cand_set)
coverage = len(shared) / len(claim_set)
else:
token_overlap = coverage = 0.0
claim_tokens, cand_tokens = tokenize(norm_claim), tokenize(norm_cand)
lcs_ratio = _lcs_length(claim_tokens, cand_tokens) / len(claim_tokens) if claim_tokens else 0.0
edit_sim = _edit_similarity(norm_claim, norm_cand)
claim_chars, cand_chars = set(norm_claim.replace(" ", "")), set(norm_cand.replace(" ", ""))
char_overlap = len(claim_chars & cand_chars) / len(claim_chars | cand_chars) if (claim_chars or cand_chars) else 0.0
claim_flat, cand_flat = norm_claim.replace(" ", ""), norm_cand.replace(" ", "")
is_substring = int(bool(claim_flat) and bool(cand_flat) and (claim_flat in cand_flat or cand_flat in claim_flat))
diacritic_sim = (
_edit_similarity(_light_normalize(claim), _light_normalize(candidate), max_len=800) if is_quran else edit_sim
)
short = len(claim_tokens) < 4
if is_quran:
w = (
dict(coverage=0.35, diacritic_sim=0.30, lcs_ratio=0.15, token_overlap=0.10, char_overlap=0.05, edit_sim=0.05)
if short
else dict(coverage=0.25, diacritic_sim=0.30, lcs_ratio=0.20, token_overlap=0.10, char_overlap=0.05, edit_sim=0.10)
)
composite = (
w["coverage"] * coverage + w["diacritic_sim"] * diacritic_sim + w["lcs_ratio"] * lcs_ratio
+ w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
)
if is_substring and diacritic_sim >= 0.60:
composite = max(composite, 0.88)
elif is_substring:
composite = max(composite, 0.75)
else:
w = (
dict(coverage=0.40, lcs_ratio=0.20, token_overlap=0.20, char_overlap=0.10, edit_sim=0.10)
if short
else dict(coverage=0.30, lcs_ratio=0.28, token_overlap=0.18, char_overlap=0.12, edit_sim=0.12)
)
composite = (
w["coverage"] * coverage + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap
+ w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim
)
if is_substring:
composite = max(composite, 0.82)
return {
"token_overlap": round(token_overlap, 4),
"coverage": round(coverage, 4),
"lcs_ratio": round(lcs_ratio, 4),
"edit_sim": round(edit_sim, 4),
"diacritic_sim": round(diacritic_sim, 4),
"char_overlap": round(char_overlap, 4),
"is_substring": is_substring,
"composite": round(composite, 4),
}
def best_match_score(claim: str, candidates: List[dict], content_type: str = "Ayah"):
"""Return ``(score, candidate_with_signals)`` for the best-scoring candidate."""
best_score, best_candidate = 0.0, None
for candidate in candidates:
signals = compute_signals(claim, candidate.get("text", ""), content_type)
if signals["composite"] > best_score:
best_score, best_candidate = signals["composite"], {**candidate, "signals": signals}
return best_score, best_candidate
# --------------------------------------------------------------------------------------------------------------
# Detection (Subtask 1A)
# --------------------------------------------------------------------------------------------------------------
@dataclass
class DetectedSpan:
start: int
end: int # exclusive
label: str # 'Ayah' | 'Hadith'
confidence: Optional[float] # None for the rule backend
source: str # 'bert' | 'rules' | 'given'
text: str = ""
QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]"
def trim_span(text: str, start: int, end: int):
"""Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
while start < end and text[start] in QUOTE_CHARS:
start += 1
while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:":
end -= 1
return start, end
AYAH_TRIGGERS = [
"قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه",
"سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى",
"ذكر الله", "﴿",
]
HADITH_TRIGGERS = [
"رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه",
"الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم",
]
_FORMULA_WORDS = {
normalize_for_matching(w)
for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split()
}
_BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")]
class RuleDetector:
"""Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
def __init__(self, retriever: Optional[SourceRetriever] = None, min_words: int = 3, context_chars: int = 110,
min_corpus_cov: float = 0.6) -> None:
self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
@staticmethod
def _segments(text: str):
segments = []
quote_positions = [m.start() for m in re.finditer('"', text)]
if len(quote_positions) % 2 == 0:
pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
else: # a stray quote: fall back to every consecutive pair
pairs = zip(quote_positions, quote_positions[1:])
for a, b in pairs:
segments.append((a + 1, b))
for opener, closer in _BRACKET_PAIRS:
for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
segments.append((m.start(1), m.end(1)))
return segments
@staticmethod
def _trigger_type(context: str) -> Optional[str]:
best_end, best_label = -1, None
for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
for trigger in triggers:
pos = context.rfind(trigger)
if pos >= 0 and pos + len(trigger) > best_end:
best_end, best_label = pos + len(trigger), label
return best_label
def _corpus_coverage(self, span: str):
"""Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
if self.kb is None:
return 0.0, 0.0
quran_words = set(tokenize(normalize_strict(span)))
if not quran_words:
return 0.0, 0.0
quran_cov = max(
(len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words)
for c in self.kb.search_quran_ayahs(span, top_k=5)),
default=0.0,
)
hadith_words = set(tokenize(normalize_lenient(span)))
hadith_cov = max(
(len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
for c in self.kb.search_hadith(span, top_k=5)),
default=0.0,
) if hadith_words else 0.0
return quran_cov, hadith_cov
def detect(self, text: str) -> List[DetectedSpan]:
candidates = []
for start, end in self._segments(text):
start, end = trim_span(text, start, end)
if end <= start:
continue
inner = text[start:end]
words = [w for w in normalize_for_matching(inner).split() if w]
if len(words) < self.min_words or len(inner) > 3000:
continue
if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
continue
trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
quran_cov, hadith_cov = self._corpus_coverage(inner)
label = None
if trigger:
label = trigger
other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
if other >= 0.8 and mine < 0.5:
label = "Hadith" if trigger == "Ayah" else "Ayah"
elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
if label is None:
continue
candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label))
candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
taken = []
for _, _, _, start, end, label in candidates:
if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken):
taken.append((start, end, label))
taken.sort()
return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken]
LABEL2ID = {"O": 0, "B-Ayah": 1, "I-Ayah": 2, "B-Hadith": 3, "I-Hadith": 4}
ID2LABEL = {v: k for k, v in LABEL2ID.items()}
def _token_labels_to_char_spans(offsets, token_labels):
spans, current = [], None
for (start, end), label_id in zip(offsets, token_labels):
if label_id == -100:
continue
name = ID2LABEL[label_id]
if name == "O":
if current is not None:
spans.append(current)
current = None
continue
prefix, entity = name.split("-")
if prefix == "B" or current is None or current["label"] != entity:
if current is not None:
spans.append(current)
current = {"label": entity, "start": start, "end": end}
else:
current["end"] = end
if current is not None:
spans.append(current)
return spans
def _merge_adjacent_spans(spans, max_gap: int = 1):
if not spans:
return []
spans = sorted(spans, key=lambda s: s["start"])
merged = [dict(spans[0])]
for span in spans[1:]:
last = merged[-1]
if span["label"] == last["label"] and 0 <= span["start"] - last["end"] <= max_gap:
last["end"] = max(last["end"], span["end"])
else:
merged.append(dict(span))
return merged
class BertDetector:
"""Fine-tuned token classifier (BIO tags, sliding window). Requires ``torch`` and ``transformers`` plus a
trained model directory; training code lives in the research notebook."""
def __init__(self, model_dir: str, device: Optional[str] = None, max_length: int = 512, stride: int = 256) -> None:
import torch
from transformers import AutoModelForTokenClassification, AutoTokenizer
self.torch = torch
self.device = torch.device(device or ("cuda" if torch.cuda.is_available() else "cpu"))
self.tokenizer = AutoTokenizer.from_pretrained(model_dir)
self.model = AutoModelForTokenClassification.from_pretrained(model_dir).to(self.device).eval()
self.max_length, self.stride = max_length, stride
def detect(self, text: str) -> List[DetectedSpan]:
torch = self.torch
if not text.strip():
return []
encoded = self.tokenizer(text, return_offsets_mapping=True, truncation=False, return_tensors="pt")
ids = encoded["input_ids"][0]
offsets = encoded["offset_mapping"][0].tolist()
total, n_labels = len(ids), len(LABEL2ID)
logits_sum, counts = torch.zeros(total, n_labels), torch.zeros(total)
start = 0
with torch.no_grad():
while start < total:
end = min(start + self.max_length, total)
window = ids[start:end].unsqueeze(0).to(self.device)
logits = self.model(input_ids=window, attention_mask=torch.ones_like(window)).logits[0].cpu()
logits_sum[start:end] += logits
counts[start:end] += 1
if end == total:
break
start += self.stride
logits_sum /= counts.unsqueeze(1).clamp(min=1)
probs = torch.softmax(logits_sum, dim=-1)
predictions = torch.argmax(logits_sum, dim=-1).tolist()
predictions = [(-100 if a == b else p) for p, (a, b) in zip(predictions, offsets)] # skip special tokens
spans = _merge_adjacent_spans(_token_labels_to_char_spans(offsets, predictions), max_gap=1)
detected = []
for span in spans:
start_c, end_c = trim_span(text, span["start"], span["end"])
if end_c <= start_c:
continue
token_idx = [i for i, (x, y) in enumerate(offsets) if y > x and x >= span["start"] and y <= span["end"]]
label = "Ayah" if span["label"] == "Ayah" else "Hadith"
tag_ids = [LABEL2ID[f"B-{label}"], LABEL2ID[f"I-{label}"]]
confidence = float(sum(probs[i, tag_ids].sum() for i in token_idx) / len(token_idx)) if token_idx else None
detected.append(DetectedSpan(start_c, end_c, label, confidence, "bert", text[start_c:end_c]))
return detected
def build_detector(kind: str = "rules", retriever: Optional[SourceRetriever] = None, model_dir: Optional[str] = None,
rules_use_corpus: bool = False):
"""``kind``: 'rules' | 'bert' | 'auto' (BERT if a model directory loads, otherwise rules)."""
if kind == "rules":
return RuleDetector(retriever if rules_use_corpus else None)
if kind == "bert":
return BertDetector(model_dir)
if model_dir:
try:
return BertDetector(model_dir)
except Exception as exc:
logger.warning("Could not load BERT detector from %s (%s); using rule-based detector", model_dir, exc)
return RuleDetector(retriever if rules_use_corpus else None)
# --------------------------------------------------------------------------------------------------------------
# Verification (Subtask 1B)
# --------------------------------------------------------------------------------------------------------------
@dataclass
class Verification:
verdict: str # 'Correct' | 'Incorrect'
confidence: float
best_score: float
method: str
source: Optional[dict] = None # best matching source record (with 'signals')
n_candidates: int = 0
retrieval_top: float = 0.0
class Verifier:
"""Compares a quotation with retrieved candidates and returns a verdict with its supporting evidence."""
def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None:
self.kb = retriever
self.cfg = config or VerifierConfig()
def verify(self, span_text: str, content_type: str) -> Verification:
if not span_text or not span_text.strip():
return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span")
if content_type == "Ayah":
return self._verify_quran(span_text)
if content_type == "Hadith":
return self._verify_hadith(span_text)
return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type")
def _verify_quran(self, span: str) -> Verification:
cfg = self.cfg
candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k)
if not candidates:
return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates")
score, best = best_match_score(span, candidates, "Ayah")
signals = best.get("signals", {}) if best else {}
coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
n = len(candidates)
if is_substring and coverage >= cfg.quran_min_coverage:
return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match")
if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage:
return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass")
if score <= cfg.quran_uncertain_low:
return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail")
strong = sum(
1 for cand in candidates[:10]
if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75
)
if strong >= 2:
return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov")
return self._result("Incorrect", 0.58, score, best, n, "borderline_default")
def _verify_hadith(self, span: str) -> Verification:
cfg = self.cfg
candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k)
if not candidates:
return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates")
top_retrieval = candidates[0].get("retrieval_score", 0.0)
score, best = best_match_score(span, candidates, "Hadith")
signals = best.get("signals", {}) if best else {}
coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0)
n = len(candidates)
if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard:
return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval)
if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage:
return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval)
if score <= cfg.hadith_uncertain_low:
return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval)
moderate = sum(
1 for cand in candidates[:8]
if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55
)
if moderate >= 2 and top_retrieval >= 0.30:
return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval)
if top_retrieval < 0.25 or score < 0.45:
return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval)
return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval)
@staticmethod
def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification:
return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4))
_MARKER = re.compile(r"^\(\d+\)$")
def compare(span_text: str, source_text: str) -> dict:
"""Word-level comparison between the quotation and the source (the 'evidence' view)."""
span_words = span_text.split()
source_words = [w for w in source_text.split() if not _MARKER.match(w)]
span_norm = [normalize_for_matching(w) for w in span_words]
source_norm = [normalize_for_matching(w) for w in source_words]
matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False)
blocks = [b for b in matcher.get_matching_blocks() if b.size > 0]
if blocks and len(source_words) > 2 * len(span_words) + 10: # long source (e.g. Hadith with chain): keep matched region
lo, hi = max(0, blocks[0].b - 3), min(len(source_words), blocks[-1].b + blocks[-1].size + 3)
source_words, source_norm = source_words[lo:hi], source_norm[lo:hi]
matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False)
operations, missing, extra = [], [], []
for tag, i1, i2, j1, j2 in matcher.get_opcodes():
operations.append({"op": tag, "span": " ".join(span_words[i1:i2]), "source": " ".join(source_words[j1:j2])})
if tag in ("delete", "replace"):
extra += span_words[i1:i2]
if tag in ("insert", "replace"):
missing += source_words[j1:j2]
return {
"word_similarity": round(matcher.ratio(), 3),
"word_diff": operations,
"missing_from_span": missing,
"extra_in_span": extra,
"source_excerpt": " ".join(source_words),
}
# --------------------------------------------------------------------------------------------------------------
# Idgham rendering (published mushaf convention used by the Subtask 1C gold corrections)
# --------------------------------------------------------------------------------------------------------------
_SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B"
_TANWEEN = set("\u064B\u064C\u064D")
_IDGHAM_AFTER_NOON = set("نمرل")
_IDGHAM_AFTER_LAM = set("لر")
_DIACRITIC_CHARS = set(
"\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A"
"\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670"
"\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4"
"\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED"
)
def _base_letters(word: str) -> str:
return "".join(c for c in word if c not in _DIACRITIC_CHARS)
def _insert_shadda(word: str) -> str:
return word if not word else word[0] + _SHADDA + word[1:]
def _ends_with_tanween(word: str) -> bool:
if not word:
return False
if word[-1] in _TANWEEN:
return True
return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN
def apply_idgham(text: str, extended: bool = True) -> str:
"""Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda.
Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across
ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention
is inconsistent there."""
words = text.split(" ")
noon_set = _IDGHAM_AFTER_NOON if extended else set("نم")
i = 0
while i < len(words):
word = words[i]
if _MARKER.match(word) or not word:
i += 1
continue
j = i + 1
while j < len(words) and _MARKER.match(words[j]):
j += 1
if j < len(words):
base = _base_letters(words[j])
first = base[0] if base else ""
if word.endswith("\u0646" + _SUKUN) and first in noon_set:
words[i], words[j] = word[:-1], _insert_shadda(words[j])
elif _ends_with_tanween(word) and first in noon_set:
words[j] = _insert_shadda(words[j])
elif word.endswith("\u0645" + _SUKUN) and first == "\u0645":
words[i], words[j] = word[:-1], _insert_shadda(words[j])
elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM:
words[i], words[j] = word[:-1], _insert_shadda(words[j])
i += 1
return " ".join(words)
# --------------------------------------------------------------------------------------------------------------
# Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text
# --------------------------------------------------------------------------------------------------------------
@dataclass
class CorrectionMatch:
kind: str # 'Ayah' | 'Hadith'
strength: float # coverage (Quran) / symmetric containment (Hadith)
full_ratio: float
text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers)
source: dict # reference metadata
display: str = "" # clean human-readable version
class Corrector:
def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None:
self.kb = retriever
self.cfg = config or CorrectorConfig()
def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]:
return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text)
def match_quran(self, query_text: str) -> Optional[CorrectionMatch]:
kb = self.kb
query_norm = normalize_for_matching(query_text)
query_words = content_words(query_norm.split())
if not query_words:
return None
query_len = len(query_norm)
memo: Dict[tuple, tuple] = {}
best = None # (key, coverage, ratio, surah, start, end)
for seed in kb.quran_seed_ayahs(query_words, top_k=25):
surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"]
ayahs = kb.quran_by_surah[surah]
min_ayah, max_ayah = min(ayahs), max(ayahs)
for offset in range(3):
start = ayah - offset
if start < min_ayah:
continue
window_len = -1
for length in range(1, self.cfg.max_window + 1):
end = start + length - 1
if end > max_ayah:
break
window_len += len(kb.q_norm_match[ayahs[end]]) + 1
len_diff = abs(window_len - query_len)
upper_bound = min(1.0, window_len / max(query_len, 1))
if best is not None: # exact-result pruning
best_cov, best_neg = best[0][0], best[0][1]
if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg):
continue
key_pos = (surah, start, end)
if key_pos in memo:
continue
window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1))
matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False)
matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
coverage = matched / max(query_len, 1)
key = (coverage, -len_diff, matcher.ratio())
memo[key_pos] = key
if best is None or key > best[0]:
best = (key, coverage, key[2], surah, start, end)
if best is None:
return None
_, coverage, ratio, surah, start, end = best
display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1))
return CorrectionMatch(
"Ayah", coverage, ratio, self._ayah_text(surah, start, end),
{"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"],
"ayah_start": start, "ayah_end": end},
display,
)
def _ayah_text(self, surah: int, start: int, end: int) -> str:
kb, multi = self.kb, end > start
parts = []
for a in range(start, end + 1):
text = kb.quran[kb.quran_by_surah[surah][a]]["text"]
parts.append(f"{text} ({a})" if multi else text)
return apply_idgham(" ".join(parts)).replace("\u0640", "")
def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]:
kb = self.kb
query_norm = normalize_for_matching(query_text)
query_words = content_words(query_norm.split())
if not query_words:
return None
query_len = len(query_norm)
best = None # (key, idx, field, coverage, candidate_coverage, ratio)
for idx, _ in kb.vote(kb.h_content_index, kb.hadith_idf, query_words, self.cfg.hadith_top_k):
record = kb.hadith[idx]
for field_name in ("norm_matn", "norm_full"):
text = record[field_name]
if not text:
continue
upper_bound = min(1.0, query_len / max(len(text), 1))
if best is not None and upper_bound < best[0][0]:
continue
matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False)
matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4)
coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1)
key = (min(coverage, candidate_cov), matcher.ratio())
if best is None or key > best[0]:
best = (key, idx, field_name, coverage, candidate_cov, key[1])
if best is None:
return None
key, idx, field_name, _, _, ratio = best
record = kb.hadith[idx]
text = (record["matn"] if field_name == "norm_matn" else record["full"]).strip()
return CorrectionMatch(
"Hadith", key[0], ratio, text,
{"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"],
"field": "matn" if field_name == "norm_matn" else "full_text"},
text,
)
def is_exact_ayah(self, span_text: str) -> bool:
"""True if the quote (diacritics-insensitive) is a contiguous piece of 1-5 consecutive ayahs."""
kb, query_norm = self.kb, normalize_for_matching(span_text)
if not query_norm:
return False
for seed in kb.quran_seed_ayahs(content_words(query_norm.split()), 25):
ayah = kb.quran[seed]
ayahs = kb.quran_by_surah[ayah["surah_id"]]
for offset in range(4):
start = ayah["ayah_id"] - offset
for length in range(1, 6):
ids = [ayahs.get(x) for x in range(start, start + length)]
if None in ids:
break
if query_norm in " ".join(kb.q_norm_match[i] for i in ids):
return True
return False
# --------------------------------------------------------------------------------------------------------------
# End-to-end pipeline
# --------------------------------------------------------------------------------------------------------------
STATUS_INFO = {
"VERIFIED": {"ar": "موثّق — النص مطابق للمصدر", "en": "Verified — matches the source", "group": "verified"},
"CORRECTED": {"ar": "مُصحَّح — دليل قوي على نص المصدر", "en": "Mismatch — source-backed correction available", "group": "mismatch"},
"UNSUPPORTED": {"ar": "غير مدعوم — لا يوجد مصدر مطابق في المراجع", "en": "Mismatch — no matching source in the corpus", "group": "mismatch"},
"HUMAN_REVIEW": {"ar": "مراجعة بشرية — الدليل غير كافٍ", "en": "Needs human review — insufficient evidence", "group": "review"},
}
class IslamicContentVerifier:
"""Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review."""
def __init__(self, retriever: Optional[SourceRetriever] = None, detector: str = "rules",
model_dir: Optional[str] = None, config: Optional[PipelineConfig] = None) -> None:
self.cfg = config or PipelineConfig()
self.retriever = retriever or SourceRetriever()
self.verifier = Verifier(self.retriever, self.cfg.verifier)
self.corrector = Corrector(self.retriever, self.cfg.corrector)
self.detector = build_detector(detector, self.retriever, model_dir)
self.detector_name = type(self.detector).__name__
# ---- public API -----------------------------------------------------------------------------------------
def analyze(self, text: str) -> dict:
"""Run the full pipeline on a generated text."""
text = self._validate(text)
started = time.time()
spans = self.detector.detect(text) if text.strip() else []
detect_seconds = time.time() - started
result = self._analyze_spans(text, spans)
result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)}
return result
def analyze_spans(self, text: str, spans: List[dict]) -> dict:
"""Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode)."""
given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans]
return self._analyze_spans(self._validate(text), given)
# ---- internals ------------------------------------------------------------------------------------------
@staticmethod
def _validate(text: str) -> str:
if not isinstance(text, str):
raise TypeError("Input text must be a string")
if len(text) > MAX_INPUT_CHARS:
raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}")
return text
def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict:
reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))]
counts = {status: 0 for status in STATUS_INFO}
for report in reports:
counts[report["status"]] += 1
return {
"input_text": text,
"detector": self.detector_name,
"spans": reports,
"corrected_text": self._apply_corrections(text, reports),
"summary": {
"n_spans": len(reports),
"n_ayah": sum(r["type"] == "Ayah" for r in reports),
"n_hadith": sum(r["type"] == "Hadith" for r in reports),
**counts,
"needs_human_review": counts["HUMAN_REVIEW"] > 0,
},
}
def _process_span(self, index: int, span: DetectedSpan) -> dict:
try:
return self._decide(index, span)
except Exception: # a single failing quotation must not break the whole report
logger.exception("Failed to process span %d", index)
report = self._empty_report(index, span)
self._finalize(report, "HUMAN_REVIEW", "internal_error: this quotation could not be processed automatically")
return report
@staticmethod
def _empty_report(index: int, span: DetectedSpan) -> dict:
return {
"id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text,
"detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)},
"verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0},
"evidence": None, "correction": None, "suggestion": None,
}
@staticmethod
def _finalize(report: dict, status: str, reason: str) -> None:
info = STATUS_INFO[status]
report.update(status=status, status_ar=info["ar"], status_en=info["en"], group=info["group"], reason=reason)
def _decide(self, index: int, span: DetectedSpan) -> dict:
cfg, corr_cfg = self.cfg, self.cfg.corrector
verification = self.verifier.verify(span.text, span.label)
report = self._empty_report(index, span)
report["verification"] = {
"verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score,
"method": verification.method, "n_candidates": verification.n_candidates,
}
report["evidence"] = self._evidence(span, verification)
def proposal(match: Optional[CorrectionMatch]) -> Optional[dict]:
if match is None:
return None
return {
"text": match.text, "display_text": match.display, "source": match.source,
"match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4),
"comparison": compare(span.text, match.display),
}
if verification.verdict == "Correct":
exact = self.corrector.is_exact_ayah(span.text) if span.label == "Ayah" else None
report["verification"]["exact_match"] = exact
if verification.method.startswith("borderline") or verification.confidence < cfg.verified_min_conf:
status, reason = "HUMAN_REVIEW", "weak_verification: matched a source but with low confidence"
elif exact is False:
status = "HUMAN_REVIEW"
reason = "near_match: the quote is close to a source ayah but NOT identical (words missing, added or changed)"
report["suggestion"] = proposal(self.corrector.match(span.text, span.label))
else:
status, reason = "VERIFIED", f"matched source ({verification.method})"
else:
match = self.corrector.match(span.text, span.label)
strong, low = (corr_cfg.quran_strong, corr_cfg.quran_low) if span.label == "Ayah" else (corr_cfg.hadith_strong, corr_cfg.hadith_low)
candidate = proposal(match)
if match is None or match.strength < low:
confident_abstain = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline")
if match is None or match.strength < cfg.unsupported_strength or confident_abstain:
status, reason = "UNSUPPORTED", "no source in the corpus matches this quotation"
else:
status, reason = "HUMAN_REVIEW", "insufficient_evidence: no clear source and low verification confidence"
report["suggestion"] = candidate
elif match.strength >= strong and match.full_ratio >= corr_cfg.min_full_ratio:
status = "CORRECTED"
reason = f"strong match to {self.describe_source(match.source)} (strength {match.strength:.2f})"
report["correction"] = {**candidate, "applied": True}
else:
status = "HUMAN_REVIEW"
reason = (f"candidate source found ({self.describe_source(match.source)}, strength {match.strength:.2f}) "
"but evidence is not strong enough for automatic correction")
report["suggestion"] = candidate
if span.confidence is not None and span.confidence < cfg.min_detection_conf and status != "HUMAN_REVIEW":
reason = f"low detection confidence ({span.confidence:.2f}); was {status}: {reason}"
status = "HUMAN_REVIEW"
if report["correction"]:
report["suggestion"], report["correction"] = {**report["correction"], "applied": False}, None
self._finalize(report, status, reason)
return report
@staticmethod
def _evidence(span: DetectedSpan, verification: Verification) -> Optional[dict]:
source = verification.source
if not source:
return None
if span.label == "Ayah":
reference = {"type": "Quran", "surah_id": source["surah_id"], "surah_name": source["surah_name"], "ayah": source["ayah_id"]}
else:
reference = {"type": "Hadith", "hadithID": source["hadithID"], "book": source["book"], "title": source["title"]}
return {"source": reference, "signals": source.get("signals"), "comparison": compare(span.text, source["text"])}
@staticmethod
def describe_source(source: dict) -> str:
"""Human-readable reference, e.g. ``Quran الفاتحة 1-3`` or ``Hadith #123``."""
if source["type"] == "Quran":
start, end = source["ayah_start"], source["ayah_end"]
return f"Quran {source['surah_name']} {start}" + (f"-{end}" if end != start else "")
return f"Hadith #{source['hadithID']}"
@staticmethod
def _apply_corrections(text: str, reports: List[dict]) -> str:
out = text
for report in sorted(reports, key=lambda r: r["start"], reverse=True):
if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"):
out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:]
return out