Download verifier.py from Ghada-99-Ragab/Islamic-Content-Verifier: direct link, hf CLI and curl.
- Browser
- Download file 45.8 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Islamic-Content-Verifier/resolve/main/verifier.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Islamic-Content-Verifier/verifier.py
-
curl -L -o verifier.py https://huggingface.co/spaces/Ghada-99-Ragab/Islamic-Content-Verifier/resolve/main/verifier.py
45.8 kB
| """Quotation detection, evidence-based verification and source-backed correction for Quran and Hadith. | |
| This module consolidates the logic of the IslamicEval 2025 research notebook | |
| (``research/IslamicEval_Unified.ipynb``) into one pipeline: | |
| input text -> detection (1A) -> source retrieval -> verification (1B) -> evidence | |
| -> strong evidence : source-backed correction (1C) | |
| -> weak / ambiguous : human review | |
| -> no source found : reported as unsupported, never "fixed" | |
| Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated. | |
| """ | |
| from __future__ import annotations | |
| import difflib | |
| import logging | |
| import re | |
| import time | |
| import unicodedata | |
| from dataclasses import dataclass, field | |
| from typing import Dict, List, Optional, Sequence | |
| from retrieval import ( | |
| SourceRetriever, | |
| content_words, | |
| normalize_for_matching, | |
| normalize_lenient, | |
| normalize_strict, | |
| tokenize, | |
| ) | |
| try: # optional accelerator; identical formula (1 - distance / max_len) | |
| from rapidfuzz.distance import Levenshtein as _RapidLevenshtein | |
| except ImportError: # pragma: no cover | |
| _RapidLevenshtein = None | |
| logger = logging.getLogger(__name__) | |
| MAX_INPUT_CHARS = 20_000 | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Configuration (all tunable numbers live here) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class VerifierConfig: | |
| """Thresholds calibrated in Subtask 1B of the research notebook.""" | |
| quran_correct_threshold: float = 0.94 | |
| quran_uncertain_low: float = 0.45 | |
| quran_min_coverage: float = 0.40 | |
| hadith_correct_threshold: float = 0.79 | |
| hadith_uncertain_low: float = 0.30 | |
| hadith_min_coverage: float = 0.70 | |
| quran_top_k: int = 25 | |
| hadith_top_k: int = 15 | |
| hadith_retrieval_guard: float = 0.20 | |
| class CorrectorConfig: | |
| """Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the | |
| case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact | |
| fragment boundaries of a Hadith quotation cannot be reproduced reliably.""" | |
| max_window: int = 8 | |
| hadith_top_k: int = 40 | |
| quran_strong: float = 0.65 | |
| min_full_ratio: float = 0.40 | |
| quran_low: float = 0.55 | |
| hadith_strong: float = 1.01 | |
| hadith_low: float = 0.45 | |
| class PipelineConfig: | |
| verifier: VerifierConfig = field(default_factory=VerifierConfig) | |
| corrector: CorrectorConfig = field(default_factory=CorrectorConfig) | |
| min_detection_conf: float = 0.60 # detector confidence below this -> human review | |
| verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review | |
| unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently | |
| unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Similarity signals (Subtask 1B) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| def _lcs_length(a: Sequence[str], b: Sequence[str]) -> int: | |
| m, n = len(a), len(b) | |
| if m == 0 or n == 0: | |
| return 0 | |
| if m < n: | |
| a, b, m, n = b, a, n, m | |
| prev = [0] * (n + 1) | |
| for i in range(m): | |
| curr = [0] * (n + 1) | |
| for j in range(n): | |
| curr[j + 1] = prev[j] + 1 if a[i] == b[j] else max(curr[j], prev[j + 1]) | |
| prev = curr | |
| return prev[n] | |
| def _edit_similarity(a: str, b: str, max_len: int = 600) -> float: | |
| """Normalised Levenshtein similarity in [0, 1] on the first ``max_len`` characters.""" | |
| a, b = a[:max_len], b[:max_len] | |
| if a == b: | |
| return 1.0 | |
| if not a or not b: | |
| return 0.0 | |
| if _RapidLevenshtein is not None: | |
| return float(_RapidLevenshtein.normalized_similarity(a, b)) | |
| prev = list(range(len(b) + 1)) | |
| for i, ca in enumerate(a): | |
| curr = [i + 1] | |
| for j, cb in enumerate(b): | |
| curr.append(min(curr[j] + 1, prev[j + 1] + 1, prev[j] + (ca != cb))) | |
| prev = curr | |
| return 1.0 - prev[-1] / max(len(a), len(b)) | |
| def _light_normalize(text: str) -> str: | |
| """Keeps diacritics (so diacritic changes lower the score) but drops punctuation and tatweel.""" | |
| text = unicodedata.normalize("NFC", text) | |
| text = re.sub(r"[،؛؟!.,:;'\"()\[\]{}<>«»\-_/\\|@#$%^&*+=~`\u0640]", " ", text) | |
| return re.sub(r"\s+", " ", text).strip() | |
| def compute_signals(claim: str, candidate: str, content_type: str = "Ayah") -> Dict[str, float]: | |
| """Word-, character- and sequence-level similarity indicators between a quotation and a candidate source.""" | |
| is_quran = content_type == "Ayah" | |
| normalize = normalize_strict if is_quran else normalize_lenient | |
| norm_claim, norm_cand = normalize(claim), normalize(candidate) | |
| claim_set, cand_set = set(tokenize(norm_claim)), set(tokenize(norm_cand)) | |
| if claim_set and cand_set: | |
| shared = claim_set & cand_set | |
| token_overlap = len(shared) / len(claim_set | cand_set) | |
| coverage = len(shared) / len(claim_set) | |
| else: | |
| token_overlap = coverage = 0.0 | |
| claim_tokens, cand_tokens = tokenize(norm_claim), tokenize(norm_cand) | |
| lcs_ratio = _lcs_length(claim_tokens, cand_tokens) / len(claim_tokens) if claim_tokens else 0.0 | |
| edit_sim = _edit_similarity(norm_claim, norm_cand) | |
| claim_chars, cand_chars = set(norm_claim.replace(" ", "")), set(norm_cand.replace(" ", "")) | |
| char_overlap = len(claim_chars & cand_chars) / len(claim_chars | cand_chars) if (claim_chars or cand_chars) else 0.0 | |
| claim_flat, cand_flat = norm_claim.replace(" ", ""), norm_cand.replace(" ", "") | |
| is_substring = int(bool(claim_flat) and bool(cand_flat) and (claim_flat in cand_flat or cand_flat in claim_flat)) | |
| diacritic_sim = ( | |
| _edit_similarity(_light_normalize(claim), _light_normalize(candidate), max_len=800) if is_quran else edit_sim | |
| ) | |
| short = len(claim_tokens) < 4 | |
| if is_quran: | |
| w = ( | |
| dict(coverage=0.35, diacritic_sim=0.30, lcs_ratio=0.15, token_overlap=0.10, char_overlap=0.05, edit_sim=0.05) | |
| if short | |
| else dict(coverage=0.25, diacritic_sim=0.30, lcs_ratio=0.20, token_overlap=0.10, char_overlap=0.05, edit_sim=0.10) | |
| ) | |
| composite = ( | |
| w["coverage"] * coverage + w["diacritic_sim"] * diacritic_sim + w["lcs_ratio"] * lcs_ratio | |
| + w["token_overlap"] * token_overlap + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim | |
| ) | |
| if is_substring and diacritic_sim >= 0.60: | |
| composite = max(composite, 0.88) | |
| elif is_substring: | |
| composite = max(composite, 0.75) | |
| else: | |
| w = ( | |
| dict(coverage=0.40, lcs_ratio=0.20, token_overlap=0.20, char_overlap=0.10, edit_sim=0.10) | |
| if short | |
| else dict(coverage=0.30, lcs_ratio=0.28, token_overlap=0.18, char_overlap=0.12, edit_sim=0.12) | |
| ) | |
| composite = ( | |
| w["coverage"] * coverage + w["lcs_ratio"] * lcs_ratio + w["token_overlap"] * token_overlap | |
| + w["char_overlap"] * char_overlap + w["edit_sim"] * edit_sim | |
| ) | |
| if is_substring: | |
| composite = max(composite, 0.82) | |
| return { | |
| "token_overlap": round(token_overlap, 4), | |
| "coverage": round(coverage, 4), | |
| "lcs_ratio": round(lcs_ratio, 4), | |
| "edit_sim": round(edit_sim, 4), | |
| "diacritic_sim": round(diacritic_sim, 4), | |
| "char_overlap": round(char_overlap, 4), | |
| "is_substring": is_substring, | |
| "composite": round(composite, 4), | |
| } | |
| def best_match_score(claim: str, candidates: List[dict], content_type: str = "Ayah"): | |
| """Return ``(score, candidate_with_signals)`` for the best-scoring candidate.""" | |
| best_score, best_candidate = 0.0, None | |
| for candidate in candidates: | |
| signals = compute_signals(claim, candidate.get("text", ""), content_type) | |
| if signals["composite"] > best_score: | |
| best_score, best_candidate = signals["composite"], {**candidate, "signals": signals} | |
| return best_score, best_candidate | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Detection (Subtask 1A) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class DetectedSpan: | |
| start: int | |
| end: int # exclusive | |
| label: str # 'Ayah' | 'Hadith' | |
| confidence: Optional[float] # None for the rule backend | |
| source: str # 'bert' | 'rules' | 'given' | |
| text: str = "" | |
| QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]" | |
| def trim_span(text: str, start: int, end: int): | |
| """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks).""" | |
| while start < end and text[start] in QUOTE_CHARS: | |
| start += 1 | |
| while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:": | |
| end -= 1 | |
| return start, end | |
| AYAH_TRIGGERS = [ | |
| "قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه", | |
| "سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى", | |
| "ذكر الله", "﴿", | |
| ] | |
| HADITH_TRIGGERS = [ | |
| "رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه", | |
| "الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم", | |
| ] | |
| _FORMULA_WORDS = { | |
| normalize_for_matching(w) | |
| for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split() | |
| } | |
| _BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")] | |
| class RuleDetector: | |
| """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU).""" | |
| def __init__(self, retriever: Optional[SourceRetriever] = None, min_words: int = 3, context_chars: int = 110, | |
| min_corpus_cov: float = 0.6) -> None: | |
| self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov | |
| def _segments(text: str): | |
| segments = [] | |
| quote_positions = [m.start() for m in re.finditer('"', text)] | |
| if len(quote_positions) % 2 == 0: | |
| pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs | |
| else: # a stray quote: fall back to every consecutive pair | |
| pairs = zip(quote_positions, quote_positions[1:]) | |
| for a, b in pairs: | |
| segments.append((a + 1, b)) | |
| for opener, closer in _BRACKET_PAIRS: | |
| for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S): | |
| segments.append((m.start(1), m.end(1))) | |
| return segments | |
| def _trigger_type(context: str) -> Optional[str]: | |
| best_end, best_label = -1, None | |
| for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)): | |
| for trigger in triggers: | |
| pos = context.rfind(trigger) | |
| if pos >= 0 and pos + len(trigger) > best_end: | |
| best_end, best_label = pos + len(trigger), label | |
| return best_label | |
| def _corpus_coverage(self, span: str): | |
| """Highest word coverage of the span by any top Quran ayah / Hadith candidate.""" | |
| if self.kb is None: | |
| return 0.0, 0.0 | |
| quran_words = set(tokenize(normalize_strict(span))) | |
| if not quran_words: | |
| return 0.0, 0.0 | |
| quran_cov = max( | |
| (len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words) | |
| for c in self.kb.search_quran_ayahs(span, top_k=5)), | |
| default=0.0, | |
| ) | |
| hadith_words = set(tokenize(normalize_lenient(span))) | |
| hadith_cov = max( | |
| (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words) | |
| for c in self.kb.search_hadith(span, top_k=5)), | |
| default=0.0, | |
| ) if hadith_words else 0.0 | |
| return quran_cov, hadith_cov | |
| def detect(self, text: str) -> List[DetectedSpan]: | |
| candidates = [] | |
| for start, end in self._segments(text): | |
| start, end = trim_span(text, start, end) | |
| if end <= start: | |
| continue | |
| inner = text[start:end] | |
| words = [w for w in normalize_for_matching(inner).split() if w] | |
| if len(words) < self.min_words or len(inner) > 3000: | |
| continue | |
| if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6: | |
| continue | |
| trigger = self._trigger_type(text[max(0, start - self.context_chars):start]) | |
| quran_cov, hadith_cov = self._corpus_coverage(inner) | |
| label = None | |
| if trigger: | |
| label = trigger | |
| other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov) | |
| if other >= 0.8 and mine < 0.5: | |
| label = "Hadith" if trigger == "Ayah" else "Ayah" | |
| elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4: | |
| label = "Ayah" if quran_cov >= hadith_cov else "Hadith" | |
| if label is None: | |
| continue | |
| candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label)) | |
| candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length | |
| taken = [] | |
| for _, _, _, start, end, label in candidates: | |
| if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken): | |
| taken.append((start, end, label)) | |
| taken.sort() | |
| return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken] | |
| LABEL2ID = {"O": 0, "B-Ayah": 1, "I-Ayah": 2, "B-Hadith": 3, "I-Hadith": 4} | |
| ID2LABEL = {v: k for k, v in LABEL2ID.items()} | |
| def _token_labels_to_char_spans(offsets, token_labels): | |
| spans, current = [], None | |
| for (start, end), label_id in zip(offsets, token_labels): | |
| if label_id == -100: | |
| continue | |
| name = ID2LABEL[label_id] | |
| if name == "O": | |
| if current is not None: | |
| spans.append(current) | |
| current = None | |
| continue | |
| prefix, entity = name.split("-") | |
| if prefix == "B" or current is None or current["label"] != entity: | |
| if current is not None: | |
| spans.append(current) | |
| current = {"label": entity, "start": start, "end": end} | |
| else: | |
| current["end"] = end | |
| if current is not None: | |
| spans.append(current) | |
| return spans | |
| def _merge_adjacent_spans(spans, max_gap: int = 1): | |
| if not spans: | |
| return [] | |
| spans = sorted(spans, key=lambda s: s["start"]) | |
| merged = [dict(spans[0])] | |
| for span in spans[1:]: | |
| last = merged[-1] | |
| if span["label"] == last["label"] and 0 <= span["start"] - last["end"] <= max_gap: | |
| last["end"] = max(last["end"], span["end"]) | |
| else: | |
| merged.append(dict(span)) | |
| return merged | |
| class BertDetector: | |
| """Fine-tuned token classifier (BIO tags, sliding window). Requires ``torch`` and ``transformers`` plus a | |
| trained model directory; training code lives in the research notebook.""" | |
| def __init__(self, model_dir: str, device: Optional[str] = None, max_length: int = 512, stride: int = 256) -> None: | |
| import torch | |
| from transformers import AutoModelForTokenClassification, AutoTokenizer | |
| self.torch = torch | |
| self.device = torch.device(device or ("cuda" if torch.cuda.is_available() else "cpu")) | |
| self.tokenizer = AutoTokenizer.from_pretrained(model_dir) | |
| self.model = AutoModelForTokenClassification.from_pretrained(model_dir).to(self.device).eval() | |
| self.max_length, self.stride = max_length, stride | |
| def detect(self, text: str) -> List[DetectedSpan]: | |
| torch = self.torch | |
| if not text.strip(): | |
| return [] | |
| encoded = self.tokenizer(text, return_offsets_mapping=True, truncation=False, return_tensors="pt") | |
| ids = encoded["input_ids"][0] | |
| offsets = encoded["offset_mapping"][0].tolist() | |
| total, n_labels = len(ids), len(LABEL2ID) | |
| logits_sum, counts = torch.zeros(total, n_labels), torch.zeros(total) | |
| start = 0 | |
| with torch.no_grad(): | |
| while start < total: | |
| end = min(start + self.max_length, total) | |
| window = ids[start:end].unsqueeze(0).to(self.device) | |
| logits = self.model(input_ids=window, attention_mask=torch.ones_like(window)).logits[0].cpu() | |
| logits_sum[start:end] += logits | |
| counts[start:end] += 1 | |
| if end == total: | |
| break | |
| start += self.stride | |
| logits_sum /= counts.unsqueeze(1).clamp(min=1) | |
| probs = torch.softmax(logits_sum, dim=-1) | |
| predictions = torch.argmax(logits_sum, dim=-1).tolist() | |
| predictions = [(-100 if a == b else p) for p, (a, b) in zip(predictions, offsets)] # skip special tokens | |
| spans = _merge_adjacent_spans(_token_labels_to_char_spans(offsets, predictions), max_gap=1) | |
| detected = [] | |
| for span in spans: | |
| start_c, end_c = trim_span(text, span["start"], span["end"]) | |
| if end_c <= start_c: | |
| continue | |
| token_idx = [i for i, (x, y) in enumerate(offsets) if y > x and x >= span["start"] and y <= span["end"]] | |
| label = "Ayah" if span["label"] == "Ayah" else "Hadith" | |
| tag_ids = [LABEL2ID[f"B-{label}"], LABEL2ID[f"I-{label}"]] | |
| confidence = float(sum(probs[i, tag_ids].sum() for i in token_idx) / len(token_idx)) if token_idx else None | |
| detected.append(DetectedSpan(start_c, end_c, label, confidence, "bert", text[start_c:end_c])) | |
| return detected | |
| def build_detector(kind: str = "rules", retriever: Optional[SourceRetriever] = None, model_dir: Optional[str] = None, | |
| rules_use_corpus: bool = False): | |
| """``kind``: 'rules' | 'bert' | 'auto' (BERT if a model directory loads, otherwise rules).""" | |
| if kind == "rules": | |
| return RuleDetector(retriever if rules_use_corpus else None) | |
| if kind == "bert": | |
| return BertDetector(model_dir) | |
| if model_dir: | |
| try: | |
| return BertDetector(model_dir) | |
| except Exception as exc: | |
| logger.warning("Could not load BERT detector from %s (%s); using rule-based detector", model_dir, exc) | |
| return RuleDetector(retriever if rules_use_corpus else None) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Verification (Subtask 1B) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class Verification: | |
| verdict: str # 'Correct' | 'Incorrect' | |
| confidence: float | |
| best_score: float | |
| method: str | |
| source: Optional[dict] = None # best matching source record (with 'signals') | |
| n_candidates: int = 0 | |
| retrieval_top: float = 0.0 | |
| class Verifier: | |
| """Compares a quotation with retrieved candidates and returns a verdict with its supporting evidence.""" | |
| def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None: | |
| self.kb = retriever | |
| self.cfg = config or VerifierConfig() | |
| def verify(self, span_text: str, content_type: str) -> Verification: | |
| if not span_text or not span_text.strip(): | |
| return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span") | |
| if content_type == "Ayah": | |
| return self._verify_quran(span_text) | |
| if content_type == "Hadith": | |
| return self._verify_hadith(span_text) | |
| return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type") | |
| def _verify_quran(self, span: str) -> Verification: | |
| cfg = self.cfg | |
| candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k) | |
| if not candidates: | |
| return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates") | |
| score, best = best_match_score(span, candidates, "Ayah") | |
| signals = best.get("signals", {}) if best else {} | |
| coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) | |
| n = len(candidates) | |
| if is_substring and coverage >= cfg.quran_min_coverage: | |
| return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match") | |
| if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage: | |
| return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass") | |
| if score <= cfg.quran_uncertain_low: | |
| return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail") | |
| strong = sum( | |
| 1 for cand in candidates[:10] | |
| if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75 | |
| ) | |
| if strong >= 2: | |
| return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov") | |
| return self._result("Incorrect", 0.58, score, best, n, "borderline_default") | |
| def _verify_hadith(self, span: str) -> Verification: | |
| cfg = self.cfg | |
| candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k) | |
| if not candidates: | |
| return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates") | |
| top_retrieval = candidates[0].get("retrieval_score", 0.0) | |
| score, best = best_match_score(span, candidates, "Hadith") | |
| signals = best.get("signals", {}) if best else {} | |
| coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) | |
| n = len(candidates) | |
| if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard: | |
| return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval) | |
| if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage: | |
| return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval) | |
| if score <= cfg.hadith_uncertain_low: | |
| return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval) | |
| moderate = sum( | |
| 1 for cand in candidates[:8] | |
| if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55 | |
| ) | |
| if moderate >= 2 and top_retrieval >= 0.30: | |
| return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval) | |
| if top_retrieval < 0.25 or score < 0.45: | |
| return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval) | |
| return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval) | |
| def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification: | |
| return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4)) | |
| _MARKER = re.compile(r"^\(\d+\)$") | |
| def compare(span_text: str, source_text: str) -> dict: | |
| """Word-level comparison between the quotation and the source (the 'evidence' view).""" | |
| span_words = span_text.split() | |
| source_words = [w for w in source_text.split() if not _MARKER.match(w)] | |
| span_norm = [normalize_for_matching(w) for w in span_words] | |
| source_norm = [normalize_for_matching(w) for w in source_words] | |
| matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False) | |
| blocks = [b for b in matcher.get_matching_blocks() if b.size > 0] | |
| if blocks and len(source_words) > 2 * len(span_words) + 10: # long source (e.g. Hadith with chain): keep matched region | |
| lo, hi = max(0, blocks[0].b - 3), min(len(source_words), blocks[-1].b + blocks[-1].size + 3) | |
| source_words, source_norm = source_words[lo:hi], source_norm[lo:hi] | |
| matcher = difflib.SequenceMatcher(None, span_norm, source_norm, autojunk=False) | |
| operations, missing, extra = [], [], [] | |
| for tag, i1, i2, j1, j2 in matcher.get_opcodes(): | |
| operations.append({"op": tag, "span": " ".join(span_words[i1:i2]), "source": " ".join(source_words[j1:j2])}) | |
| if tag in ("delete", "replace"): | |
| extra += span_words[i1:i2] | |
| if tag in ("insert", "replace"): | |
| missing += source_words[j1:j2] | |
| return { | |
| "word_similarity": round(matcher.ratio(), 3), | |
| "word_diff": operations, | |
| "missing_from_span": missing, | |
| "extra_in_span": extra, | |
| "source_excerpt": " ".join(source_words), | |
| } | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Idgham rendering (published mushaf convention used by the Subtask 1C gold corrections) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| _SUKUN, _SHADDA, _FATHATAN = "\u0652", "\u0651", "\u064B" | |
| _TANWEEN = set("\u064B\u064C\u064D") | |
| _IDGHAM_AFTER_NOON = set("نمرل") | |
| _IDGHAM_AFTER_LAM = set("لر") | |
| _DIACRITIC_CHARS = set( | |
| "\u0610\u0611\u0612\u0613\u0614\u0615\u0616\u0617\u0618\u0619\u061A" | |
| "\u064B\u064C\u064D\u064E\u064F\u0650\u0651\u0652\u0670" | |
| "\u06D6\u06D7\u06D8\u06D9\u06DA\u06DB\u06DC\u06DF\u06E0\u06E1\u06E2\u06E3\u06E4" | |
| "\u06E7\u06E8\u06EA\u06EB\u06EC\u06ED" | |
| ) | |
| def _base_letters(word: str) -> str: | |
| return "".join(c for c in word if c not in _DIACRITIC_CHARS) | |
| def _insert_shadda(word: str) -> str: | |
| return word if not word else word[0] + _SHADDA + word[1:] | |
| def _ends_with_tanween(word: str) -> bool: | |
| if not word: | |
| return False | |
| if word[-1] in _TANWEEN: | |
| return True | |
| return len(word) >= 2 and word[-1] in "اى" and word[-2] == _FATHATAN | |
| def apply_idgham(text: str, extended: bool = True) -> str: | |
| """Convert the flat Quran text into the mushaf rendering that marks assimilation with a shadda. | |
| Covers noon sakinah / tanween before ن م ر ل, meem sakinah before م and lam sakinah before ل ر, also across | |
| ayah-number markers such as ``(20)``. The و / ي cases are excluded on purpose because the mushaf convention | |
| is inconsistent there.""" | |
| words = text.split(" ") | |
| noon_set = _IDGHAM_AFTER_NOON if extended else set("نم") | |
| i = 0 | |
| while i < len(words): | |
| word = words[i] | |
| if _MARKER.match(word) or not word: | |
| i += 1 | |
| continue | |
| j = i + 1 | |
| while j < len(words) and _MARKER.match(words[j]): | |
| j += 1 | |
| if j < len(words): | |
| base = _base_letters(words[j]) | |
| first = base[0] if base else "" | |
| if word.endswith("\u0646" + _SUKUN) and first in noon_set: | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| elif _ends_with_tanween(word) and first in noon_set: | |
| words[j] = _insert_shadda(words[j]) | |
| elif word.endswith("\u0645" + _SUKUN) and first == "\u0645": | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| elif extended and word.endswith("\u0644" + _SUKUN) and first in _IDGHAM_AFTER_LAM: | |
| words[i], words[j] = word[:-1], _insert_shadda(words[j]) | |
| i += 1 | |
| return " ".join(words) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class CorrectionMatch: | |
| kind: str # 'Ayah' | 'Hadith' | |
| strength: float # coverage (Quran) / symmetric containment (Hadith) | |
| full_ratio: float | |
| text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers) | |
| source: dict # reference metadata | |
| display: str = "" # clean human-readable version | |
| class Corrector: | |
| def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None: | |
| self.kb = retriever | |
| self.cfg = config or CorrectorConfig() | |
| def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]: | |
| return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text) | |
| def match_quran(self, query_text: str) -> Optional[CorrectionMatch]: | |
| kb = self.kb | |
| query_norm = normalize_for_matching(query_text) | |
| query_words = content_words(query_norm.split()) | |
| if not query_words: | |
| return None | |
| query_len = len(query_norm) | |
| memo: Dict[tuple, tuple] = {} | |
| best = None # (key, coverage, ratio, surah, start, end) | |
| for seed in kb.quran_seed_ayahs(query_words, top_k=25): | |
| surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"] | |
| ayahs = kb.quran_by_surah[surah] | |
| min_ayah, max_ayah = min(ayahs), max(ayahs) | |
| for offset in range(3): | |
| start = ayah - offset | |
| if start < min_ayah: | |
| continue | |
| window_len = -1 | |
| for length in range(1, self.cfg.max_window + 1): | |
| end = start + length - 1 | |
| if end > max_ayah: | |
| break | |
| window_len += len(kb.q_norm_match[ayahs[end]]) + 1 | |
| len_diff = abs(window_len - query_len) | |
| upper_bound = min(1.0, window_len / max(query_len, 1)) | |
| if best is not None: # exact-result pruning | |
| best_cov, best_neg = best[0][0], best[0][1] | |
| if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg): | |
| continue | |
| key_pos = (surah, start, end) | |
| if key_pos in memo: | |
| continue | |
| window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1)) | |
| matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False) | |
| matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) | |
| coverage = matched / max(query_len, 1) | |
| key = (coverage, -len_diff, matcher.ratio()) | |
| memo[key_pos] = key | |
| if best is None or key > best[0]: | |
| best = (key, coverage, key[2], surah, start, end) | |
| if best is None: | |
| return None | |
| _, coverage, ratio, surah, start, end = best | |
| display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1)) | |
| return CorrectionMatch( | |
| "Ayah", coverage, ratio, self._ayah_text(surah, start, end), | |
| {"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"], | |
| "ayah_start": start, "ayah_end": end}, | |
| display, | |
| ) | |
| def _ayah_text(self, surah: int, start: int, end: int) -> str: | |
| kb, multi = self.kb, end > start | |
| parts = [] | |
| for a in range(start, end + 1): | |
| text = kb.quran[kb.quran_by_surah[surah][a]]["text"] | |
| parts.append(f"{text} ({a})" if multi else text) | |
| return apply_idgham(" ".join(parts)).replace("\u0640", "") | |
| def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]: | |
| kb = self.kb | |
| query_norm = normalize_for_matching(query_text) | |
| query_words = content_words(query_norm.split()) | |
| if not query_words: | |
| return None | |
| query_len = len(query_norm) | |
| best = None # (key, idx, field, coverage, candidate_coverage, ratio) | |
| for idx, _ in kb.vote(kb.h_content_index, kb.hadith_idf, query_words, self.cfg.hadith_top_k): | |
| record = kb.hadith[idx] | |
| for field_name in ("norm_matn", "norm_full"): | |
| text = record[field_name] | |
| if not text: | |
| continue | |
| upper_bound = min(1.0, query_len / max(len(text), 1)) | |
| if best is not None and upper_bound < best[0][0]: | |
| continue | |
| matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False) | |
| matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) | |
| coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1) | |
| key = (min(coverage, candidate_cov), matcher.ratio()) | |
| if best is None or key > best[0]: | |
| best = (key, idx, field_name, coverage, candidate_cov, key[1]) | |
| if best is None: | |
| return None | |
| key, idx, field_name, _, _, ratio = best | |
| record = kb.hadith[idx] | |
| text = (record["matn"] if field_name == "norm_matn" else record["full"]).strip() | |
| return CorrectionMatch( | |
| "Hadith", key[0], ratio, text, | |
| {"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"], | |
| "field": "matn" if field_name == "norm_matn" else "full_text"}, | |
| text, | |
| ) | |
| def is_exact_ayah(self, span_text: str) -> bool: | |
| """True if the quote (diacritics-insensitive) is a contiguous piece of 1-5 consecutive ayahs.""" | |
| kb, query_norm = self.kb, normalize_for_matching(span_text) | |
| if not query_norm: | |
| return False | |
| for seed in kb.quran_seed_ayahs(content_words(query_norm.split()), 25): | |
| ayah = kb.quran[seed] | |
| ayahs = kb.quran_by_surah[ayah["surah_id"]] | |
| for offset in range(4): | |
| start = ayah["ayah_id"] - offset | |
| for length in range(1, 6): | |
| ids = [ayahs.get(x) for x in range(start, start + length)] | |
| if None in ids: | |
| break | |
| if query_norm in " ".join(kb.q_norm_match[i] for i in ids): | |
| return True | |
| return False | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # End-to-end pipeline | |
| # -------------------------------------------------------------------------------------------------------------- | |
| STATUS_INFO = { | |
| "VERIFIED": {"ar": "موثّق — النص مطابق للمصدر", "en": "Verified — matches the source", "group": "verified"}, | |
| "CORRECTED": {"ar": "مُصحَّح — دليل قوي على نص المصدر", "en": "Mismatch — source-backed correction available", "group": "mismatch"}, | |
| "UNSUPPORTED": {"ar": "غير مدعوم — لا يوجد مصدر مطابق في المراجع", "en": "Mismatch — no matching source in the corpus", "group": "mismatch"}, | |
| "HUMAN_REVIEW": {"ar": "مراجعة بشرية — الدليل غير كافٍ", "en": "Needs human review — insufficient evidence", "group": "review"}, | |
| } | |
| class IslamicContentVerifier: | |
| """Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review.""" | |
| def __init__(self, retriever: Optional[SourceRetriever] = None, detector: str = "rules", | |
| model_dir: Optional[str] = None, config: Optional[PipelineConfig] = None) -> None: | |
| self.cfg = config or PipelineConfig() | |
| self.retriever = retriever or SourceRetriever() | |
| self.verifier = Verifier(self.retriever, self.cfg.verifier) | |
| self.corrector = Corrector(self.retriever, self.cfg.corrector) | |
| self.detector = build_detector(detector, self.retriever, model_dir) | |
| self.detector_name = type(self.detector).__name__ | |
| # ---- public API ----------------------------------------------------------------------------------------- | |
| def analyze(self, text: str) -> dict: | |
| """Run the full pipeline on a generated text.""" | |
| text = self._validate(text) | |
| started = time.time() | |
| spans = self.detector.detect(text) if text.strip() else [] | |
| detect_seconds = time.time() - started | |
| result = self._analyze_spans(text, spans) | |
| result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)} | |
| return result | |
| def analyze_spans(self, text: str, spans: List[dict]) -> dict: | |
| """Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode).""" | |
| given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans] | |
| return self._analyze_spans(self._validate(text), given) | |
| # ---- internals ------------------------------------------------------------------------------------------ | |
| def _validate(text: str) -> str: | |
| if not isinstance(text, str): | |
| raise TypeError("Input text must be a string") | |
| if len(text) > MAX_INPUT_CHARS: | |
| raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}") | |
| return text | |
| def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict: | |
| reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))] | |
| counts = {status: 0 for status in STATUS_INFO} | |
| for report in reports: | |
| counts[report["status"]] += 1 | |
| return { | |
| "input_text": text, | |
| "detector": self.detector_name, | |
| "spans": reports, | |
| "corrected_text": self._apply_corrections(text, reports), | |
| "summary": { | |
| "n_spans": len(reports), | |
| "n_ayah": sum(r["type"] == "Ayah" for r in reports), | |
| "n_hadith": sum(r["type"] == "Hadith" for r in reports), | |
| **counts, | |
| "needs_human_review": counts["HUMAN_REVIEW"] > 0, | |
| }, | |
| } | |
| def _process_span(self, index: int, span: DetectedSpan) -> dict: | |
| try: | |
| return self._decide(index, span) | |
| except Exception: # a single failing quotation must not break the whole report | |
| logger.exception("Failed to process span %d", index) | |
| report = self._empty_report(index, span) | |
| self._finalize(report, "HUMAN_REVIEW", "internal_error: this quotation could not be processed automatically") | |
| return report | |
| def _empty_report(index: int, span: DetectedSpan) -> dict: | |
| return { | |
| "id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text, | |
| "detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)}, | |
| "verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0}, | |
| "evidence": None, "correction": None, "suggestion": None, | |
| } | |
| def _finalize(report: dict, status: str, reason: str) -> None: | |
| info = STATUS_INFO[status] | |
| report.update(status=status, status_ar=info["ar"], status_en=info["en"], group=info["group"], reason=reason) | |
| def _decide(self, index: int, span: DetectedSpan) -> dict: | |
| cfg, corr_cfg = self.cfg, self.cfg.corrector | |
| verification = self.verifier.verify(span.text, span.label) | |
| report = self._empty_report(index, span) | |
| report["verification"] = { | |
| "verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score, | |
| "method": verification.method, "n_candidates": verification.n_candidates, | |
| } | |
| report["evidence"] = self._evidence(span, verification) | |
| def proposal(match: Optional[CorrectionMatch]) -> Optional[dict]: | |
| if match is None: | |
| return None | |
| return { | |
| "text": match.text, "display_text": match.display, "source": match.source, | |
| "match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4), | |
| "comparison": compare(span.text, match.display), | |
| } | |
| if verification.verdict == "Correct": | |
| exact = self.corrector.is_exact_ayah(span.text) if span.label == "Ayah" else None | |
| report["verification"]["exact_match"] = exact | |
| if verification.method.startswith("borderline") or verification.confidence < cfg.verified_min_conf: | |
| status, reason = "HUMAN_REVIEW", "weak_verification: matched a source but with low confidence" | |
| elif exact is False: | |
| status = "HUMAN_REVIEW" | |
| reason = "near_match: the quote is close to a source ayah but NOT identical (words missing, added or changed)" | |
| report["suggestion"] = proposal(self.corrector.match(span.text, span.label)) | |
| else: | |
| status, reason = "VERIFIED", f"matched source ({verification.method})" | |
| else: | |
| match = self.corrector.match(span.text, span.label) | |
| strong, low = (corr_cfg.quran_strong, corr_cfg.quran_low) if span.label == "Ayah" else (corr_cfg.hadith_strong, corr_cfg.hadith_low) | |
| candidate = proposal(match) | |
| if match is None or match.strength < low: | |
| confident_abstain = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline") | |
| if match is None or match.strength < cfg.unsupported_strength or confident_abstain: | |
| status, reason = "UNSUPPORTED", "no source in the corpus matches this quotation" | |
| else: | |
| status, reason = "HUMAN_REVIEW", "insufficient_evidence: no clear source and low verification confidence" | |
| report["suggestion"] = candidate | |
| elif match.strength >= strong and match.full_ratio >= corr_cfg.min_full_ratio: | |
| status = "CORRECTED" | |
| reason = f"strong match to {self.describe_source(match.source)} (strength {match.strength:.2f})" | |
| report["correction"] = {**candidate, "applied": True} | |
| else: | |
| status = "HUMAN_REVIEW" | |
| reason = (f"candidate source found ({self.describe_source(match.source)}, strength {match.strength:.2f}) " | |
| "but evidence is not strong enough for automatic correction") | |
| report["suggestion"] = candidate | |
| if span.confidence is not None and span.confidence < cfg.min_detection_conf and status != "HUMAN_REVIEW": | |
| reason = f"low detection confidence ({span.confidence:.2f}); was {status}: {reason}" | |
| status = "HUMAN_REVIEW" | |
| if report["correction"]: | |
| report["suggestion"], report["correction"] = {**report["correction"], "applied": False}, None | |
| self._finalize(report, status, reason) | |
| return report | |
| def _evidence(span: DetectedSpan, verification: Verification) -> Optional[dict]: | |
| source = verification.source | |
| if not source: | |
| return None | |
| if span.label == "Ayah": | |
| reference = {"type": "Quran", "surah_id": source["surah_id"], "surah_name": source["surah_name"], "ayah": source["ayah_id"]} | |
| else: | |
| reference = {"type": "Hadith", "hadithID": source["hadithID"], "book": source["book"], "title": source["title"]} | |
| return {"source": reference, "signals": source.get("signals"), "comparison": compare(span.text, source["text"])} | |
| def describe_source(source: dict) -> str: | |
| """Human-readable reference, e.g. ``Quran الفاتحة 1-3`` or ``Hadith #123``.""" | |
| if source["type"] == "Quran": | |
| start, end = source["ayah_start"], source["ayah_end"] | |
| return f"Quran {source['surah_name']} {start}" + (f"-{end}" if end != start else "") | |
| return f"Hadith #{source['hadithID']}" | |
| def _apply_corrections(text: str, reports: List[dict]) -> str: | |
| out = text | |
| for report in sorted(reports, key=lambda r: r["start"], reverse=True): | |
| if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"): | |
| out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:] | |
| return out | |