Spaces:
Running
Running
Download verifier.py from Ghada-99-Ragab/lastfinal2: direct link, hf CLI and curl.
- Browser
- Download file 32 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/lastfinal2/resolve/main/verifier.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/lastfinal2/verifier.py
-
curl -L -o verifier.py https://huggingface.co/spaces/Ghada-99-Ragab/lastfinal2/resolve/main/verifier.py
32 kB
| """Evidence-based verification and source-backed correction of Quran and Hadith quotations. | |
| Pipeline (see ``IslamicContentVerifier``): | |
| input text -> detection -> BM25 retrieval -> alignment (phonetic skeleton, sliding window, LCS, dynamic gap) | |
| -> exact match : verified | |
| -> altered Quran passage : mismatch with the exact source text as the correction | |
| -> anything uncertain : human review (no correction is produced) | |
| -> nothing similar found : no matching source | |
| Safety rule: a correction is only ever the exact text of a retrieved source. Nothing is generated. | |
| """ | |
| from __future__ import annotations | |
| import difflib | |
| import logging | |
| import time | |
| from dataclasses import dataclass, field | |
| from typing import Dict, List, Optional | |
| from alignment import align | |
| from detector import DetectedSpan | |
| from scanner import HybridDetector | |
| from idgham import apply_idgham | |
| from normalization import content_words, normalize_for_matching | |
| from retrieval import SourceRetriever | |
| from similarity import best_match_score, compute_signals | |
| logger = logging.getLogger(__name__) | |
| MAX_INPUT_CHARS = 20_000 | |
| SHORT_QUOTE_WORDS = 3 # quotations shorter than this (found by the rules) always go to human review | |
| MIN_EXACT_TOKENS = 3 # shorter quotations are too generic to be verified by exact containment alone | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Configuration (all tunable numbers live here) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class VerifierConfig: | |
| """Thresholds calibrated in Subtask 1B of the research notebook.""" | |
| quran_correct_threshold: float = 0.94 | |
| quran_uncertain_low: float = 0.45 | |
| quran_min_coverage: float = 0.40 | |
| hadith_correct_threshold: float = 0.88 | |
| hadith_uncertain_low: float = 0.30 | |
| hadith_min_coverage: float = 0.70 | |
| quran_top_k: int = 25 | |
| hadith_top_k: int = 15 | |
| hadith_retrieval_guard: float = 0.20 | |
| class CorrectorConfig: | |
| """Correction is proposed only when match strength >= ``*_strong``; between ``*_low`` and ``*_strong`` the | |
| case goes to human review. Hadith is never auto-corrected (``hadith_strong`` > 1), by design: the exact | |
| fragment boundaries of a Hadith quotation cannot be reproduced reliably, and narrations differ legitimately.""" | |
| max_window: int = 8 | |
| hadith_top_k: int = 40 | |
| quran_strong: float = 0.65 | |
| min_full_ratio: float = 0.40 | |
| quran_low: float = 0.55 | |
| hadith_strong: float = 1.01 | |
| hadith_low: float = 0.45 | |
| class PipelineConfig: | |
| verifier: VerifierConfig = field(default_factory=VerifierConfig) | |
| corrector: CorrectorConfig = field(default_factory=CorrectorConfig) | |
| verified_min_conf: float = 0.75 # "Correct" verdicts weaker than this -> human review | |
| unsupported_min_conf: float = 0.70 # "Incorrect + no source" needs this confidence to abstain confidently | |
| unsupported_strength: float = 0.35 # ...or the best source match is this weak (nothing similar exists) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Verification (Subtask 1B) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class Verification: | |
| verdict: str # 'Correct' | 'Incorrect' | |
| confidence: float | |
| best_score: float | |
| method: str | |
| source: Optional[dict] = None # best matching source record (with 'signals') | |
| n_candidates: int = 0 | |
| retrieval_top: float = 0.0 | |
| class Verifier: | |
| """Scores a quotation against retrieved candidates and returns a verdict with its supporting source.""" | |
| def __init__(self, retriever: SourceRetriever, config: Optional[VerifierConfig] = None) -> None: | |
| self.kb = retriever | |
| self.cfg = config or VerifierConfig() | |
| def verify(self, span_text: str, content_type: str) -> Verification: | |
| if not span_text or not span_text.strip(): | |
| return self._result("Incorrect", 0.95, 0.0, None, 0, "empty_span") | |
| if content_type == "Ayah": | |
| return self._verify_quran(span_text) | |
| if content_type == "Hadith": | |
| return self._verify_hadith(span_text) | |
| return self._result("Incorrect", 0.5, 0.0, None, 0, "unknown_type") | |
| def _verify_quran(self, span: str) -> Verification: | |
| cfg = self.cfg | |
| candidates = self.kb.search_quran_ayahs(span, top_k=cfg.quran_top_k) | |
| if not candidates: | |
| return self._result("Incorrect", 0.8, 0.0, None, 0, "no_candidates") | |
| score, best = best_match_score(span, candidates, "Ayah") | |
| signals = best.get("signals", {}) if best else {} | |
| coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) | |
| n = len(candidates) | |
| if is_substring and coverage >= cfg.quran_min_coverage: | |
| return self._result("Correct", min(0.98, 0.85 + score * 0.15), score, best, n, "substring_match") | |
| if score >= cfg.quran_correct_threshold and coverage >= cfg.quran_min_coverage: | |
| return self._result("Correct", min(0.95, 0.70 + score * 0.25), score, best, n, "threshold_pass") | |
| if score <= cfg.quran_uncertain_low: | |
| return self._result("Incorrect", min(0.95, 0.70 + (1 - score) * 0.25), score, best, n, "threshold_fail") | |
| strong = sum( | |
| 1 for cand in candidates[:10] | |
| if (s := compute_signals(span, cand.get("text", ""), "Ayah"))["coverage"] >= 0.80 and s["lcs_ratio"] >= 0.75 | |
| ) | |
| if strong >= 2: | |
| return self._result("Correct", 0.60 + min(0.20, strong * 0.05), score, best, n, "borderline_multi_cov") | |
| return self._result("Incorrect", 0.58, score, best, n, "borderline_default") | |
| def _verify_hadith(self, span: str) -> Verification: | |
| cfg = self.cfg | |
| candidates = self.kb.search_hadith(span, top_k=cfg.hadith_top_k) | |
| if not candidates: | |
| return self._result("Incorrect", 0.75, 0.0, None, 0, "no_candidates") | |
| top_retrieval = candidates[0].get("retrieval_score", 0.0) | |
| score, best = best_match_score(span, candidates, "Hadith") | |
| signals = best.get("signals", {}) if best else {} | |
| coverage, is_substring = signals.get("coverage", 0.0), signals.get("is_substring", 0) | |
| n = len(candidates) | |
| if is_substring and coverage >= cfg.hadith_min_coverage and top_retrieval >= cfg.hadith_retrieval_guard: | |
| return self._result("Correct", min(0.97, 0.80 + score * 0.17), score, best, n, "substring_match", top_retrieval) | |
| if score >= cfg.hadith_correct_threshold and coverage >= cfg.hadith_min_coverage: | |
| return self._result("Correct", min(0.92, 0.65 + score * 0.27), score, best, n, "threshold_pass", top_retrieval) | |
| if score <= cfg.hadith_uncertain_low: | |
| return self._result("Incorrect", min(0.90, 0.65 + (1 - score) * 0.25), score, best, n, "threshold_fail", top_retrieval) | |
| moderate = sum( | |
| 1 for cand in candidates[:8] | |
| if (s := compute_signals(span, cand.get("text", ""), "Hadith"))["coverage"] >= 0.65 and s["lcs_ratio"] >= 0.55 | |
| ) | |
| if moderate >= 2 and top_retrieval >= 0.30: | |
| return self._result("Correct", 0.58 + min(0.22, moderate * 0.06), score, best, n, "borderline_multi_cov", top_retrieval) | |
| if top_retrieval < 0.25 or score < 0.45: | |
| return self._result("Incorrect", 0.60, score, best, n, "borderline_low_retrieval", top_retrieval) | |
| return self._result("Incorrect", 0.55, score, best, n, "borderline_default", top_retrieval) | |
| def _result(verdict, confidence, score, best, n_candidates, method, top_retrieval=0.0) -> Verification: | |
| return Verification(verdict, round(confidence, 4), round(score, 4), method, best, n_candidates, round(top_retrieval, 4)) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # Correction (Subtask 1C): locate the true ayah window / Hadith record and return its exact text | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class CorrectionMatch: | |
| kind: str # 'Ayah' | 'Hadith' | |
| strength: float # coverage (Quran) / symmetric containment (Hadith) | |
| full_ratio: float | |
| text: str # proposed correction in the official 1C format (idgham + '(n)' ayah markers) | |
| source: dict # reference metadata | |
| display: str = "" # clean human-readable version | |
| class Corrector: | |
| def __init__(self, retriever: SourceRetriever, config: Optional[CorrectorConfig] = None) -> None: | |
| self.kb = retriever | |
| self.cfg = config or CorrectorConfig() | |
| def match(self, span_text: str, content_type: str) -> Optional[CorrectionMatch]: | |
| return self.match_quran(span_text) if content_type == "Ayah" else self.match_hadith(span_text) | |
| def match_quran(self, query_text: str) -> Optional[CorrectionMatch]: | |
| """Best window of 1-8 consecutive ayahs (dynamic sliding window over BM25 seeds, exact-result pruning).""" | |
| kb = self.kb | |
| query_norm = normalize_for_matching(query_text) | |
| query_words = content_words(query_norm.split()) | |
| if not query_words: | |
| return None | |
| query_len = len(query_norm) | |
| memo: Dict[tuple, tuple] = {} | |
| best = None # (key, coverage, ratio, surah, start, end) | |
| for seed in kb.quran_seed_ayahs(query_words, top_k=25): | |
| surah, ayah = kb.quran[seed]["surah_id"], kb.quran[seed]["ayah_id"] | |
| ayahs = kb.quran_by_surah[surah] | |
| min_ayah, max_ayah = min(ayahs), max(ayahs) | |
| for offset in range(3): | |
| start = ayah - offset | |
| if start < min_ayah: | |
| continue | |
| window_len = -1 | |
| for length in range(1, self.cfg.max_window + 1): | |
| end = start + length - 1 | |
| if end > max_ayah: | |
| break | |
| window_len += len(kb.q_norm_match[ayahs[end]]) + 1 | |
| len_diff = abs(window_len - query_len) | |
| upper_bound = min(1.0, window_len / max(query_len, 1)) | |
| if best is not None: # exact-result pruning | |
| best_cov, best_neg = best[0][0], best[0][1] | |
| if upper_bound < best_cov or (upper_bound == best_cov and -len_diff < best_neg): | |
| continue | |
| key_pos = (surah, start, end) | |
| if key_pos in memo: | |
| continue | |
| window = " ".join(kb.q_norm_match[ayahs[a]] for a in range(start, end + 1)) | |
| matcher = difflib.SequenceMatcher(None, query_norm, window, autojunk=False) | |
| matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) | |
| coverage = matched / max(query_len, 1) | |
| key = (coverage, -len_diff, matcher.ratio()) | |
| memo[key_pos] = key | |
| if best is None or key > best[0]: | |
| best = (key, coverage, key[2], surah, start, end) | |
| if best is None: | |
| return None | |
| _, coverage, ratio, surah, start, end = best | |
| display = " ".join(kb.quran[kb.quran_by_surah[surah][a]]["text"] for a in range(start, end + 1)) | |
| return CorrectionMatch( | |
| "Ayah", coverage, ratio, self._ayah_text(surah, start, end), | |
| {"type": "Quran", "surah_id": surah, "surah_name": kb.quran[kb.quran_by_surah[surah][start]]["surah_name"], | |
| "ayah_start": start, "ayah_end": end}, | |
| display, | |
| ) | |
| def _ayah_text(self, surah: int, start: int, end: int) -> str: | |
| kb, multi = self.kb, end > start | |
| parts = [] | |
| for a in range(start, end + 1): | |
| text = kb.quran[kb.quran_by_surah[surah][a]]["text"] | |
| parts.append(f"{text} ({a})" if multi else text) | |
| return apply_idgham(" ".join(parts)).replace("\u0640", "") | |
| def match_hadith(self, query_text: str) -> Optional[CorrectionMatch]: | |
| kb = self.kb | |
| query_norm = normalize_for_matching(query_text) | |
| query_words = content_words(query_norm.split()) | |
| if not query_words: | |
| return None | |
| query_len = len(query_norm) | |
| best = None # (key, idx, field, ratio) | |
| for idx in kb.hadith_candidates(query_words, self.cfg.hadith_top_k): | |
| for field_name in ("matn", "full"): | |
| text = kb.hadith_norm(idx, field_name) | |
| if not text: | |
| continue | |
| upper_bound = min(1.0, query_len / max(len(text), 1)) | |
| if best is not None and upper_bound < best[0][0]: | |
| continue | |
| matcher = difflib.SequenceMatcher(None, query_norm, text, autojunk=False) | |
| matched = sum(b.size for b in matcher.get_matching_blocks() if b.size >= 4) | |
| coverage, candidate_cov = matched / max(query_len, 1), matched / max(len(text), 1) | |
| key = (min(coverage, candidate_cov), matcher.ratio()) | |
| if best is None or key > best[0]: | |
| best = (key, idx, field_name, key[1]) | |
| if best is None: | |
| return None | |
| key, idx, field_name, ratio = best | |
| record = kb.hadith[idx] | |
| text = record[field_name].strip() | |
| return CorrectionMatch( | |
| "Hadith", key[0], ratio, text, | |
| {"type": "Hadith", "hadithID": record["hadithID"], "book": record["book"], "title": record["title"], | |
| "field": "matn" if field_name == "matn" else "full_text"}, | |
| text, | |
| ) | |
| # -------------------------------------------------------------------------------------------------------------- | |
| # End-to-end pipeline | |
| # -------------------------------------------------------------------------------------------------------------- | |
| STATUS_INFO = { | |
| "VERIFIED": {"ar": "موثّق", "group": "verified"}, | |
| "CORRECTED": {"ar": "غير مطابق — يوجد تصحيح من المصدر", "group": "mismatch"}, | |
| "UNSUPPORTED": {"ar": "غير مطابق — لا يوجد مصدر مطابق", "group": "mismatch"}, | |
| "HUMAN_REVIEW": {"ar": "يحتاج مراجعة بشرية", "group": "review"}, | |
| } | |
| class IslamicContentVerifier: | |
| """Detect quotations, verify them against the corpora and decide: verified, corrected, unsupported or review.""" | |
| def __init__(self, retriever: Optional[SourceRetriever] = None, config: Optional[PipelineConfig] = None, | |
| use_scanner: bool = True, decouple_triggers: bool = True) -> None: | |
| self.cfg = config or PipelineConfig() | |
| self.retriever = retriever or SourceRetriever() | |
| self.verifier = Verifier(self.retriever, self.cfg.verifier) | |
| self.corrector = Corrector(self.retriever, self.cfg.corrector) | |
| self.detector = HybridDetector(self.retriever, use_scanner, rules_use_corpus=decouple_triggers, | |
| decouple_triggers=decouple_triggers) | |
| self.detector_name = type(self.detector).__name__ | |
| # ---- public API ----------------------------------------------------------------------------------------- | |
| def detect(self, text: str) -> List[DetectedSpan]: | |
| return self.detector.detect(self._validate(text)) if text.strip() else [] | |
| def needs_hadith(self, text: str) -> bool: | |
| """True if the text contains a Hadith quotation (so the large Hadith index must be loaded).""" | |
| return any(span.label == "Hadith" for span in self.detect(text)) | |
| def analyze(self, text: str) -> dict: | |
| """Run the full pipeline on a generated text.""" | |
| text = self._validate(text) | |
| started = time.time() | |
| spans = self.detector.detect(text) if text.strip() else [] | |
| detect_seconds = time.time() - started | |
| result = self._analyze_spans(text, spans) | |
| result["timings"] = {"detect_s": round(detect_seconds, 3), "total_s": round(time.time() - started, 3)} | |
| return result | |
| def analyze_spans(self, text: str, spans: List[dict]) -> dict: | |
| """Skip detection and use given spans ``[{label, start, end}]`` (evaluation / oracle mode).""" | |
| given = [DetectedSpan(s["start"], s["end"], s["label"], None, "given", text[s["start"]:s["end"]]) for s in spans] | |
| return self._analyze_spans(self._validate(text), given) | |
| def analyze_detected(self, text: str, spans: List[DetectedSpan]) -> dict: | |
| """Verify spans produced elsewhere (e.g. detector + hosted model merged by ``camelbert_adapter.merge_spans``).""" | |
| return self._analyze_spans(self._validate(text), spans) | |
| # ---- internals ------------------------------------------------------------------------------------------ | |
| def _validate(text: str) -> str: | |
| if not isinstance(text, str): | |
| raise TypeError("Input text must be a string") | |
| if len(text) > MAX_INPUT_CHARS: | |
| raise ValueError(f"Input is too long ({len(text)} characters); the limit is {MAX_INPUT_CHARS}") | |
| return text | |
| def _analyze_spans(self, text: str, spans: List[DetectedSpan]) -> dict: | |
| reports = [self._process_span(i + 1, span) for i, span in enumerate(sorted(spans, key=lambda s: s.start))] | |
| counts = {status: 0 for status in STATUS_INFO} | |
| for report in reports: | |
| counts[report["status"]] += 1 | |
| return { | |
| "input_text": text, | |
| "detector": self.detector_name, | |
| "spans": reports, | |
| "corrected_text": self._apply_corrections(text, reports), | |
| "summary": { | |
| "n_spans": len(reports), | |
| "n_ayah": sum(r["type"] == "Ayah" for r in reports), | |
| "n_hadith": sum(r["type"] == "Hadith" for r in reports), | |
| **counts, | |
| "needs_human_review": counts["HUMAN_REVIEW"] > 0, | |
| }, | |
| } | |
| def _process_span(self, index: int, span: DetectedSpan) -> dict: | |
| try: | |
| return self._decide(index, span) | |
| except Exception: # a single failing quotation must not break the whole report | |
| logger.exception("Failed to process span %d", index) | |
| report = self._empty_report(index, span) | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "internal_error"}) | |
| return report | |
| def _empty_report(index: int, span: DetectedSpan) -> dict: | |
| return { | |
| "id": index, "type": span.label, "start": span.start, "end": span.end, "text": span.text, | |
| "detection": {"backend": span.source, "confidence": None if span.confidence is None else round(span.confidence, 4)}, | |
| "verification": {"verdict": "Incorrect", "confidence": 0.0, "score": 0.0, "method": "error", "n_candidates": 0}, | |
| "evidence": None, "correction": None, "suggestion": None, "notes": [], | |
| } | |
| def _finalize(report: dict, status: str, reason: dict) -> None: | |
| info = STATUS_INFO[status] | |
| report.update(status=status, status_ar=info["ar"], group=info["group"], reason=reason) | |
| def source_label(source: dict) -> str: | |
| """Plain reference used in reasons, e.g. ``سورة البقرة 153`` or ``حديث رقم 5``.""" | |
| if source["type"] == "Quran": | |
| start, end = source["ayah_start"], source["ayah_end"] | |
| return f"سورة {source['surah_name']} {start}" + (f"–{end}" if end != start else "") | |
| return f"حديث رقم {source['hadithID']}" | |
| def _decide(self, index: int, span: DetectedSpan) -> dict: | |
| report = self._empty_report(index, span) | |
| verification = self.verifier.verify(span.text, span.label) | |
| report["verification"] = { | |
| "verdict": verification.verdict, "confidence": verification.confidence, "score": verification.best_score, | |
| "method": verification.method, "n_candidates": verification.n_candidates, | |
| } | |
| whole = self._whole_ayah(span.text) if span.label == "Ayah" else None | |
| if whole is not None: | |
| self._verified_whole_ayah(report, span, whole) # a complete ayah, however short or common its words | |
| elif span.label == "Ayah": | |
| self._decide_quran(report, span, verification) | |
| else: | |
| self._decide_hadith(report, span, verification) | |
| if whole is None and span.source == "rules" and len(normalize_for_matching(span.text).split()) < SHORT_QUOTE_WORDS: | |
| # two words cannot identify a source: never claim "verified" or "wrong", and never correct | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "too_short"}) | |
| report["correction"] = None | |
| report["suggestion"] = None | |
| elif report["status"] in ("UNSUPPORTED", "HUMAN_REVIEW"): | |
| self._cross_check(report, span) | |
| if report["status"] == "VERIFIED" and span.hint and span.hint != span.label: | |
| report["notes"].append({"code": "is_ayah" if span.label == "Ayah" else "is_hadith", "source": self.source_label(report["evidence"]["source"]), "misattributed": True}) | |
| return report | |
| # ---- helpers for the decision --------------------------------------------------------------------------- | |
| def _whole_ayah(self, text: str) -> Optional[int]: | |
| """Index of the ayah whose complete text equals ``text`` (two words or more), else None.""" | |
| if getattr(self, "_whole_map", None) is None: | |
| mapping: Dict[str, int] = {} | |
| for i, norm in enumerate(self.retriever.q_norm_match): | |
| if len(norm.split()) >= 2: | |
| mapping.setdefault(norm, i) | |
| self._whole_map = mapping | |
| norm = normalize_for_matching(text) | |
| return self._whole_map.get(norm) if len(norm.split()) >= 2 else None | |
| def _verified_whole_ayah(self, report: dict, span: DetectedSpan, idx: int) -> None: | |
| record = self.retriever.quran[idx] | |
| source_ref = {"type": "Quran", "surah_id": record["surah_id"], "surah_name": record["surah_name"], | |
| "ayah_start": record["ayah_id"], "ayah_end": record["ayah_id"]} | |
| alignment = align(span.text, record["text"], self.retriever.quran_vocabulary) | |
| report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, record["text"], "Ayah"), | |
| "comparison": alignment} | |
| report["verification"].update(verdict="Correct", confidence=0.98, score=1.0, method="whole_ayah") | |
| self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)}) | |
| if alignment["diacritic_notes"]: | |
| report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])}) | |
| def _cross_check(self, report: dict, span: DetectedSpan) -> None: | |
| """A quotation that fails against its claimed corpus but is found verbatim in the other one is probably | |
| mis-attributed (a Hadith presented as an ayah, or the reverse): say so.""" | |
| if len(normalize_for_matching(span.text).split()) < 3: | |
| return | |
| try: | |
| if span.label == "Ayah": | |
| other = self.verifier._verify_hadith(span.text) | |
| if other.verdict == "Correct" and other.method == "substring_match" and other.source: | |
| report["notes"].append({"code": "is_hadith", "source": f"حديث رقم {other.source['hadithID']}"}) | |
| else: | |
| other = self.verifier._verify_quran(span.text) | |
| if other.verdict == "Correct" and other.method == "substring_match" and other.source: | |
| c = other.source | |
| report["notes"].append({"code": "is_ayah", "source": f"سورة {c['surah_name']} {c['ayah_id']}"}) | |
| except Exception: # a diagnostic note must never break the decision | |
| logger.exception("cross-check failed") | |
| def _proposal(span: DetectedSpan, match: Optional[CorrectionMatch]) -> Optional[dict]: | |
| if match is None: | |
| return None | |
| return {"text": match.text, "display_text": match.display, "source": match.source, | |
| "match_strength": round(match.strength, 4), "full_ratio": round(match.full_ratio, 4)} | |
| def _decide_quran(self, report: dict, span: DetectedSpan, verification: Verification) -> None: | |
| cfg, corr_cfg = self.cfg, self.cfg.corrector | |
| match = self.corrector.match_quran(span.text) | |
| if match is not None: | |
| source_text, source_ref = match.display, match.source | |
| elif verification.source: | |
| candidate = verification.source | |
| source_text = candidate["text"] | |
| source_ref = {"type": "Quran", "surah_id": candidate["surah_id"], "surah_name": candidate["surah_name"], | |
| "ayah_start": candidate["ayah_id"], "ayah_end": candidate["ayah_id"]} | |
| else: | |
| source_text, source_ref = "", None | |
| alignment = align(span.text, source_text, self.retriever.quran_vocabulary) if source_text else None | |
| if source_ref: | |
| report["evidence"] = {"source": source_ref, "signals": compute_signals(span.text, source_text, "Ayah"), | |
| "comparison": alignment} | |
| proposal = self._proposal(span, match) | |
| n_tokens = len(content_words(normalize_for_matching(span.text).split())) if span.text else 0 | |
| has_tokens = alignment is not None and alignment["exact"] and len(alignment["word_diff"]) >= 1 | |
| if has_tokens and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= MIN_EXACT_TOKENS: | |
| self._finalize(report, "VERIFIED", {"code": "exact_match", "source": self.source_label(source_ref)}) | |
| if alignment["diacritic_notes"]: | |
| report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])}) | |
| if alignment["orthographic_variants"]: | |
| report["notes"].append({"code": "orthographic_variant", "n": alignment["orthographic_variants"]}) | |
| return | |
| local = bool(alignment and alignment.get("near") and alignment.get("word_similarity", 0) >= 0.75 and alignment.get("source_excerpt")) | |
| strong = match is not None and match.strength >= corr_cfg.quran_strong and (match.full_ratio >= corr_cfg.min_full_ratio or local) | |
| if strong: | |
| self._finalize(report, "CORRECTED", {"code": "altered_passage", "source": self.source_label(source_ref), | |
| "n": alignment["mismatches"] if alignment else 0, | |
| "reordered": bool(alignment and alignment["reordered"])}) | |
| # replace only the passage the author quoted, not the whole ayah that contains it | |
| excerpt = alignment["source_excerpt"] if alignment and alignment["source_excerpt"] else proposal["display_text"] | |
| report["correction"] = {**proposal, "display_text": excerpt, "applied": True} | |
| return | |
| if verification.verdict == "Correct": | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"}) | |
| report["suggestion"] = proposal | |
| elif match is None or match.strength < corr_cfg.quran_low: | |
| confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline") | |
| if match is None or match.strength < cfg.unsupported_strength or confident: | |
| self._finalize(report, "UNSUPPORTED", {"code": "no_source"}) | |
| else: | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"}) | |
| report["suggestion"] = proposal | |
| else: | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "candidate_not_strong", "source": self.source_label(source_ref), | |
| "strength": round(match.strength, 2)}) | |
| report["suggestion"] = proposal | |
| def _decide_hadith(self, report: dict, span: DetectedSpan, verification: Verification) -> None: | |
| cfg, corr_cfg = self.cfg, self.cfg.corrector | |
| best = verification.source | |
| alignment = None | |
| if best: | |
| alignment = align(span.text, best["text"]) | |
| report["evidence"] = { | |
| "source": {"type": "Hadith", "hadithID": best["hadithID"], "book": best["book"], "title": best["title"]}, | |
| "signals": best.get("signals"), "comparison": alignment, | |
| } | |
| exact_tokens = alignment is not None and alignment["exact"] and sum(len(op["span"].split()) for op in alignment["word_diff"]) >= 4 | |
| if verification.verdict == "Correct" or exact_tokens: | |
| altered = (not exact_tokens and alignment is not None and not alignment["exact"] and alignment["mismatches"] >= 1) | |
| if altered: # any added, dropped or changed word can alter the meaning (خيرًا / شرًّا, or a missing "لا"): only an exact match is verified | |
| match = self.corrector.match_hadith(span.text) | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_altered", "n": alignment["mismatches"], | |
| "source": f"حديث رقم {best['hadithID']}"}) | |
| report["suggestion"] = self._proposal(span, match) | |
| report["verification"]["verdict"] = "Incorrect" | |
| elif exact_tokens or (not verification.method.startswith("borderline") and verification.confidence >= cfg.verified_min_conf): | |
| self._finalize(report, "VERIFIED", {"code": "exact_match" if exact_tokens else "close_match", | |
| "source": f"حديث رقم {best['hadithID']}"}) | |
| if alignment and not alignment["exact"]: | |
| report["notes"].append({"code": "hadith_minor_diffs", "n": alignment["mismatches"]}) | |
| if alignment and alignment["diacritic_notes"]: | |
| report["notes"].append({"code": "diacritic_conflict", "n": len(alignment["diacritic_notes"])}) | |
| else: | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "weak_match"}) | |
| return | |
| match = self.corrector.match_hadith(span.text) | |
| proposal = self._proposal(span, match) | |
| if match is None or match.strength < corr_cfg.hadith_low: | |
| confident = verification.confidence >= cfg.unsupported_min_conf and not verification.method.startswith("borderline") | |
| if match is None or match.strength < cfg.unsupported_strength or confident: | |
| self._finalize(report, "UNSUPPORTED", {"code": "no_source"}) | |
| else: | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "insufficient_evidence"}) | |
| report["suggestion"] = proposal | |
| else: # a similar narration exists, but Hadith is never corrected automatically | |
| self._finalize(report, "HUMAN_REVIEW", {"code": "hadith_candidate", "source": self.source_label(match.source), | |
| "strength": round(match.strength, 2)}) | |
| report["suggestion"] = proposal | |
| def _apply_corrections(text: str, reports: List[dict]) -> str: | |
| out = text | |
| for report in sorted(reports, key=lambda r: r["start"], reverse=True): | |
| if report["status"] == "CORRECTED" and report["correction"] and report["correction"].get("applied"): | |
| out = out[: report["start"]] + report["correction"]["display_text"] + out[report["end"]:] | |
| return out | |