"""Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries. The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah / Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books). Introductory phrases such as "قال الله تعالى" or "قال رسول الله ﷺ" are only a tie-breaker / fallback hint, never a requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded.""" from __future__ import annotations import re from dataclasses import dataclass from typing import List, Optional from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize # -------------------------------------------------------------------------------------------------------------- @dataclass class DetectedSpan: start: int end: int # exclusive label: str # 'Ayah' | 'Hadith' confidence: Optional[float] # None for the rule backend source: str # 'rules' | 'given' text: str = "" QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]" def trim_span(text: str, start: int, end: int): """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks).""" while start < end and text[start] in QUOTE_CHARS: start += 1 while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:": end -= 1 return start, end AYAH_TRIGGERS = [ "قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه", "سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى", "ذكر الله", "﴿", ] HADITH_TRIGGERS = [ "رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه", "الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم", ] _FORMULA_WORDS = { normalize_for_matching(w) for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split() } _BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")] class RuleDetector: """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU).""" def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110, min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5, quoted_min_cov_hadith: float = 0.75) -> None: """``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited.""" self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov self.decouple_triggers = decouple_triggers and retriever is not None self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith @staticmethod def _segments(text: str): segments = [] quote_positions = [m.start() for m in re.finditer('"', text)] if len(quote_positions) % 2 == 0: pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs else: # a stray quote: fall back to every consecutive pair pairs = zip(quote_positions, quote_positions[1:]) for a, b in pairs: segments.append((a + 1, b)) for opener, closer in _BRACKET_PAIRS: for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S): segments.append((m.start(1), m.end(1))) return segments @staticmethod def _trigger_type(context: str) -> Optional[str]: best_end, best_label = -1, None for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)): for trigger in triggers: pos = context.rfind(trigger) if pos >= 0 and pos + len(trigger) > best_end: best_end, best_label = pos + len(trigger), label return best_label def _corpus_coverage(self, span: str): """Highest word coverage of the span by any top Quran ayah / Hadith candidate.""" if self.kb is None: return 0.0, 0.0 quran_words = set(tokenize(normalize_strict(span))) if not quran_words: return 0.0, 0.0 quran_cov = max( (len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words) for c in self.kb.search_quran_ayahs(span, top_k=5)), default=0.0, ) hadith_words = set(tokenize(normalize_lenient(span))) hadith_cov = max( (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words) for c in self.kb.search_hadith(span, top_k=5)), default=0.0, ) if hadith_words else 0.0 return quran_cov, hadith_cov def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]: """Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose coverage is too low for the corpus to decide on its own.""" quran_ok = quran_cov >= self.quoted_min_cov hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them if (quran_ok or hadith_ok) and n_words >= 4: if trigger and abs(quran_cov - hadith_cov) < 0.1: return trigger if quran_ok and hadith_ok: return "Ayah" if quran_cov >= hadith_cov else "Hadith" return "Ayah" if quran_ok else "Hadith" return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported def detect(self, text: str) -> List[DetectedSpan]: candidates = [] for start, end in self._segments(text): start, end = trim_span(text, start, end) if end <= start: continue inner = text[start:end] words = [w for w in normalize_for_matching(inner).split() if w] if len(words) < self.min_words or len(inner) > 3000: continue if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6: continue trigger = self._trigger_type(text[max(0, start - self.context_chars):start]) quran_cov, hadith_cov = self._corpus_coverage(inner) label = None if self.decouple_triggers: label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov) elif trigger: label = trigger other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov) if other >= 0.8 and mine < 0.5: label = "Hadith" if trigger == "Ayah" else "Ayah" elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4: label = "Ayah" if quran_cov >= hadith_cov else "Hadith" if label is None: continue candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label)) candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length taken = [] for _, _, _, start, end, label in candidates: if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken): taken.append((start, end, label)) taken.sort() return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken]