Islamic2 / detector.py
Ghada-99-Ragab's picture
Upload 21 files
6cb74c6 verified
Raw History Blame Contribute Delete
6.67 kB
"""Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.
The detector is rule-based: quotation marks and brackets mark candidate segments, introductory phrases ("قال الله
تعالى", "قال رسول الله") decide the type, and an optional corpus lookup corrects the type when the text clearly
belongs to the other corpus. No model weights are loaded."""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import List, Optional
from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize
# --------------------------------------------------------------------------------------------------------------
@dataclass
class DetectedSpan:
start: int
end: int # exclusive
label: str # 'Ayah' | 'Hadith'
confidence: Optional[float] # None for the rule backend
source: str # 'rules' | 'given'
text: str = ""
QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]"
def trim_span(text: str, start: int, end: int):
"""Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
while start < end and text[start] in QUOTE_CHARS:
start += 1
while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:":
end -= 1
return start, end
AYAH_TRIGGERS = [
"قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه",
"سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى",
"ذكر الله", "﴿",
]
HADITH_TRIGGERS = [
"رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه",
"الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم",
]
_FORMULA_WORDS = {
normalize_for_matching(w)
for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split()
}
_BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")]
class RuleDetector:
"""Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
min_corpus_cov: float = 0.6) -> None:
self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
@staticmethod
def _segments(text: str):
segments = []
quote_positions = [m.start() for m in re.finditer('"', text)]
if len(quote_positions) % 2 == 0:
pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
else: # a stray quote: fall back to every consecutive pair
pairs = zip(quote_positions, quote_positions[1:])
for a, b in pairs:
segments.append((a + 1, b))
for opener, closer in _BRACKET_PAIRS:
for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
segments.append((m.start(1), m.end(1)))
return segments
@staticmethod
def _trigger_type(context: str) -> Optional[str]:
best_end, best_label = -1, None
for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
for trigger in triggers:
pos = context.rfind(trigger)
if pos >= 0 and pos + len(trigger) > best_end:
best_end, best_label = pos + len(trigger), label
return best_label
def _corpus_coverage(self, span: str):
"""Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
if self.kb is None:
return 0.0, 0.0
quran_words = set(tokenize(normalize_strict(span)))
if not quran_words:
return 0.0, 0.0
quran_cov = max(
(len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words)
for c in self.kb.search_quran_ayahs(span, top_k=5)),
default=0.0,
)
hadith_words = set(tokenize(normalize_lenient(span)))
hadith_cov = max(
(len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
for c in self.kb.search_hadith(span, top_k=5)),
default=0.0,
) if hadith_words else 0.0
return quran_cov, hadith_cov
def detect(self, text: str) -> List[DetectedSpan]:
candidates = []
for start, end in self._segments(text):
start, end = trim_span(text, start, end)
if end <= start:
continue
inner = text[start:end]
words = [w for w in normalize_for_matching(inner).split() if w]
if len(words) < self.min_words or len(inner) > 3000:
continue
if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
continue
trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
quran_cov, hadith_cov = self._corpus_coverage(inner)
label = None
if trigger:
label = trigger
other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
if other >= 0.8 and mine < 0.5:
label = "Hadith" if trigger == "Ayah" else "Ayah"
elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
if label is None:
continue
candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label))
candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
taken = []
for _, _, _, start, end, label in candidates:
if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken):
taken.append((start, end, label))
taken.sort()
return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken]