Finallll / detector.py
Ghada-99-Ragab's picture
Upload 26 files
597dbb9 verified
Raw History Blame Contribute Delete
8.41 kB
"""Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.
The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah /
Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books).
Introductory phrases such as "ู‚ุงู„ ุงู„ู„ู‡ ุชุนุงู„ู‰" or "ู‚ุงู„ ุฑุณูˆู„ ุงู„ู„ู‡ ๏ทบ" are only a tie-breaker / fallback hint, never a
requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded."""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import List, Optional
from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize
# --------------------------------------------------------------------------------------------------------------
@dataclass
class DetectedSpan:
start: int
end: int # exclusive
label: str # 'Ayah' | 'Hadith'
confidence: Optional[float] # None for the rule backend
source: str # 'rules' | 'given'
text: str = ""
QUOTE_CHARS = " \t\r\n\"โ€œโ€ยซยป๏ดฟ๏ดพ{}()[]"
def trim_span(text: str, start: int, end: int):
"""Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
while start < end and text[start] in QUOTE_CHARS:
start += 1
while end > start and text[end - 1] in QUOTE_CHARS + ".ุŒ,ุ›:":
end -= 1
return start, end
AYAH_TRIGGERS = [
"ู‚ุงู„ ุงู„ู„ู‡", "ู‚ูˆู„ู‡ ุชุนุงู„ู‰", "ู‚ุงู„ ุชุนุงู„ู‰", "ูŠู‚ูˆู„ ุงู„ู„ู‡", "ูŠู‚ูˆู„ ุชุนุงู„ู‰", "ู‚ุงู„ ุณุจุญุงู†ู‡", "ู‚ูˆู„ู‡ ุณุจุญุงู†ู‡", "ููŠ ูƒุชุงุจู‡",
"ุณูˆุฑุฉ", "ุงู„ุขูŠุฉ", "ุงู„ุงูŠุฉ", "ุงู„ุขูŠุงุช", "ุงู„ู‚ุฑุขู†", "ุงู„ู‚ุฑุงู†", "ูƒุชุงุจ ุงู„ู„ู‡", "ุนุฒ ูˆุฌู„", "ุฌู„ ุฌู„ุงู„ู‡", "ูู‚ุงู„ ุชุนุงู„ู‰",
"ุฐูƒุฑ ุงู„ู„ู‡", "๏ดฟ",
]
HADITH_TRIGGERS = [
"ุฑุณูˆู„ ุงู„ู„ู‡", "ุงู„ู†ุจูŠ", "ุตู„ู‰ ุงู„ู„ู‡ ุนู„ูŠู‡ ูˆุณู„ู…", "๏ทบ", "ุนู„ูŠู‡ ุงู„ุตู„ุงุฉ ูˆุงู„ุณู„ุงู…", "ุญุฏูŠุซ", "ุฑูˆุงู‡", "ุฑูˆู‰", "ู…ุชูู‚ ุนู„ูŠู‡",
"ุงู„ุญุฏูŠุซ", "ูู‚ุงู„", "ู‚ุงู„ ุต", "ุตู„ู‰ ุงู„ู„ู‡ ุนู„ูŠู‡", "ูˆุณู„ู…",
]
_FORMULA_WORDS = {
normalize_for_matching(w)
for w in "ู‚ุงู„ ู‚ุงู„ุช ุฑุณูˆู„ ุงู„ู„ู‡ ุตู„ู‰ ุนู„ูŠู‡ ูˆุณู„ู… ุงู„ู†ุจูŠ ุชุนุงู„ู‰ ุณุจุญุงู†ู‡ ุนุฒ ูˆุฌู„ ูู‚ุงู„ ูŠู‚ูˆู„ ุงู„ูƒุฑูŠู… ุงู„ุดุฑูŠู ุงู„ุญุฏูŠุซ ุงู„ุขูŠุฉ ุฑูˆู‰ ุฑูˆุงู‡ ุนู† ุฃู† ุฃู†ู‡ ุงู„ุจุฎุงุฑูŠ ูˆู…ุณู„ู…".split()
}
_BRACKET_PAIRS = [("โ€œ", "โ€"), ("ยซ", "ยป"), ("๏ดฟ", "๏ดพ"), ("{", "}"), ("(", ")"), ("[", "]")]
class RuleDetector:
"""Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5,
quoted_min_cov_hadith: float = 0.75) -> None:
"""``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a
hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited."""
self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
self.decouple_triggers = decouple_triggers and retriever is not None
self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith
@staticmethod
def _segments(text: str):
segments = []
quote_positions = [m.start() for m in re.finditer('"', text)]
if len(quote_positions) % 2 == 0:
pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
else: # a stray quote: fall back to every consecutive pair
pairs = zip(quote_positions, quote_positions[1:])
for a, b in pairs:
segments.append((a + 1, b))
for opener, closer in _BRACKET_PAIRS:
for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
segments.append((m.start(1), m.end(1)))
return segments
@staticmethod
def _trigger_type(context: str) -> Optional[str]:
best_end, best_label = -1, None
for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
for trigger in triggers:
pos = context.rfind(trigger)
if pos >= 0 and pos + len(trigger) > best_end:
best_end, best_label = pos + len(trigger), label
return best_label
def _corpus_coverage(self, span: str):
"""Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
if self.kb is None:
return 0.0, 0.0
quran_words = set(tokenize(normalize_strict(span)))
if not quran_words:
return 0.0, 0.0
quran_cov = max(
(len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words)
for c in self.kb.search_quran_ayahs(span, top_k=5)),
default=0.0,
)
hadith_words = set(tokenize(normalize_lenient(span)))
hadith_cov = max(
(len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
for c in self.kb.search_hadith(span, top_k=5)),
default=0.0,
) if hadith_words else 0.0
return quran_cov, hadith_cov
def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]:
"""Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose
coverage is too low for the corpus to decide on its own."""
quran_ok = quran_cov >= self.quoted_min_cov
hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them
if (quran_ok or hadith_ok) and n_words >= 4:
if trigger and abs(quran_cov - hadith_cov) < 0.1:
return trigger
if quran_ok and hadith_ok:
return "Ayah" if quran_cov >= hadith_cov else "Hadith"
return "Ayah" if quran_ok else "Hadith"
return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported
def detect(self, text: str) -> List[DetectedSpan]:
candidates = []
for start, end in self._segments(text):
start, end = trim_span(text, start, end)
if end <= start:
continue
inner = text[start:end]
words = [w for w in normalize_for_matching(inner).split() if w]
if len(words) < self.min_words or len(inner) > 3000:
continue
if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
continue
trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
quran_cov, hadith_cov = self._corpus_coverage(inner)
label = None
if self.decouple_triggers:
label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov)
elif trigger:
label = trigger
other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
if other >= 0.8 and mine < 0.5:
label = "Hadith" if trigger == "Ayah" else "Ayah"
elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
if label is None:
continue
candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label))
candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
taken = []
for _, _, _, start, end, label in candidates:
if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken):
taken.append((start, end, label))
taken.sort()
return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken]