Spaces:
Running
Running
File size: 8,409 Bytes
597dbb9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 | """Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.
The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah /
Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books).
Introductory phrases such as "ูุงู ุงููู ุชุนุงูู" or "ูุงู ุฑุณูู ุงููู ๏ทบ" are only a tie-breaker / fallback hint, never a
requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded."""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import List, Optional
from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize
# --------------------------------------------------------------------------------------------------------------
@dataclass
class DetectedSpan:
start: int
end: int # exclusive
label: str # 'Ayah' | 'Hadith'
confidence: Optional[float] # None for the rule backend
source: str # 'rules' | 'given'
text: str = ""
QUOTE_CHARS = " \t\r\n\"โโยซยป๏ดฟ๏ดพ{}()[]"
def trim_span(text: str, start: int, end: int):
"""Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
while start < end and text[start] in QUOTE_CHARS:
start += 1
while end > start and text[end - 1] in QUOTE_CHARS + ".ุ,ุ:":
end -= 1
return start, end
AYAH_TRIGGERS = [
"ูุงู ุงููู", "ูููู ุชุนุงูู", "ูุงู ุชุนุงูู", "ูููู ุงููู", "ูููู ุชุนุงูู", "ูุงู ุณุจุญุงูู", "ูููู ุณุจุญุงูู", "ูู ูุชุงุจู",
"ุณูุฑุฉ", "ุงูุขูุฉ", "ุงูุงูุฉ", "ุงูุขูุงุช", "ุงููุฑุขู", "ุงููุฑุงู", "ูุชุงุจ ุงููู", "ุนุฒ ูุฌู", "ุฌู ุฌูุงูู", "ููุงู ุชุนุงูู",
"ุฐูุฑ ุงููู", "๏ดฟ",
]
HADITH_TRIGGERS = [
"ุฑุณูู ุงููู", "ุงููุจู", "ุตูู ุงููู ุนููู ูุณูู
", "๏ทบ", "ุนููู ุงูุตูุงุฉ ูุงูุณูุงู
", "ุญุฏูุซ", "ุฑูุงู", "ุฑูู", "ู
ุชูู ุนููู",
"ุงูุญุฏูุซ", "ููุงู", "ูุงู ุต", "ุตูู ุงููู ุนููู", "ูุณูู
",
]
_FORMULA_WORDS = {
normalize_for_matching(w)
for w in "ูุงู ูุงูุช ุฑุณูู ุงููู ุตูู ุนููู ูุณูู
ุงููุจู ุชุนุงูู ุณุจุญุงูู ุนุฒ ูุฌู ููุงู ูููู ุงููุฑูู
ุงูุดุฑูู ุงูุญุฏูุซ ุงูุขูุฉ ุฑูู ุฑูุงู ุนู ุฃู ุฃูู ุงูุจุฎุงุฑู ูู
ุณูู
".split()
}
_BRACKET_PAIRS = [("โ", "โ"), ("ยซ", "ยป"), ("๏ดฟ", "๏ดพ"), ("{", "}"), ("(", ")"), ("[", "]")]
class RuleDetector:
"""Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""
def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5,
quoted_min_cov_hadith: float = 0.75) -> None:
"""``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a
hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited."""
self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
self.decouple_triggers = decouple_triggers and retriever is not None
self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith
@staticmethod
def _segments(text: str):
segments = []
quote_positions = [m.start() for m in re.finditer('"', text)]
if len(quote_positions) % 2 == 0:
pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs
else: # a stray quote: fall back to every consecutive pair
pairs = zip(quote_positions, quote_positions[1:])
for a, b in pairs:
segments.append((a + 1, b))
for opener, closer in _BRACKET_PAIRS:
for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
segments.append((m.start(1), m.end(1)))
return segments
@staticmethod
def _trigger_type(context: str) -> Optional[str]:
best_end, best_label = -1, None
for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
for trigger in triggers:
pos = context.rfind(trigger)
if pos >= 0 and pos + len(trigger) > best_end:
best_end, best_label = pos + len(trigger), label
return best_label
def _corpus_coverage(self, span: str):
"""Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
if self.kb is None:
return 0.0, 0.0
quran_words = set(tokenize(normalize_strict(span)))
if not quran_words:
return 0.0, 0.0
quran_cov = max(
(len(quran_words & set(tokenize(normalize_strict(c["text"])))) / len(quran_words)
for c in self.kb.search_quran_ayahs(span, top_k=5)),
default=0.0,
)
hadith_words = set(tokenize(normalize_lenient(span)))
hadith_cov = max(
(len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
for c in self.kb.search_hadith(span, top_k=5)),
default=0.0,
) if hadith_words else 0.0
return quran_cov, hadith_cov
def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]:
"""Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose
coverage is too low for the corpus to decide on its own."""
quran_ok = quran_cov >= self.quoted_min_cov
hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them
if (quran_ok or hadith_ok) and n_words >= 4:
if trigger and abs(quran_cov - hadith_cov) < 0.1:
return trigger
if quran_ok and hadith_ok:
return "Ayah" if quran_cov >= hadith_cov else "Hadith"
return "Ayah" if quran_ok else "Hadith"
return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported
def detect(self, text: str) -> List[DetectedSpan]:
candidates = []
for start, end in self._segments(text):
start, end = trim_span(text, start, end)
if end <= start:
continue
inner = text[start:end]
words = [w for w in normalize_for_matching(inner).split() if w]
if len(words) < self.min_words or len(inner) > 3000:
continue
if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
continue
trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
quran_cov, hadith_cov = self._corpus_coverage(inner)
label = None
if self.decouple_triggers:
label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov)
elif trigger:
label = trigger
other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
if other >= 0.8 and mine < 0.5:
label = "Hadith" if trigger == "Ayah" else "Ayah"
elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
if label is None:
continue
candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label))
candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length
taken = []
for _, _, _, start, end, label in candidates:
if all(end <= t_start or start >= t_end for t_start, t_end, _ in taken):
taken.append((start, end, label))
taken.sort()
return [DetectedSpan(s, e, label, None, "rules", text[s:e]) for s, e, label in taken]
|