Spaces:
Running
Running
Download detector.py from Ghada-99-Ragab/lastfinal2: direct link, hf CLI and curl.
- Browser
- Download file 9.84 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/lastfinal2/resolve/main/detector.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/lastfinal2/detector.py
-
curl -L -o detector.py https://huggingface.co/spaces/Ghada-99-Ragab/lastfinal2/resolve/main/detector.py
9.84 kB
| """Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries. | |
| The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah / | |
| Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books). | |
| Introductory phrases such as "قال الله تعالى" or "قال رسول الله ﷺ" are only a tie-breaker / fallback hint, never a | |
| requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded.""" | |
| from __future__ import annotations | |
| import re | |
| from dataclasses import dataclass | |
| from typing import List, Optional | |
| from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize | |
| # -------------------------------------------------------------------------------------------------------------- | |
| class DetectedSpan: | |
| start: int | |
| end: int # exclusive | |
| label: str # 'Ayah' | 'Hadith' | |
| confidence: Optional[float] # None for the rule backend | |
| source: str # 'rules' | 'given' | |
| text: str = "" | |
| hint: Optional[str] = None # type suggested by an introductory phrase, if any (may disagree with ``label``) | |
| QUOTE_CHARS = " \t\r\n\"“”«»﴿﴾{}()[]" | |
| def trim_span(text: str, start: int, end: int): | |
| """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks).""" | |
| while start < end and text[start] in QUOTE_CHARS: | |
| start += 1 | |
| while end > start and text[end - 1] in QUOTE_CHARS + ".،,؛:": | |
| end -= 1 | |
| return start, end | |
| AYAH_TRIGGERS = [ | |
| "قال الله", "قوله تعالى", "قال تعالى", "يقول الله", "يقول تعالى", "قال سبحانه", "قوله سبحانه", "في كتابه", | |
| "سورة", "الآية", "الاية", "الآيات", "القرآن", "القران", "كتاب الله", "عز وجل", "جل جلاله", "فقال تعالى", | |
| "ذكر الله", "﴿", | |
| ] | |
| HADITH_TRIGGERS = [ | |
| "رسول الله", "النبي", "صلى الله عليه وسلم", "ﷺ", "عليه الصلاة والسلام", "حديث", "رواه", "روى", "متفق عليه", | |
| "الحديث", "فقال", "قال ص", "صلى الله عليه", "وسلم", | |
| ] | |
| _FORMULA_WORDS = { | |
| normalize_for_matching(w) | |
| for w in "قال قالت رسول الله صلى عليه وسلم النبي تعالى سبحانه عز وجل فقال يقول الكريم الشريف الحديث الآية روى رواه عن أن أنه البخاري ومسلم".split() | |
| } | |
| _BRACKET_PAIRS = [("“", "”"), ("«", "»"), ("﴿", "﴾"), ("{", "}"), ("(", ")"), ("[", "]")] | |
| class RuleDetector: | |
| """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU).""" | |
| def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110, | |
| min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5, | |
| quoted_min_cov_hadith: float = 0.75) -> None: | |
| """``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a | |
| hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited.""" | |
| self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov | |
| self.decouple_triggers = decouple_triggers and retriever is not None | |
| self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith | |
| def _segments(text: str): | |
| segments = [] | |
| quote_positions = [m.start() for m in re.finditer('"', text)] | |
| if len(quote_positions) % 2 == 0: | |
| pairs = zip(quote_positions[0::2], quote_positions[1::2]) # opening/closing pairs | |
| else: # a stray quote: fall back to every consecutive pair | |
| pairs = zip(quote_positions, quote_positions[1:]) | |
| for a, b in pairs: | |
| segments.append((a + 1, b)) | |
| for opener, closer in _BRACKET_PAIRS: | |
| for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S): | |
| segments.append((m.start(1), m.end(1))) | |
| return segments | |
| def _trigger_type(context: str) -> Optional[str]: | |
| best_end, best_label = -1, None | |
| for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)): | |
| for trigger in triggers: | |
| pos = context.rfind(trigger) | |
| if pos >= 0 and pos + len(trigger) > best_end: | |
| best_end, best_label = pos + len(trigger), label | |
| return best_label | |
| def _corpus_coverage(self, span: str): | |
| """Highest word coverage of the span by any top Quran ayah / Hadith candidate.""" | |
| if self.kb is None: | |
| return 0.0, 0.0 | |
| quran_words = set(tokenize(normalize_strict(span))) | |
| if not quran_words: | |
| return 0.0, 0.0 | |
| quran_cov = self._quran_window_coverage(quran_words, span) | |
| hadith_words = set(tokenize(normalize_lenient(span))) | |
| hadith_cov = max( | |
| (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words) | |
| for c in self.kb.search_hadith(span, top_k=5)), | |
| default=0.0, | |
| ) if hadith_words else 0.0 | |
| return quran_cov, hadith_cov | |
| def _quran_window_coverage(self, quran_words: set, span: str) -> float: | |
| """Best word coverage of the span by a single ayah or by 2-3 consecutive ayahs around a top candidate (a quotation | |
| often runs across an ayah boundary, and no single ayah then covers it).""" | |
| kb, best = self.kb, 0.0 | |
| for cand in kb.search_quran_ayahs(span, top_k=5): | |
| surah = kb.quran_by_surah.get(cand["surah_id"], {}) | |
| for first in range(cand["ayah_id"] - 2, cand["ayah_id"] + 1): | |
| words: set = set() | |
| for length in (1, 2, 3): | |
| idx = surah.get(first + length - 1) | |
| if idx is None: | |
| break | |
| ayah_words = set(tokenize(normalize_strict(kb.quran[idx]["text"]))) | |
| if length > 1 and not any(len(w) >= 4 for w in quran_words & ayah_words): | |
| break # every ayah of a multi-ayah window must contribute a real (not particle-like) word of the quotation | |
| words |= ayah_words | |
| if first <= cand["ayah_id"] <= first + length - 1: | |
| best = max(best, len(quran_words & words) / len(quran_words)) | |
| return best | |
| def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]: | |
| """Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose | |
| coverage is too low for the corpus to decide on its own.""" | |
| quran_ok = quran_cov >= self.quoted_min_cov | |
| hadith_ok = hadith_cov >= self.quoted_min_cov_hadith # Hadith records are long: stricter, ordinary prose overlaps them | |
| if (quran_ok or hadith_ok) and n_words >= 4: | |
| if trigger and ((quran_ok if trigger == "Ayah" else hadith_ok)) and abs(quran_cov - hadith_cov) < 0.25: | |
| return trigger | |
| if quran_ok and hadith_ok: | |
| return "Ayah" if quran_cov >= hadith_cov else "Hadith" | |
| return "Ayah" if quran_ok else "Hadith" | |
| return trigger # may be None: a delimited segment that matches nothing and has no hint is not reported | |
| def detect(self, text: str) -> List[DetectedSpan]: | |
| candidates = [] | |
| for start, end in self._segments(text): | |
| start, end = trim_span(text, start, end) | |
| if end <= start: | |
| continue | |
| inner = text[start:end] | |
| words = [w for w in normalize_for_matching(inner).split() if w] | |
| if len(words) < 2 or len(inner) > 3000: | |
| continue | |
| trigger = self._trigger_type(text[max(0, start - self.context_chars):start]) | |
| if len(words) < self.min_words and not trigger: # a very short quote is only taken after an introductory phrase | |
| continue | |
| if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6: | |
| continue | |
| quran_cov, hadith_cov = self._corpus_coverage(inner) | |
| label = None | |
| if self.decouple_triggers: | |
| label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov) | |
| elif trigger: | |
| label = trigger | |
| other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov) | |
| if other >= 0.8 and mine < 0.5: | |
| label = "Hadith" if trigger == "Ayah" else "Ayah" | |
| elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4: | |
| label = "Ayah" if quran_cov >= hadith_cov else "Hadith" | |
| if label is None: | |
| continue | |
| candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label, trigger)) | |
| candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True) # trigger first, then corpus match, then length | |
| taken = [] | |
| for _, _, _, start, end, label, hint in candidates: | |
| if all(end <= t_start or start >= t_end for t_start, t_end, _, _ in taken): | |
| taken.append((start, end, label, hint)) | |
| taken.sort() | |
| return [DetectedSpan(s, e, label, None, "rules", text[s:e], hint) for s, e, label, hint in taken] | |