File size: 9,842 Bytes
0dff1a5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
"""Quotation detection (Subtask 1A): finds Quran / Hadith quotations in a text and their boundaries.

The detector is rule-based: quotation marks and brackets mark candidate segments. The *type* of a segment (Ayah /
Hadith) is decided by the reference corpora themselves (word coverage against the Quran and the six Hadith books).
Introductory phrases such as "ู‚ุงู„ ุงู„ู„ู‡ ุชุนุงู„ู‰" or "ู‚ุงู„ ุฑุณูˆู„ ุงู„ู„ู‡ ๏ทบ" are only a tie-breaker / fallback hint, never a
requirement: a quotation is found and typed even when no such phrase precedes it. No model weights are loaded."""
from __future__ import annotations

import re
from dataclasses import dataclass
from typing import List, Optional

from normalization import normalize_for_matching, normalize_lenient, normalize_strict, tokenize


# --------------------------------------------------------------------------------------------------------------
@dataclass
class DetectedSpan:
    start: int
    end: int                        # exclusive
    label: str                      # 'Ayah' | 'Hadith'
    confidence: Optional[float]     # None for the rule backend
    source: str                     # 'rules' | 'given'
    text: str = ""
    hint: Optional[str] = None      # type suggested by an introductory phrase, if any (may disagree with ``label``)


QUOTE_CHARS = " \t\r\n\"โ€œโ€ยซยป๏ดฟ๏ดพ{}()[]"


def trim_span(text: str, start: int, end: int):
    """Drop whitespace and quotation marks at both edges (gold spans exclude the quote marks)."""
    while start < end and text[start] in QUOTE_CHARS:
        start += 1
    while end > start and text[end - 1] in QUOTE_CHARS + ".ุŒ,ุ›:":
        end -= 1
    return start, end


AYAH_TRIGGERS = [
    "ู‚ุงู„ ุงู„ู„ู‡", "ู‚ูˆู„ู‡ ุชุนุงู„ู‰", "ู‚ุงู„ ุชุนุงู„ู‰", "ูŠู‚ูˆู„ ุงู„ู„ู‡", "ูŠู‚ูˆู„ ุชุนุงู„ู‰", "ู‚ุงู„ ุณุจุญุงู†ู‡", "ู‚ูˆู„ู‡ ุณุจุญุงู†ู‡", "ููŠ ูƒุชุงุจู‡",
    "ุณูˆุฑุฉ", "ุงู„ุขูŠุฉ", "ุงู„ุงูŠุฉ", "ุงู„ุขูŠุงุช", "ุงู„ู‚ุฑุขู†", "ุงู„ู‚ุฑุงู†", "ูƒุชุงุจ ุงู„ู„ู‡", "ุนุฒ ูˆุฌู„", "ุฌู„ ุฌู„ุงู„ู‡", "ูู‚ุงู„ ุชุนุงู„ู‰",
    "ุฐูƒุฑ ุงู„ู„ู‡", "๏ดฟ",
]
HADITH_TRIGGERS = [
    "ุฑุณูˆู„ ุงู„ู„ู‡", "ุงู„ู†ุจูŠ", "ุตู„ู‰ ุงู„ู„ู‡ ุนู„ูŠู‡ ูˆุณู„ู…", "๏ทบ", "ุนู„ูŠู‡ ุงู„ุตู„ุงุฉ ูˆุงู„ุณู„ุงู…", "ุญุฏูŠุซ", "ุฑูˆุงู‡", "ุฑูˆู‰", "ู…ุชูู‚ ุนู„ูŠู‡",
    "ุงู„ุญุฏูŠุซ", "ูู‚ุงู„", "ู‚ุงู„ ุต", "ุตู„ู‰ ุงู„ู„ู‡ ุนู„ูŠู‡", "ูˆุณู„ู…",
]
_FORMULA_WORDS = {
    normalize_for_matching(w)
    for w in "ู‚ุงู„ ู‚ุงู„ุช ุฑุณูˆู„ ุงู„ู„ู‡ ุตู„ู‰ ุนู„ูŠู‡ ูˆุณู„ู… ุงู„ู†ุจูŠ ุชุนุงู„ู‰ ุณุจุญุงู†ู‡ ุนุฒ ูˆุฌู„ ูู‚ุงู„ ูŠู‚ูˆู„ ุงู„ูƒุฑูŠู… ุงู„ุดุฑูŠู ุงู„ุญุฏูŠุซ ุงู„ุขูŠุฉ ุฑูˆู‰ ุฑูˆุงู‡ ุนู† ุฃู† ุฃู†ู‡ ุงู„ุจุฎุงุฑูŠ ูˆู…ุณู„ู…".split()
}
_BRACKET_PAIRS = [("โ€œ", "โ€"), ("ยซ", "ยป"), ("๏ดฟ", "๏ดพ"), ("{", "}"), ("(", ")"), ("[", "]")]


class RuleDetector:
    """Quotation-mark and trigger-phrase detector with an optional corpus lookup (no training, no GPU)."""

    def __init__(self, retriever=None, min_words: int = 3, context_chars: int = 110,
                 min_corpus_cov: float = 0.6, decouple_triggers: bool = True, quoted_min_cov: float = 0.5,
                 quoted_min_cov_hadith: float = 0.75) -> None:
        """``decouple_triggers``: type a delimited segment from the corpora first and use introductory phrases only as a
        hint (needs a retriever). ``quoted_min_cov``: lower coverage bar for text the author explicitly delimited."""
        self.kb, self.min_words, self.context_chars, self.min_corpus_cov = retriever, min_words, context_chars, min_corpus_cov
        self.decouple_triggers = decouple_triggers and retriever is not None
        self.quoted_min_cov, self.quoted_min_cov_hadith = quoted_min_cov, quoted_min_cov_hadith

    @staticmethod
    def _segments(text: str):
        segments = []
        quote_positions = [m.start() for m in re.finditer('"', text)]
        if len(quote_positions) % 2 == 0:
            pairs = zip(quote_positions[0::2], quote_positions[1::2])   # opening/closing pairs
        else:   # a stray quote: fall back to every consecutive pair
            pairs = zip(quote_positions, quote_positions[1:])
        for a, b in pairs:
            segments.append((a + 1, b))
        for opener, closer in _BRACKET_PAIRS:
            for m in re.finditer(re.escape(opener) + r"(.*?)" + re.escape(closer), text, re.S):
                segments.append((m.start(1), m.end(1)))
        return segments

    @staticmethod
    def _trigger_type(context: str) -> Optional[str]:
        best_end, best_label = -1, None
        for label, triggers in (("Ayah", AYAH_TRIGGERS), ("Hadith", HADITH_TRIGGERS)):
            for trigger in triggers:
                pos = context.rfind(trigger)
                if pos >= 0 and pos + len(trigger) > best_end:
                    best_end, best_label = pos + len(trigger), label
        return best_label

    def _corpus_coverage(self, span: str):
        """Highest word coverage of the span by any top Quran ayah / Hadith candidate."""
        if self.kb is None:
            return 0.0, 0.0
        quran_words = set(tokenize(normalize_strict(span)))
        if not quran_words:
            return 0.0, 0.0
        quran_cov = self._quran_window_coverage(quran_words, span)
        hadith_words = set(tokenize(normalize_lenient(span)))
        hadith_cov = max(
            (len(hadith_words & set(tokenize(normalize_lenient(c["text"])))) / len(hadith_words)
             for c in self.kb.search_hadith(span, top_k=5)),
            default=0.0,
        ) if hadith_words else 0.0
        return quran_cov, hadith_cov

    def _quran_window_coverage(self, quran_words: set, span: str) -> float:
        """Best word coverage of the span by a single ayah or by 2-3 consecutive ayahs around a top candidate (a quotation
        often runs across an ayah boundary, and no single ayah then covers it)."""
        kb, best = self.kb, 0.0
        for cand in kb.search_quran_ayahs(span, top_k=5):
            surah = kb.quran_by_surah.get(cand["surah_id"], {})
            for first in range(cand["ayah_id"] - 2, cand["ayah_id"] + 1):
                words: set = set()
                for length in (1, 2, 3):
                    idx = surah.get(first + length - 1)
                    if idx is None:
                        break
                    ayah_words = set(tokenize(normalize_strict(kb.quran[idx]["text"])))
                    if length > 1 and not any(len(w) >= 4 for w in quran_words & ayah_words):
                        break   # every ayah of a multi-ayah window must contribute a real (not particle-like) word of the quotation
                    words |= ayah_words
                    if first <= cand["ayah_id"] <= first + length - 1:
                        best = max(best, len(quran_words & words) / len(quran_words))
        return best

    def _label_from_corpus(self, n_words: int, trigger: Optional[str], quran_cov: float, hadith_cov: float) -> Optional[str]:
        """Corpus-first typing. The introductory phrase only breaks near-ties or types an altered quotation whose
        coverage is too low for the corpus to decide on its own."""
        quran_ok = quran_cov >= self.quoted_min_cov
        hadith_ok = hadith_cov >= self.quoted_min_cov_hadith   # Hadith records are long: stricter, ordinary prose overlaps them
        if (quran_ok or hadith_ok) and n_words >= 4:
            if trigger and ((quran_ok if trigger == "Ayah" else hadith_ok)) and abs(quran_cov - hadith_cov) < 0.25:
                return trigger
            if quran_ok and hadith_ok:
                return "Ayah" if quran_cov >= hadith_cov else "Hadith"
            return "Ayah" if quran_ok else "Hadith"
        return trigger   # may be None: a delimited segment that matches nothing and has no hint is not reported

    def detect(self, text: str) -> List[DetectedSpan]:
        candidates = []
        for start, end in self._segments(text):
            start, end = trim_span(text, start, end)
            if end <= start:
                continue
            inner = text[start:end]
            words = [w for w in normalize_for_matching(inner).split() if w]
            if len(words) < 2 or len(inner) > 3000:
                continue
            trigger = self._trigger_type(text[max(0, start - self.context_chars):start])
            if len(words) < self.min_words and not trigger:   # a very short quote is only taken after an introductory phrase
                continue
            if sum(w in _FORMULA_WORDS for w in words) / len(words) >= 0.6:
                continue
            quran_cov, hadith_cov = self._corpus_coverage(inner)
            label = None
            if self.decouple_triggers:
                label = self._label_from_corpus(len(words), trigger, quran_cov, hadith_cov)
            elif trigger:
                label = trigger
                other, mine = (hadith_cov, quran_cov) if trigger == "Ayah" else (quran_cov, hadith_cov)
                if other >= 0.8 and mine < 0.5:
                    label = "Hadith" if trigger == "Ayah" else "Ayah"
            elif max(quran_cov, hadith_cov) >= self.min_corpus_cov and len(words) >= 4:
                label = "Ayah" if quran_cov >= hadith_cov else "Hadith"
            if label is None:
                continue
            candidates.append((bool(trigger), max(quran_cov, hadith_cov), end - start, start, end, label, trigger))

        candidates.sort(key=lambda c: (c[0], c[1], c[2]), reverse=True)   # trigger first, then corpus match, then length
        taken = []
        for _, _, _, start, end, label, hint in candidates:
            if all(end <= t_start or start >= t_end for t_start, t_end, _, _ in taken):
                taken.append((start, end, label, hint))
        taken.sort()
        return [DetectedSpan(s, e, label, None, "rules", text[s:e], hint) for s, e, label, hint in taken]