File size: 7,098 Bytes
8f2ee72
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
"""Build the pre-tokenised search indexes shipped in ``index/``.

    python index_builder.py            # writes index/quran.idx.gz and index/hadith.idx.gz

An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the
application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read
back only by this project; never load an index file from an untrusted source.
"""
from __future__ import annotations

import gzip
import json
import math
import pickle
import sys
import time
import zlib
from array import array
from collections import Counter, defaultdict
from pathlib import Path
from typing import Dict, List, Sequence, Tuple

from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize

INDEX_VERSION = 3
BM25_K1, BM25_B = 1.2, 0.75


class BM25Index:
    """Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``."""

    def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None:
        self.postings, self.doc_len = postings, doc_len
        self.n_docs = len(doc_len)
        self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0
        self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()}

    def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]:
        scores: Dict[int, float] = defaultdict(float)
        for term in set(terms):
            entry = self.postings.get(term)
            if entry is None:
                continue
            idf = self.idf[term]
            for doc, tf in zip(*entry):
                norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len
                scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm)
        return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k]


def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]:
    docs: Dict[str, array] = defaultdict(lambda: array("I"))
    tfs: Dict[str, array] = defaultdict(lambda: array("H"))
    doc_len = array("I")
    for doc_id, tokens in enumerate(documents):
        doc_len.append(len(tokens))
        for term, tf in Counter(tokens).items():
            docs[term].append(doc_id)
            tfs[term].append(min(tf, 65535))
    return {term: (docs[term], tfs[term]) for term in docs}, doc_len


def _read_json(path: Path) -> list:
    opener = gzip.open if path.suffix == ".gz" else open
    with opener(path, "rt", encoding="utf-8") as handle:
        data = json.load(handle)
    if not isinstance(data, list):
        raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list")
    return data



def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]:
    """Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped
    trigrams that survive a single substituted or inserted word."""
    keys = [phonetic_key(t) for t in tokens]
    out = []
    for i in range(len(keys) - 2):
        out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i))
        if i + 3 < len(keys):
            out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i))
            out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i))
    return out


def build_quran_index(path: Path) -> dict:
    records, norm, word_count = [], [], []
    all_index: Dict[str, List[int]] = defaultdict(list)
    by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
    documents: List[List[str]] = []
    anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
    for entry in _read_json(Path(path)):
        text = (entry.get("ayah_text") or "").strip()
        if not text:
            continue
        idx = len(records)
        records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""),
                        "ayah_id": entry.get("ayah_id"), "text": text})
        by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
        norm.append(normalize_for_matching(text))
        strict_tokens = tokenize(normalize_strict(text))
        word_count.append(len(strict_tokens))
        for word in set(strict_tokens):
            all_index[word].append(idx)
        documents.append(content_words(norm[-1].split()))
        for key, pos in anchor_keys(norm[-1].split()):
            anchors[key].append((idx, pos))
    postings, doc_len = _postings(documents)
    return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count,
            "all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len,
            "anchors": dict(anchors)}


def build_hadith_index(path: Path) -> dict:
    records, documents = [], []
    for entry in _read_json(Path(path)):
        if not entry:
            continue
        matn = (entry.get("Matn") or "").strip() or None
        full = (entry.get("hadithTxt") or "").strip() or None
        if not (matn or full):
            continue
        records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"),
                        "matn": matn, "full": full})
        words = content_words(normalize_for_matching(full or matn).split())
        if matn and full:   # words of the matn that the full text may lack
            extra = set(content_words(normalize_for_matching(matn).split())) - set(words)
            words += sorted(extra)
        documents.append(words)
    postings, doc_len = _postings(documents)
    return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len}


def save_index(index: dict, path: Path) -> None:
    path = Path(path)
    path.parent.mkdir(parents=True, exist_ok=True)
    with gzip.open(path, "wb", compresslevel=6) as handle:
        pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL)


def load_index(path: Path) -> dict:
    with gzip.open(path, "rb") as handle:
        index = pickle.load(handle)   # produced by save_index() from this project's own data
    if index.get("version") != INDEX_VERSION:
        raise ValueError("Index version mismatch; rebuild with index_builder.py")
    return index


def main() -> None:
    base = Path(__file__).resolve().parent
    out = base / "index"
    for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"),
                                  ("hadith", build_hadith_index, base / "data" / "hadith.json")):
        started = time.time()
        source = source if source.is_file() else source.with_name(source.name + ".gz")
        index = builder(source)
        save_index(index, out / f"{name}.idx.gz")
        size = (out / f"{name}.idx.gz").stat().st_size / 1e6
        print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)")


if __name__ == "__main__":
    sys.exit(main())