Spaces:
Running
Running
File size: 7,098 Bytes
8f2ee72 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 | """Build the pre-tokenised search indexes shipped in ``index/``.
python index_builder.py # writes index/quran.idx.gz and index/hadith.idx.gz
An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the
application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read
back only by this project; never load an index file from an untrusted source.
"""
from __future__ import annotations
import gzip
import json
import math
import pickle
import sys
import time
import zlib
from array import array
from collections import Counter, defaultdict
from pathlib import Path
from typing import Dict, List, Sequence, Tuple
from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize
INDEX_VERSION = 3
BM25_K1, BM25_B = 1.2, 0.75
class BM25Index:
"""Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``."""
def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None:
self.postings, self.doc_len = postings, doc_len
self.n_docs = len(doc_len)
self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0
self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()}
def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]:
scores: Dict[int, float] = defaultdict(float)
for term in set(terms):
entry = self.postings.get(term)
if entry is None:
continue
idf = self.idf[term]
for doc, tf in zip(*entry):
norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len
scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm)
return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k]
def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]:
docs: Dict[str, array] = defaultdict(lambda: array("I"))
tfs: Dict[str, array] = defaultdict(lambda: array("H"))
doc_len = array("I")
for doc_id, tokens in enumerate(documents):
doc_len.append(len(tokens))
for term, tf in Counter(tokens).items():
docs[term].append(doc_id)
tfs[term].append(min(tf, 65535))
return {term: (docs[term], tfs[term]) for term in docs}, doc_len
def _read_json(path: Path) -> list:
opener = gzip.open if path.suffix == ".gz" else open
with opener(path, "rt", encoding="utf-8") as handle:
data = json.load(handle)
if not isinstance(data, list):
raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list")
return data
def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]:
"""Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped
trigrams that survive a single substituted or inserted word."""
keys = [phonetic_key(t) for t in tokens]
out = []
for i in range(len(keys) - 2):
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i))
if i + 3 < len(keys):
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i))
out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i))
return out
def build_quran_index(path: Path) -> dict:
records, norm, word_count = [], [], []
all_index: Dict[str, List[int]] = defaultdict(list)
by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
documents: List[List[str]] = []
anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
for entry in _read_json(Path(path)):
text = (entry.get("ayah_text") or "").strip()
if not text:
continue
idx = len(records)
records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""),
"ayah_id": entry.get("ayah_id"), "text": text})
by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
norm.append(normalize_for_matching(text))
strict_tokens = tokenize(normalize_strict(text))
word_count.append(len(strict_tokens))
for word in set(strict_tokens):
all_index[word].append(idx)
documents.append(content_words(norm[-1].split()))
for key, pos in anchor_keys(norm[-1].split()):
anchors[key].append((idx, pos))
postings, doc_len = _postings(documents)
return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count,
"all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len,
"anchors": dict(anchors)}
def build_hadith_index(path: Path) -> dict:
records, documents = [], []
for entry in _read_json(Path(path)):
if not entry:
continue
matn = (entry.get("Matn") or "").strip() or None
full = (entry.get("hadithTxt") or "").strip() or None
if not (matn or full):
continue
records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"),
"matn": matn, "full": full})
words = content_words(normalize_for_matching(full or matn).split())
if matn and full: # words of the matn that the full text may lack
extra = set(content_words(normalize_for_matching(matn).split())) - set(words)
words += sorted(extra)
documents.append(words)
postings, doc_len = _postings(documents)
return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len}
def save_index(index: dict, path: Path) -> None:
path = Path(path)
path.parent.mkdir(parents=True, exist_ok=True)
with gzip.open(path, "wb", compresslevel=6) as handle:
pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL)
def load_index(path: Path) -> dict:
with gzip.open(path, "rb") as handle:
index = pickle.load(handle) # produced by save_index() from this project's own data
if index.get("version") != INDEX_VERSION:
raise ValueError("Index version mismatch; rebuild with index_builder.py")
return index
def main() -> None:
base = Path(__file__).resolve().parent
out = base / "index"
for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"),
("hadith", build_hadith_index, base / "data" / "hadith.json")):
started = time.time()
source = source if source.is_file() else source.with_name(source.name + ".gz")
index = builder(source)
save_index(index, out / f"{name}.idx.gz")
size = (out / f"{name}.idx.gz").stat().st_size / 1e6
print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)")
if __name__ == "__main__":
sys.exit(main())
|