lastfinal2 / index_builder.py
Ghada-99-Ragab's picture
Upload 26 files
0dff1a5 verified
Raw History Blame Contribute Delete
7.1 kB
"""Build the pre-tokenised search indexes shipped in ``index/``.
python index_builder.py # writes index/quran.idx.gz and index/hadith.idx.gz
An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the
application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read
back only by this project; never load an index file from an untrusted source.
"""
from __future__ import annotations
import gzip
import json
import math
import pickle
import sys
import time
import zlib
from array import array
from collections import Counter, defaultdict
from pathlib import Path
from typing import Dict, List, Sequence, Tuple
from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize
INDEX_VERSION = 3
BM25_K1, BM25_B = 1.2, 0.75
class BM25Index:
"""Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``."""
def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None:
self.postings, self.doc_len = postings, doc_len
self.n_docs = len(doc_len)
self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0
self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()}
def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]:
scores: Dict[int, float] = defaultdict(float)
for term in set(terms):
entry = self.postings.get(term)
if entry is None:
continue
idf = self.idf[term]
for doc, tf in zip(*entry):
norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len
scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm)
return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k]
def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]:
docs: Dict[str, array] = defaultdict(lambda: array("I"))
tfs: Dict[str, array] = defaultdict(lambda: array("H"))
doc_len = array("I")
for doc_id, tokens in enumerate(documents):
doc_len.append(len(tokens))
for term, tf in Counter(tokens).items():
docs[term].append(doc_id)
tfs[term].append(min(tf, 65535))
return {term: (docs[term], tfs[term]) for term in docs}, doc_len
def _read_json(path: Path) -> list:
opener = gzip.open if path.suffix == ".gz" else open
with opener(path, "rt", encoding="utf-8") as handle:
data = json.load(handle)
if not isinstance(data, list):
raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list")
return data
def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]:
"""Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped
trigrams that survive a single substituted or inserted word."""
keys = [phonetic_key(t) for t in tokens]
out = []
for i in range(len(keys) - 2):
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i))
if i + 3 < len(keys):
out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i))
out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i))
return out
def build_quran_index(path: Path) -> dict:
records, norm, word_count = [], [], []
all_index: Dict[str, List[int]] = defaultdict(list)
by_surah: Dict[int, Dict[int, int]] = defaultdict(dict)
documents: List[List[str]] = []
anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list)
for entry in _read_json(Path(path)):
text = (entry.get("ayah_text") or "").strip()
if not text:
continue
idx = len(records)
records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""),
"ayah_id": entry.get("ayah_id"), "text": text})
by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx
norm.append(normalize_for_matching(text))
strict_tokens = tokenize(normalize_strict(text))
word_count.append(len(strict_tokens))
for word in set(strict_tokens):
all_index[word].append(idx)
documents.append(content_words(norm[-1].split()))
for key, pos in anchor_keys(norm[-1].split()):
anchors[key].append((idx, pos))
postings, doc_len = _postings(documents)
return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count,
"all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len,
"anchors": dict(anchors)}
def build_hadith_index(path: Path) -> dict:
records, documents = [], []
for entry in _read_json(Path(path)):
if not entry:
continue
matn = (entry.get("Matn") or "").strip() or None
full = (entry.get("hadithTxt") or "").strip() or None
if not (matn or full):
continue
records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"),
"matn": matn, "full": full})
words = content_words(normalize_for_matching(full or matn).split())
if matn and full: # words of the matn that the full text may lack
extra = set(content_words(normalize_for_matching(matn).split())) - set(words)
words += sorted(extra)
documents.append(words)
postings, doc_len = _postings(documents)
return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len}
def save_index(index: dict, path: Path) -> None:
path = Path(path)
path.parent.mkdir(parents=True, exist_ok=True)
with gzip.open(path, "wb", compresslevel=6) as handle:
pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL)
def load_index(path: Path) -> dict:
with gzip.open(path, "rb") as handle:
index = pickle.load(handle) # produced by save_index() from this project's own data
if index.get("version") != INDEX_VERSION:
raise ValueError("Index version mismatch; rebuild with index_builder.py")
return index
def main() -> None:
base = Path(__file__).resolve().parent
out = base / "index"
for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"),
("hadith", build_hadith_index, base / "data" / "hadith.json")):
started = time.time()
source = source if source.is_file() else source.with_name(source.name + ".gz")
index = builder(source)
save_index(index, out / f"{name}.idx.gz")
size = (out / f"{name}.idx.gz").stat().st_size / 1e6
print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)")
if __name__ == "__main__":
sys.exit(main())