Spaces:
Running
Running
Download index_builder.py from Ghada-99-Ragab/Finallll: direct link, hf CLI and curl.
- Browser
- Download file 7.1 kB
-
https://huggingface.co/spaces/Ghada-99-Ragab/Finallll/resolve/main/index_builder.py
- Command line
-
hf download hf://spaces/Ghada-99-Ragab/Finallll/index_builder.py
-
curl -L -o index_builder.py https://huggingface.co/spaces/Ghada-99-Ragab/Finallll/resolve/main/index_builder.py
7.1 kB
| """Build the pre-tokenised search indexes shipped in ``index/``. | |
| python index_builder.py # writes index/quran.idx.gz and index/hadith.idx.gz | |
| An index is a gzip-compressed pickle holding the raw records, their phonetic skeletons and BM25 postings, so the | |
| application never re-tokenises the corpora at start-up. The pickles are produced locally by this script and read | |
| back only by this project; never load an index file from an untrusted source. | |
| """ | |
| from __future__ import annotations | |
| import gzip | |
| import json | |
| import math | |
| import pickle | |
| import sys | |
| import time | |
| import zlib | |
| from array import array | |
| from collections import Counter, defaultdict | |
| from pathlib import Path | |
| from typing import Dict, List, Sequence, Tuple | |
| from normalization import content_words, normalize_for_matching, normalize_strict, phonetic_key, tokenize | |
| INDEX_VERSION = 3 | |
| BM25_K1, BM25_B = 1.2, 0.75 | |
| class BM25Index: | |
| """Okapi BM25 over a postings table ``term -> (doc ids, term frequencies)``.""" | |
| def __init__(self, postings: Dict[str, Tuple[array, array]], doc_len: array) -> None: | |
| self.postings, self.doc_len = postings, doc_len | |
| self.n_docs = len(doc_len) | |
| self.avg_len = (sum(doc_len) / self.n_docs) if self.n_docs else 1.0 | |
| self.idf = {t: math.log(1 + (self.n_docs - len(d) + 0.5) / (len(d) + 0.5)) for t, (d, _) in postings.items()} | |
| def search(self, terms: Sequence[str], top_k: int) -> List[Tuple[int, float]]: | |
| scores: Dict[int, float] = defaultdict(float) | |
| for term in set(terms): | |
| entry = self.postings.get(term) | |
| if entry is None: | |
| continue | |
| idf = self.idf[term] | |
| for doc, tf in zip(*entry): | |
| norm = 1 - BM25_B + BM25_B * self.doc_len[doc] / self.avg_len | |
| scores[doc] += idf * tf * (BM25_K1 + 1) / (tf + BM25_K1 * norm) | |
| return sorted(scores.items(), key=lambda item: item[1], reverse=True)[:top_k] | |
| def _postings(documents: List[List[str]]) -> Tuple[Dict[str, Tuple[array, array]], array]: | |
| docs: Dict[str, array] = defaultdict(lambda: array("I")) | |
| tfs: Dict[str, array] = defaultdict(lambda: array("H")) | |
| doc_len = array("I") | |
| for doc_id, tokens in enumerate(documents): | |
| doc_len.append(len(tokens)) | |
| for term, tf in Counter(tokens).items(): | |
| docs[term].append(doc_id) | |
| tfs[term].append(min(tf, 65535)) | |
| return {term: (docs[term], tfs[term]) for term in docs}, doc_len | |
| def _read_json(path: Path) -> list: | |
| opener = gzip.open if path.suffix == ".gz" else open | |
| with opener(path, "rt", encoding="utf-8") as handle: | |
| data = json.load(handle) | |
| if not isinstance(data, list): | |
| raise ValueError(f"Unexpected corpus format in {path}: expected a JSON list") | |
| return data | |
| def anchor_keys(tokens: Sequence[str]) -> List[Tuple[int, int]]: | |
| """Seed keys for unannounced-quotation search: ``(crc32, position)`` for every phonetic trigram plus two gapped | |
| trigrams that survive a single substituted or inserted word.""" | |
| keys = [phonetic_key(t) for t in tokens] | |
| out = [] | |
| for i in range(len(keys) - 2): | |
| out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} {keys[i + 2]}".encode()), i)) | |
| if i + 3 < len(keys): | |
| out.append((zlib.crc32(f"{keys[i]} {keys[i + 1]} _ {keys[i + 3]}".encode()), i)) | |
| out.append((zlib.crc32(f"{keys[i]} _ {keys[i + 2]} {keys[i + 3]}".encode()), i)) | |
| return out | |
| def build_quran_index(path: Path) -> dict: | |
| records, norm, word_count = [], [], [] | |
| all_index: Dict[str, List[int]] = defaultdict(list) | |
| by_surah: Dict[int, Dict[int, int]] = defaultdict(dict) | |
| documents: List[List[str]] = [] | |
| anchors: Dict[int, List[Tuple[int, int]]] = defaultdict(list) | |
| for entry in _read_json(Path(path)): | |
| text = (entry.get("ayah_text") or "").strip() | |
| if not text: | |
| continue | |
| idx = len(records) | |
| records.append({"surah_id": entry.get("surah_id"), "surah_name": entry.get("surah_name", ""), | |
| "ayah_id": entry.get("ayah_id"), "text": text}) | |
| by_surah[entry.get("surah_id")][entry.get("ayah_id")] = idx | |
| norm.append(normalize_for_matching(text)) | |
| strict_tokens = tokenize(normalize_strict(text)) | |
| word_count.append(len(strict_tokens)) | |
| for word in set(strict_tokens): | |
| all_index[word].append(idx) | |
| documents.append(content_words(norm[-1].split())) | |
| for key, pos in anchor_keys(norm[-1].split()): | |
| anchors[key].append((idx, pos)) | |
| postings, doc_len = _postings(documents) | |
| return {"version": INDEX_VERSION, "records": records, "norm": norm, "word_count": word_count, | |
| "all_index": dict(all_index), "by_surah": dict(by_surah), "postings": postings, "doc_len": doc_len, | |
| "anchors": dict(anchors)} | |
| def build_hadith_index(path: Path) -> dict: | |
| records, documents = [], [] | |
| for entry in _read_json(Path(path)): | |
| if not entry: | |
| continue | |
| matn = (entry.get("Matn") or "").strip() or None | |
| full = (entry.get("hadithTxt") or "").strip() or None | |
| if not (matn or full): | |
| continue | |
| records.append({"hadithID": entry.get("hadithID"), "book": entry.get("BookID"), "title": entry.get("title"), | |
| "matn": matn, "full": full}) | |
| words = content_words(normalize_for_matching(full or matn).split()) | |
| if matn and full: # words of the matn that the full text may lack | |
| extra = set(content_words(normalize_for_matching(matn).split())) - set(words) | |
| words += sorted(extra) | |
| documents.append(words) | |
| postings, doc_len = _postings(documents) | |
| return {"version": INDEX_VERSION, "records": records, "postings": postings, "doc_len": doc_len} | |
| def save_index(index: dict, path: Path) -> None: | |
| path = Path(path) | |
| path.parent.mkdir(parents=True, exist_ok=True) | |
| with gzip.open(path, "wb", compresslevel=6) as handle: | |
| pickle.dump(index, handle, protocol=pickle.HIGHEST_PROTOCOL) | |
| def load_index(path: Path) -> dict: | |
| with gzip.open(path, "rb") as handle: | |
| index = pickle.load(handle) # produced by save_index() from this project's own data | |
| if index.get("version") != INDEX_VERSION: | |
| raise ValueError("Index version mismatch; rebuild with index_builder.py") | |
| return index | |
| def main() -> None: | |
| base = Path(__file__).resolve().parent | |
| out = base / "index" | |
| for name, builder, source in (("quran", build_quran_index, base / "data" / "quran.json"), | |
| ("hadith", build_hadith_index, base / "data" / "hadith.json")): | |
| started = time.time() | |
| source = source if source.is_file() else source.with_name(source.name + ".gz") | |
| index = builder(source) | |
| save_index(index, out / f"{name}.idx.gz") | |
| size = (out / f"{name}.idx.gz").stat().st_size / 1e6 | |
| print(f"{name}: {len(index['records'])} records -> index/{name}.idx.gz ({size:.1f} MB, {time.time() - started:.1f}s)") | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |