Download benchmark/lexical.py from WalidAlHassan/embeddingModelRnD: direct link, hf CLI and curl.
- Browser
- Download file 574 Bytes
-
https://huggingface.co/WalidAlHassan/embeddingModelRnD/resolve/main/benchmark/lexical.py
- Command line
-
hf download hf://WalidAlHassan/embeddingModelRnD/benchmark/lexical.py
-
curl -L -o lexical.py https://huggingface.co/WalidAlHassan/embeddingModelRnD/resolve/main/benchmark/lexical.py
574 Bytes
| """BM25 lexical index, shared across all embedding models (it doesn't depend on any of them).""" | |
| import numpy as np | |
| from rank_bm25 import BM25Okapi | |
| class BM25Index: | |
| def __init__(self, documents, book_ids): | |
| self.book_ids = book_ids | |
| tokenized = [doc.lower().split() for doc in documents] | |
| self.bm25 = BM25Okapi(tokenized) | |
| def search(self, query_text, top_k): | |
| scores = self.bm25.get_scores(query_text.lower().split()) | |
| ranked_indices = np.argsort(scores)[::-1][:top_k] | |
| return [self.book_ids[i] for i in ranked_indices] | |