Commit ·
6ced533
1
Parent(s): 6cfae5a
initial
Browse files- .gitignore +5 -0
- benchmark/__init__.py +0 -0
- benchmark/config.py +51 -0
- benchmark/data.py +46 -0
- benchmark/embedder.py +70 -0
- benchmark/fusion.py +10 -0
- benchmark/lexical.py +16 -0
- benchmark/metrics.py +54 -0
- benchmark/report.py +121 -0
- benchmark/runner.py +195 -0
- benchmark/vectorstore.py +69 -0
- boi1_eval_queries.json +61 -31
- run_benchmark.py +78 -0
.gitignore
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
__pycache__/
|
| 2 |
+
*.pyc
|
| 3 |
+
qdrant_data/
|
| 4 |
+
results/
|
| 5 |
+
.ipynb_checkpoints/
|
benchmark/__init__.py
ADDED
|
File without changes
|
benchmark/config.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Models under test and benchmark-wide settings.
|
| 2 |
+
|
| 3 |
+
`query_prefix` / `passage_prefix` are plain-text instruction prefixes prepended
|
| 4 |
+
before encoding (the E5 family expects "query: " / "passage: "; other model
|
| 5 |
+
families used here don't need one). This is applied manually rather than via
|
| 6 |
+
SentenceTransformer's `prompt_name=` API so behavior doesn't depend on whether
|
| 7 |
+
a given model repo happens to ship a config_sentence_transformers.json with
|
| 8 |
+
named prompts.
|
| 9 |
+
"""
|
| 10 |
+
|
| 11 |
+
MODELS = [
|
| 12 |
+
{
|
| 13 |
+
"name": "kazalbrur/bangla-embed-e5-small-banglish",
|
| 14 |
+
"query_prefix": "query: ",
|
| 15 |
+
"passage_prefix": "passage: ",
|
| 16 |
+
},
|
| 17 |
+
{
|
| 18 |
+
"name": "intfloat/multilingual-e5-small",
|
| 19 |
+
"query_prefix": "query: ",
|
| 20 |
+
"passage_prefix": "passage: ",
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"name": "intfloat/multilingual-e5-base",
|
| 24 |
+
"query_prefix": "query: ",
|
| 25 |
+
"passage_prefix": "passage: ",
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"name": "BAAI/bge-m3",
|
| 29 |
+
"query_prefix": "",
|
| 30 |
+
"passage_prefix": "",
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"name": "sentence-transformers/paraphrase-multilingual-mpnet-base-v2",
|
| 34 |
+
"query_prefix": "",
|
| 35 |
+
"passage_prefix": "",
|
| 36 |
+
},
|
| 37 |
+
]
|
| 38 |
+
|
| 39 |
+
QUERY_MODES = ["raw", "normalized"]
|
| 40 |
+
|
| 41 |
+
RETRIEVAL_MODES = ["dense", "hybrid"] # hybrid = BM25 + dense fused with RRF
|
| 42 |
+
|
| 43 |
+
TOP_K = 50 # depth retrieved per query; recall@10/@50 and nDCG@10 are sliced from this
|
| 44 |
+
FUSION_DEPTH = 50 # how deep each ranker (BM25, dense) is pulled before RRF fusion
|
| 45 |
+
RRF_K = 60 # RRF's rank-damping constant
|
| 46 |
+
|
| 47 |
+
BOOKS_PATH = "boi1_sample_books.json"
|
| 48 |
+
QUERIES_PATH = "boi1_eval_queries.json"
|
| 49 |
+
|
| 50 |
+
QDRANT_PATH = "./qdrant_data"
|
| 51 |
+
RESULTS_DIR = "./results"
|
benchmark/data.py
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Loading book corpus and evaluation queries."""
|
| 2 |
+
|
| 3 |
+
import json
|
| 4 |
+
from dataclasses import dataclass, field
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
@dataclass
|
| 8 |
+
class EvalQuery:
|
| 9 |
+
query: str
|
| 10 |
+
normalized_query: str
|
| 11 |
+
relevance: dict # book_id -> graded relevance (1-3)
|
| 12 |
+
categories: list = field(default_factory=list)
|
| 13 |
+
|
| 14 |
+
def text_for_mode(self, mode):
|
| 15 |
+
if mode == "normalized":
|
| 16 |
+
return self.normalized_query
|
| 17 |
+
return self.query
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def load_books(path):
|
| 21 |
+
with open(path, "r", encoding="utf-8") as f:
|
| 22 |
+
return json.load(f)
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def load_queries(path):
|
| 26 |
+
with open(path, "r", encoding="utf-8") as f:
|
| 27 |
+
raw = json.load(f)
|
| 28 |
+
return [
|
| 29 |
+
EvalQuery(
|
| 30 |
+
query=item["query"],
|
| 31 |
+
normalized_query=item.get("normalized_query", item["query"]),
|
| 32 |
+
relevance=item["relevance"],
|
| 33 |
+
categories=item.get("categories", []),
|
| 34 |
+
)
|
| 35 |
+
for item in raw
|
| 36 |
+
]
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def book_to_text(book):
|
| 40 |
+
return (
|
| 41 |
+
f"Title: {book['title']}\n"
|
| 42 |
+
f"Author: {book['author']}\n"
|
| 43 |
+
f"Genre: {', '.join(book['genres'])}\n"
|
| 44 |
+
f"Tags: {', '.join(book['tags'])}\n"
|
| 45 |
+
f"Description: {book['description']}\n"
|
| 46 |
+
)
|
benchmark/embedder.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Wraps a SentenceTransformer model with prefixing and timing instrumentation."""
|
| 2 |
+
|
| 3 |
+
import time
|
| 4 |
+
|
| 5 |
+
import torch
|
| 6 |
+
from sentence_transformers import SentenceTransformer
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
class TimedEmbedder:
|
| 10 |
+
def __init__(self, model_name, query_prefix="", passage_prefix="", device=None):
|
| 11 |
+
self.model_name = model_name
|
| 12 |
+
self.query_prefix = query_prefix
|
| 13 |
+
self.passage_prefix = passage_prefix
|
| 14 |
+
self.device = device or ("cuda" if torch.cuda.is_available() else "cpu")
|
| 15 |
+
|
| 16 |
+
start = time.perf_counter()
|
| 17 |
+
self.model = SentenceTransformer(model_name, device=self.device)
|
| 18 |
+
self.load_time_sec = time.perf_counter() - start
|
| 19 |
+
|
| 20 |
+
self.param_count = sum(p.numel() for p in self.model.parameters())
|
| 21 |
+
self.model_size_mb = sum(
|
| 22 |
+
p.numel() * p.element_size() for p in self.model.parameters()
|
| 23 |
+
) / (1024 ** 2)
|
| 24 |
+
|
| 25 |
+
if hasattr(self.model, "get_embedding_dimension"):
|
| 26 |
+
self.embedding_dim = self.model.get_embedding_dimension()
|
| 27 |
+
else:
|
| 28 |
+
self.embedding_dim = self.model.get_sentence_embedding_dimension()
|
| 29 |
+
|
| 30 |
+
def encode_passages(self, texts, batch_size=32):
|
| 31 |
+
"""Batch-encode documents; returns (embeddings, elapsed_seconds)."""
|
| 32 |
+
prefixed = [self.passage_prefix + t for t in texts]
|
| 33 |
+
start = time.perf_counter()
|
| 34 |
+
embeddings = self.model.encode(
|
| 35 |
+
prefixed,
|
| 36 |
+
batch_size=batch_size,
|
| 37 |
+
normalize_embeddings=True,
|
| 38 |
+
convert_to_numpy=True,
|
| 39 |
+
show_progress_bar=False,
|
| 40 |
+
)
|
| 41 |
+
elapsed = time.perf_counter() - start
|
| 42 |
+
return embeddings, elapsed
|
| 43 |
+
|
| 44 |
+
def encode_query(self, text):
|
| 45 |
+
"""Single-query encode; returns (embedding, elapsed_seconds)."""
|
| 46 |
+
prefixed = self.query_prefix + text
|
| 47 |
+
start = time.perf_counter()
|
| 48 |
+
embedding = self.model.encode(
|
| 49 |
+
prefixed,
|
| 50 |
+
normalize_embeddings=True,
|
| 51 |
+
convert_to_numpy=True,
|
| 52 |
+
show_progress_bar=False,
|
| 53 |
+
)
|
| 54 |
+
elapsed = time.perf_counter() - start
|
| 55 |
+
return embedding, elapsed
|
| 56 |
+
|
| 57 |
+
def peak_encode_memory_mb(self, sample_text):
|
| 58 |
+
"""Peak accelerator memory used while encoding one query (CUDA only, else None)."""
|
| 59 |
+
if self.device != "cuda":
|
| 60 |
+
return None
|
| 61 |
+
torch.cuda.synchronize()
|
| 62 |
+
torch.cuda.reset_peak_memory_stats(self.device)
|
| 63 |
+
self.encode_query(sample_text)
|
| 64 |
+
torch.cuda.synchronize()
|
| 65 |
+
return torch.cuda.max_memory_allocated(self.device) / (1024 ** 2)
|
| 66 |
+
|
| 67 |
+
def unload(self):
|
| 68 |
+
del self.model
|
| 69 |
+
if self.device == "cuda":
|
| 70 |
+
torch.cuda.empty_cache()
|
benchmark/fusion.py
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Reciprocal Rank Fusion for combining multiple ranked-id lists."""
|
| 2 |
+
|
| 3 |
+
|
| 4 |
+
def reciprocal_rank_fusion(rankings, k=60):
|
| 5 |
+
"""rankings: list of ranked-id lists (best first). Returns a single fused ranking."""
|
| 6 |
+
scores = {}
|
| 7 |
+
for ranking in rankings:
|
| 8 |
+
for rank, doc_id in enumerate(ranking, start=1):
|
| 9 |
+
scores[doc_id] = scores.get(doc_id, 0.0) + 1.0 / (k + rank)
|
| 10 |
+
return sorted(scores, key=scores.get, reverse=True)
|
benchmark/lexical.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""BM25 lexical index, shared across all embedding models (it doesn't depend on any of them)."""
|
| 2 |
+
|
| 3 |
+
import numpy as np
|
| 4 |
+
from rank_bm25 import BM25Okapi
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
class BM25Index:
|
| 8 |
+
def __init__(self, documents, book_ids):
|
| 9 |
+
self.book_ids = book_ids
|
| 10 |
+
tokenized = [doc.lower().split() for doc in documents]
|
| 11 |
+
self.bm25 = BM25Okapi(tokenized)
|
| 12 |
+
|
| 13 |
+
def search(self, query_text, top_k):
|
| 14 |
+
scores = self.bm25.get_scores(query_text.lower().split())
|
| 15 |
+
ranked_indices = np.argsort(scores)[::-1][:top_k]
|
| 16 |
+
return [self.book_ids[i] for i in ranked_indices]
|
benchmark/metrics.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Retrieval quality metrics for graded relevance judgments."""
|
| 2 |
+
|
| 3 |
+
import math
|
| 4 |
+
|
| 5 |
+
|
| 6 |
+
def recall_at_k(retrieved_ids, relevance, k):
|
| 7 |
+
"""Fraction of all known-relevant items that appear in the top-k results."""
|
| 8 |
+
relevant = {doc_id for doc_id, rel in relevance.items() if rel > 0}
|
| 9 |
+
if not relevant:
|
| 10 |
+
return None
|
| 11 |
+
top_k = set(retrieved_ids[:k])
|
| 12 |
+
return len(top_k & relevant) / len(relevant)
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def reciprocal_rank(retrieved_ids, relevance):
|
| 16 |
+
"""1 / rank of the first retrieved item with relevance > 0, else 0."""
|
| 17 |
+
relevant = {doc_id for doc_id, rel in relevance.items() if rel > 0}
|
| 18 |
+
for rank, doc_id in enumerate(retrieved_ids, start=1):
|
| 19 |
+
if doc_id in relevant:
|
| 20 |
+
return 1.0 / rank
|
| 21 |
+
return 0.0
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _dcg(gains):
|
| 25 |
+
return sum(gain / math.log2(pos + 1) for pos, gain in enumerate(gains, start=1))
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def ndcg_at_k(retrieved_ids, relevance, k):
|
| 29 |
+
"""Normalized DCG@k using the standard 2^rel - 1 gain function."""
|
| 30 |
+
gains = [
|
| 31 |
+
(2 ** relevance.get(doc_id, 0)) - 1
|
| 32 |
+
for doc_id in retrieved_ids[:k]
|
| 33 |
+
]
|
| 34 |
+
dcg = _dcg(gains)
|
| 35 |
+
|
| 36 |
+
ideal_gains = sorted(
|
| 37 |
+
((2 ** rel) - 1 for rel in relevance.values()),
|
| 38 |
+
reverse=True,
|
| 39 |
+
)[:k]
|
| 40 |
+
idcg = _dcg(ideal_gains)
|
| 41 |
+
|
| 42 |
+
if idcg == 0:
|
| 43 |
+
return 0.0
|
| 44 |
+
return dcg / idcg
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
def evaluate_ranking(retrieved_ids, relevance):
|
| 48 |
+
"""Compute all retrieval-quality metrics for a single query's ranking."""
|
| 49 |
+
return {
|
| 50 |
+
"recall@10": recall_at_k(retrieved_ids, relevance, 10),
|
| 51 |
+
"recall@50": recall_at_k(retrieved_ids, relevance, 50),
|
| 52 |
+
"mrr": reciprocal_rank(retrieved_ids, relevance),
|
| 53 |
+
"ndcg@10": ndcg_at_k(retrieved_ids, relevance, 10),
|
| 54 |
+
}
|
benchmark/report.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Flattens raw benchmark results into JSON + CSV artifacts."""
|
| 2 |
+
|
| 3 |
+
import csv
|
| 4 |
+
import json
|
| 5 |
+
import os
|
| 6 |
+
|
| 7 |
+
MODEL_FIELDS = [
|
| 8 |
+
"model",
|
| 9 |
+
"embedding_dim",
|
| 10 |
+
"model_size_mb",
|
| 11 |
+
"param_count",
|
| 12 |
+
"load_time_sec",
|
| 13 |
+
"doc_embed_time_sec",
|
| 14 |
+
"doc_throughput_docs_per_sec",
|
| 15 |
+
"raw_vector_storage_mb",
|
| 16 |
+
"qdrant_collection_disk_mb",
|
| 17 |
+
"peak_query_encode_memory_mb",
|
| 18 |
+
]
|
| 19 |
+
|
| 20 |
+
MODE_FIELDS = [
|
| 21 |
+
"mean_embedding_latency_ms",
|
| 22 |
+
"mean_retrieval_latency_ms",
|
| 23 |
+
"queries_per_sec",
|
| 24 |
+
"recall@10",
|
| 25 |
+
"recall@50",
|
| 26 |
+
"mrr",
|
| 27 |
+
"ndcg@10",
|
| 28 |
+
]
|
| 29 |
+
|
| 30 |
+
SUMMARY_FIELDS = MODEL_FIELDS[:1] + ["retrieval_mode", "query_mode"] + MODEL_FIELDS[1:] + MODE_FIELDS
|
| 31 |
+
|
| 32 |
+
CATEGORY_FIELDS = [
|
| 33 |
+
"model",
|
| 34 |
+
"retrieval_mode",
|
| 35 |
+
"query_mode",
|
| 36 |
+
"category",
|
| 37 |
+
"n",
|
| 38 |
+
"recall@10",
|
| 39 |
+
"recall@50",
|
| 40 |
+
"mrr",
|
| 41 |
+
"ndcg@10",
|
| 42 |
+
]
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def _summary_rows(results):
|
| 46 |
+
rows = []
|
| 47 |
+
for model_result in results:
|
| 48 |
+
for retrieval_mode, query_mode_results in model_result["retrieval_modes"].items():
|
| 49 |
+
for query_mode, mode_result in query_mode_results.items():
|
| 50 |
+
row = {k: model_result.get(k) for k in MODEL_FIELDS}
|
| 51 |
+
row["retrieval_mode"] = retrieval_mode
|
| 52 |
+
row["query_mode"] = query_mode
|
| 53 |
+
for metric in MODE_FIELDS:
|
| 54 |
+
row[metric] = mode_result.get(metric)
|
| 55 |
+
rows.append(row)
|
| 56 |
+
return rows
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def _category_rows(results):
|
| 60 |
+
rows = []
|
| 61 |
+
for model_result in results:
|
| 62 |
+
for retrieval_mode, query_mode_results in model_result["retrieval_modes"].items():
|
| 63 |
+
for query_mode, mode_result in query_mode_results.items():
|
| 64 |
+
for category, stats in mode_result["per_category"].items():
|
| 65 |
+
rows.append(
|
| 66 |
+
{
|
| 67 |
+
"model": model_result["model"],
|
| 68 |
+
"retrieval_mode": retrieval_mode,
|
| 69 |
+
"query_mode": query_mode,
|
| 70 |
+
"category": category,
|
| 71 |
+
**stats,
|
| 72 |
+
}
|
| 73 |
+
)
|
| 74 |
+
return rows
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def _write_csv(path, fieldnames, rows):
|
| 78 |
+
with open(path, "w", newline="", encoding="utf-8") as f:
|
| 79 |
+
writer = csv.DictWriter(f, fieldnames=fieldnames)
|
| 80 |
+
writer.writeheader()
|
| 81 |
+
for row in rows:
|
| 82 |
+
writer.writerow(row)
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def save_results(results, results_dir):
|
| 86 |
+
os.makedirs(results_dir, exist_ok=True)
|
| 87 |
+
|
| 88 |
+
raw_path = os.path.join(results_dir, "raw_results.json")
|
| 89 |
+
with open(raw_path, "w", encoding="utf-8") as f:
|
| 90 |
+
json.dump(results, f, ensure_ascii=False, indent=2)
|
| 91 |
+
|
| 92 |
+
summary_rows = _summary_rows(results)
|
| 93 |
+
summary_path = os.path.join(results_dir, "summary.csv")
|
| 94 |
+
_write_csv(summary_path, SUMMARY_FIELDS, summary_rows)
|
| 95 |
+
|
| 96 |
+
category_rows = _category_rows(results)
|
| 97 |
+
category_path = os.path.join(results_dir, "summary_by_category.csv")
|
| 98 |
+
_write_csv(category_path, CATEGORY_FIELDS, category_rows)
|
| 99 |
+
|
| 100 |
+
return {
|
| 101 |
+
"raw_results": raw_path,
|
| 102 |
+
"summary": summary_path,
|
| 103 |
+
"summary_by_category": category_path,
|
| 104 |
+
}
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def print_summary_table(results):
|
| 108 |
+
header = (
|
| 109 |
+
f"{'model':<55} {'retrieval':<8} {'mode':<11} {'dim':>5} {'size(MB)':>9} "
|
| 110 |
+
f"{'R@10':>6} {'R@50':>6} {'MRR':>6} {'nDCG@10':>8} {'lat(ms)':>8} {'q/s':>7}"
|
| 111 |
+
)
|
| 112 |
+
print(header)
|
| 113 |
+
print("-" * len(header))
|
| 114 |
+
for row in _summary_rows(results):
|
| 115 |
+
print(
|
| 116 |
+
f"{row['model']:<55} {row['retrieval_mode']:<8} {row['query_mode']:<11} "
|
| 117 |
+
f"{row['embedding_dim']:>5} {row['model_size_mb']:>9.1f} "
|
| 118 |
+
f"{row['recall@10']:>6.3f} {row['recall@50']:>6.3f} {row['mrr']:>6.3f} "
|
| 119 |
+
f"{row['ndcg@10']:>8.3f} {row['mean_retrieval_latency_ms']:>8.1f} "
|
| 120 |
+
f"{row['queries_per_sec']:>7.1f}"
|
| 121 |
+
)
|
benchmark/runner.py
ADDED
|
@@ -0,0 +1,195 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Orchestrates the per-model, per-retrieval-mode, per-query-mode benchmark run."""
|
| 2 |
+
|
| 3 |
+
import statistics
|
| 4 |
+
import time
|
| 5 |
+
|
| 6 |
+
from . import config
|
| 7 |
+
from .data import book_to_text, load_books, load_queries
|
| 8 |
+
from .embedder import TimedEmbedder
|
| 9 |
+
from .fusion import reciprocal_rank_fusion
|
| 10 |
+
from .lexical import BM25Index
|
| 11 |
+
from .metrics import evaluate_ranking
|
| 12 |
+
from .vectorstore import BookVectorStore
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def _mean(values):
|
| 16 |
+
values = [v for v in values if v is not None]
|
| 17 |
+
return statistics.mean(values) if values else None
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
def _retrieve(retrieval_mode, embedder, store, bm25_index, model_name, query_embedding, query_text, top_k, fusion_depth, rrf_k):
|
| 21 |
+
"""Returns (retrieved_book_ids, extra_retrieval_seconds) where extra excludes embedding time."""
|
| 22 |
+
start = time.perf_counter()
|
| 23 |
+
if retrieval_mode == "dense":
|
| 24 |
+
retrieved = store.search(model_name, query_embedding, top_k)
|
| 25 |
+
elif retrieval_mode == "hybrid":
|
| 26 |
+
bm25_ranked = bm25_index.search(query_text, fusion_depth)
|
| 27 |
+
dense_ranked = store.search(model_name, query_embedding, fusion_depth)
|
| 28 |
+
fused = reciprocal_rank_fusion([bm25_ranked, dense_ranked], k=rrf_k)
|
| 29 |
+
retrieved = fused[:top_k]
|
| 30 |
+
else:
|
| 31 |
+
raise ValueError(f"Unknown retrieval_mode: {retrieval_mode}")
|
| 32 |
+
elapsed = time.perf_counter() - start
|
| 33 |
+
return retrieved, elapsed
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def _evaluate(embedder, store, bm25_index, model_name, queries, retrieval_mode, query_mode, top_k, fusion_depth, rrf_k):
|
| 37 |
+
per_query = []
|
| 38 |
+
embed_latencies_sec = []
|
| 39 |
+
retrieval_latencies_sec = []
|
| 40 |
+
|
| 41 |
+
for q in queries:
|
| 42 |
+
text = q.text_for_mode(query_mode)
|
| 43 |
+
embedding, embed_latency = embedder.encode_query(text)
|
| 44 |
+
retrieved, extra_latency = _retrieve(
|
| 45 |
+
retrieval_mode, embedder, store, bm25_index, model_name,
|
| 46 |
+
embedding, text, top_k, fusion_depth, rrf_k,
|
| 47 |
+
)
|
| 48 |
+
total_latency = embed_latency + extra_latency
|
| 49 |
+
|
| 50 |
+
embed_latencies_sec.append(embed_latency)
|
| 51 |
+
retrieval_latencies_sec.append(total_latency)
|
| 52 |
+
|
| 53 |
+
metrics = evaluate_ranking(retrieved, q.relevance)
|
| 54 |
+
|
| 55 |
+
per_query.append(
|
| 56 |
+
{
|
| 57 |
+
"query": text,
|
| 58 |
+
"categories": q.categories,
|
| 59 |
+
"embedding_latency_ms": embed_latency * 1000,
|
| 60 |
+
"retrieval_latency_ms": total_latency * 1000,
|
| 61 |
+
**metrics,
|
| 62 |
+
}
|
| 63 |
+
)
|
| 64 |
+
|
| 65 |
+
per_category = {}
|
| 66 |
+
for entry in per_query:
|
| 67 |
+
for cat in entry["categories"]:
|
| 68 |
+
bucket = per_category.setdefault(
|
| 69 |
+
cat, {"n": 0, "recall@10": [], "recall@50": [], "mrr": [], "ndcg@10": []}
|
| 70 |
+
)
|
| 71 |
+
bucket["n"] += 1
|
| 72 |
+
for metric in ("recall@10", "recall@50", "mrr", "ndcg@10"):
|
| 73 |
+
bucket[metric].append(entry[metric])
|
| 74 |
+
|
| 75 |
+
per_category_summary = {
|
| 76 |
+
cat: {
|
| 77 |
+
"n": bucket["n"],
|
| 78 |
+
"recall@10": _mean(bucket["recall@10"]),
|
| 79 |
+
"recall@50": _mean(bucket["recall@50"]),
|
| 80 |
+
"mrr": _mean(bucket["mrr"]),
|
| 81 |
+
"ndcg@10": _mean(bucket["ndcg@10"]),
|
| 82 |
+
}
|
| 83 |
+
for cat, bucket in per_category.items()
|
| 84 |
+
}
|
| 85 |
+
|
| 86 |
+
mean_embed_latency_sec = _mean(embed_latencies_sec)
|
| 87 |
+
mean_retrieval_latency_sec = _mean(retrieval_latencies_sec)
|
| 88 |
+
|
| 89 |
+
return {
|
| 90 |
+
"mean_embedding_latency_ms": mean_embed_latency_sec * 1000 if mean_embed_latency_sec else None,
|
| 91 |
+
"mean_retrieval_latency_ms": mean_retrieval_latency_sec * 1000 if mean_retrieval_latency_sec else None,
|
| 92 |
+
"queries_per_sec": (1.0 / mean_retrieval_latency_sec) if mean_retrieval_latency_sec else None,
|
| 93 |
+
"recall@10": _mean(e["recall@10"] for e in per_query),
|
| 94 |
+
"recall@50": _mean(e["recall@50"] for e in per_query),
|
| 95 |
+
"mrr": _mean(e["mrr"] for e in per_query),
|
| 96 |
+
"ndcg@10": _mean(e["ndcg@10"] for e in per_query),
|
| 97 |
+
"per_query": per_query,
|
| 98 |
+
"per_category": per_category_summary,
|
| 99 |
+
}
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
def run_benchmark(
|
| 103 |
+
models=None,
|
| 104 |
+
query_modes=None,
|
| 105 |
+
retrieval_modes=None,
|
| 106 |
+
top_k=None,
|
| 107 |
+
fusion_depth=None,
|
| 108 |
+
rrf_k=None,
|
| 109 |
+
books_path=None,
|
| 110 |
+
queries_path=None,
|
| 111 |
+
qdrant_path=None,
|
| 112 |
+
):
|
| 113 |
+
models = models if models is not None else config.MODELS
|
| 114 |
+
query_modes = query_modes if query_modes is not None else config.QUERY_MODES
|
| 115 |
+
retrieval_modes = retrieval_modes if retrieval_modes is not None else config.RETRIEVAL_MODES
|
| 116 |
+
top_k = top_k or config.TOP_K
|
| 117 |
+
fusion_depth = fusion_depth or config.FUSION_DEPTH
|
| 118 |
+
rrf_k = rrf_k if rrf_k is not None else config.RRF_K
|
| 119 |
+
|
| 120 |
+
books = load_books(books_path or config.BOOKS_PATH)
|
| 121 |
+
queries = load_queries(queries_path or config.QUERIES_PATH)
|
| 122 |
+
store = BookVectorStore(qdrant_path or config.QDRANT_PATH)
|
| 123 |
+
|
| 124 |
+
documents = [book_to_text(b) for b in books]
|
| 125 |
+
book_ids = [b["book_id"] for b in books]
|
| 126 |
+
num_docs = len(books)
|
| 127 |
+
|
| 128 |
+
bm25_index = BM25Index(documents, book_ids) if "hybrid" in retrieval_modes else None
|
| 129 |
+
|
| 130 |
+
results = []
|
| 131 |
+
|
| 132 |
+
for model_cfg in models:
|
| 133 |
+
model_name = model_cfg["name"]
|
| 134 |
+
print(f"\n=== {model_name} ===")
|
| 135 |
+
|
| 136 |
+
embedder = TimedEmbedder(
|
| 137 |
+
model_name,
|
| 138 |
+
query_prefix=model_cfg.get("query_prefix", ""),
|
| 139 |
+
passage_prefix=model_cfg.get("passage_prefix", ""),
|
| 140 |
+
)
|
| 141 |
+
print(
|
| 142 |
+
f" loaded in {embedder.load_time_sec:.2f}s | "
|
| 143 |
+
f"dim={embedder.embedding_dim} | size={embedder.model_size_mb:.1f} MB "
|
| 144 |
+
f"| device={embedder.device}"
|
| 145 |
+
)
|
| 146 |
+
|
| 147 |
+
doc_embeddings, doc_embed_time = embedder.encode_passages(documents)
|
| 148 |
+
doc_throughput = num_docs / doc_embed_time if doc_embed_time else None
|
| 149 |
+
print(
|
| 150 |
+
f" embedded {num_docs} docs in {doc_embed_time:.2f}s "
|
| 151 |
+
f"({doc_throughput:.1f} docs/sec)"
|
| 152 |
+
)
|
| 153 |
+
|
| 154 |
+
store.index(model_name, doc_embeddings, books)
|
| 155 |
+
|
| 156 |
+
raw_vector_bytes = embedder.embedding_dim * 4 * num_docs
|
| 157 |
+
qdrant_disk_mb = store.collection_disk_size_mb(model_name)
|
| 158 |
+
peak_mem_mb = embedder.peak_encode_memory_mb(queries[0].query)
|
| 159 |
+
|
| 160 |
+
model_result = {
|
| 161 |
+
"model": model_name,
|
| 162 |
+
"device": embedder.device,
|
| 163 |
+
"embedding_dim": embedder.embedding_dim,
|
| 164 |
+
"param_count": embedder.param_count,
|
| 165 |
+
"model_size_mb": embedder.model_size_mb,
|
| 166 |
+
"load_time_sec": embedder.load_time_sec,
|
| 167 |
+
"doc_embed_time_sec": doc_embed_time,
|
| 168 |
+
"doc_throughput_docs_per_sec": doc_throughput,
|
| 169 |
+
"raw_vector_storage_mb": raw_vector_bytes / (1024 ** 2),
|
| 170 |
+
"qdrant_collection_disk_mb": qdrant_disk_mb,
|
| 171 |
+
"peak_query_encode_memory_mb": peak_mem_mb,
|
| 172 |
+
"retrieval_modes": {},
|
| 173 |
+
}
|
| 174 |
+
|
| 175 |
+
for retrieval_mode in retrieval_modes:
|
| 176 |
+
model_result["retrieval_modes"][retrieval_mode] = {}
|
| 177 |
+
for query_mode in query_modes:
|
| 178 |
+
print(f" evaluating retrieval_mode={retrieval_mode} query_mode={query_mode} ...")
|
| 179 |
+
mode_result = _evaluate(
|
| 180 |
+
embedder, store, bm25_index, model_name, queries,
|
| 181 |
+
retrieval_mode, query_mode, top_k, fusion_depth, rrf_k,
|
| 182 |
+
)
|
| 183 |
+
model_result["retrieval_modes"][retrieval_mode][query_mode] = mode_result
|
| 184 |
+
print(
|
| 185 |
+
f" recall@10={mode_result['recall@10']:.3f} "
|
| 186 |
+
f"recall@50={mode_result['recall@50']:.3f} "
|
| 187 |
+
f"mrr={mode_result['mrr']:.3f} "
|
| 188 |
+
f"ndcg@10={mode_result['ndcg@10']:.3f} "
|
| 189 |
+
f"retrieval_latency={mode_result['mean_retrieval_latency_ms']:.1f}ms"
|
| 190 |
+
)
|
| 191 |
+
|
| 192 |
+
results.append(model_result)
|
| 193 |
+
embedder.unload()
|
| 194 |
+
|
| 195 |
+
return results
|
benchmark/vectorstore.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Thin Qdrant wrapper: one collection per embedding model under test."""
|
| 2 |
+
|
| 3 |
+
import os
|
| 4 |
+
import re
|
| 5 |
+
from uuid import uuid4
|
| 6 |
+
|
| 7 |
+
from qdrant_client import QdrantClient
|
| 8 |
+
from qdrant_client.models import Distance, PointStruct, VectorParams
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def collection_name_for(model_name):
|
| 12 |
+
return "books__" + re.sub(r"[^a-zA-Z0-9]+", "_", model_name).strip("_")
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
def _dir_size_bytes(path):
|
| 16 |
+
total = 0
|
| 17 |
+
for root, _dirs, files in os.walk(path):
|
| 18 |
+
for name in files:
|
| 19 |
+
fp = os.path.join(root, name)
|
| 20 |
+
if os.path.isfile(fp):
|
| 21 |
+
total += os.path.getsize(fp)
|
| 22 |
+
return total
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class BookVectorStore:
|
| 26 |
+
def __init__(self, qdrant_path):
|
| 27 |
+
self.qdrant_path = qdrant_path
|
| 28 |
+
self.client = QdrantClient(path=qdrant_path)
|
| 29 |
+
|
| 30 |
+
def index(self, model_name, embeddings, books):
|
| 31 |
+
name = collection_name_for(model_name)
|
| 32 |
+
if self.client.collection_exists(name):
|
| 33 |
+
self.client.delete_collection(name)
|
| 34 |
+
|
| 35 |
+
self.client.create_collection(
|
| 36 |
+
collection_name=name,
|
| 37 |
+
vectors_config=VectorParams(
|
| 38 |
+
size=embeddings.shape[1],
|
| 39 |
+
distance=Distance.COSINE,
|
| 40 |
+
),
|
| 41 |
+
)
|
| 42 |
+
|
| 43 |
+
points = [
|
| 44 |
+
PointStruct(
|
| 45 |
+
id=str(uuid4()),
|
| 46 |
+
vector=embedding.tolist(),
|
| 47 |
+
payload={"book_id": book["book_id"]},
|
| 48 |
+
)
|
| 49 |
+
for book, embedding in zip(books, embeddings)
|
| 50 |
+
]
|
| 51 |
+
self.client.upsert(collection_name=name, points=points)
|
| 52 |
+
return name
|
| 53 |
+
|
| 54 |
+
def search(self, model_name, query_embedding, top_k):
|
| 55 |
+
name = collection_name_for(model_name)
|
| 56 |
+
results = self.client.query_points(
|
| 57 |
+
collection_name=name,
|
| 58 |
+
query=query_embedding.tolist(),
|
| 59 |
+
limit=top_k,
|
| 60 |
+
with_payload=True,
|
| 61 |
+
).points
|
| 62 |
+
return [point.payload["book_id"] for point in results]
|
| 63 |
+
|
| 64 |
+
def collection_disk_size_mb(self, model_name):
|
| 65 |
+
name = collection_name_for(model_name)
|
| 66 |
+
collection_dir = os.path.join(self.qdrant_path, "collection", name)
|
| 67 |
+
if not os.path.isdir(collection_dir):
|
| 68 |
+
return None
|
| 69 |
+
return _dir_size_bytes(collection_dir) / (1024 ** 2)
|
boi1_eval_queries.json
CHANGED
|
@@ -10,7 +10,8 @@
|
|
| 10 |
"pure_bangla",
|
| 11 |
"factual",
|
| 12 |
"literature_history"
|
| 13 |
-
]
|
|
|
|
| 14 |
},
|
| 15 |
{
|
| 16 |
"query": "রবীন্দ্রনাথ ঠাকুরের বিখ্যাত কাব্যগ্রন্থ কী কী?",
|
|
@@ -22,7 +23,8 @@
|
|
| 22 |
"pure_bangla",
|
| 23 |
"author_specific",
|
| 24 |
"poetry"
|
| 25 |
-
]
|
|
|
|
| 26 |
},
|
| 27 |
{
|
| 28 |
"query": "মুক্তিযুদ্ধ নিয়ে লেখা ভালো উপন্যাস",
|
|
@@ -37,7 +39,8 @@
|
|
| 37 |
"pure_bangla",
|
| 38 |
"intent_based",
|
| 39 |
"theme"
|
| 40 |
-
]
|
|
|
|
| 41 |
},
|
| 42 |
{
|
| 43 |
"query": "শরৎচন্দ্র চট্টোপাধ্যায়ের প্রেমের উপন্যাস",
|
|
@@ -50,7 +53,8 @@
|
|
| 50 |
"pure_bangla",
|
| 51 |
"author_specific",
|
| 52 |
"genre"
|
| 53 |
-
]
|
|
|
|
| 54 |
},
|
| 55 |
{
|
| 56 |
"query": "ভালো গোয়েন্দা কাহিনী সুপারিশ করুন",
|
|
@@ -64,7 +68,8 @@
|
|
| 64 |
"pure_bangla",
|
| 65 |
"recommendation",
|
| 66 |
"genre"
|
| 67 |
-
]
|
|
|
|
| 68 |
},
|
| 69 |
{
|
| 70 |
"query": "misir ali series er shera boi gula ki ki",
|
|
@@ -77,7 +82,8 @@
|
|
| 77 |
"banglish",
|
| 78 |
"series",
|
| 79 |
"recommendation"
|
| 80 |
-
]
|
|
|
|
| 81 |
},
|
| 82 |
{
|
| 83 |
"query": "rabindranath er kabbo grontho gula",
|
|
@@ -89,7 +95,8 @@
|
|
| 89 |
"banglish",
|
| 90 |
"author_variation",
|
| 91 |
"poetry"
|
| 92 |
-
]
|
|
|
|
| 93 |
},
|
| 94 |
{
|
| 95 |
"query": "himu series porte chai kon boi diye shuru korbo",
|
|
@@ -101,7 +108,8 @@
|
|
| 101 |
"banglish",
|
| 102 |
"series",
|
| 103 |
"recommendation"
|
| 104 |
-
]
|
|
|
|
| 105 |
},
|
| 106 |
{
|
| 107 |
"query": "sunil gangopadhyay er itihas bhittik uponnash",
|
|
@@ -114,7 +122,8 @@
|
|
| 114 |
"banglish",
|
| 115 |
"author_variation",
|
| 116 |
"genre"
|
| 117 |
-
]
|
|
|
|
| 118 |
},
|
| 119 |
{
|
| 120 |
"query": "muktijudder upore lekha kishor uponnash",
|
|
@@ -127,7 +136,8 @@
|
|
| 127 |
"banglish",
|
| 128 |
"theme",
|
| 129 |
"recommendation"
|
| 130 |
-
]
|
|
|
|
| 131 |
},
|
| 132 |
{
|
| 133 |
"query": "What is considered the first novel in Bengali literature?",
|
|
@@ -138,7 +148,8 @@
|
|
| 138 |
"categories": [
|
| 139 |
"english_to_bangla",
|
| 140 |
"factual"
|
| 141 |
-
]
|
|
|
|
| 142 |
},
|
| 143 |
{
|
| 144 |
"query": "Best science fiction books by Muhammed Zafar Iqbal",
|
|
@@ -152,7 +163,8 @@
|
|
| 152 |
"english_to_bangla",
|
| 153 |
"author_specific",
|
| 154 |
"genre"
|
| 155 |
-
]
|
|
|
|
| 156 |
},
|
| 157 |
{
|
| 158 |
"query": "Autobiography written by Taslima Nasrin",
|
|
@@ -164,7 +176,8 @@
|
|
| 164 |
"english_to_bangla",
|
| 165 |
"author_specific",
|
| 166 |
"genre"
|
| 167 |
-
]
|
|
|
|
| 168 |
},
|
| 169 |
{
|
| 170 |
"query": "Detective novels featuring Feluda",
|
|
@@ -175,7 +188,8 @@
|
|
| 175 |
"english_to_bangla",
|
| 176 |
"character_specific",
|
| 177 |
"genre"
|
| 178 |
-
]
|
|
|
|
| 179 |
},
|
| 180 |
{
|
| 181 |
"query": "Rabindranath er nationalism niye lekha boi",
|
|
@@ -187,7 +201,8 @@
|
|
| 187 |
"mixed_script",
|
| 188 |
"theme",
|
| 189 |
"author_variation"
|
| 190 |
-
]
|
|
|
|
| 191 |
},
|
| 192 |
{
|
| 193 |
"query": "Zafar Iqbal er muktijuddho niye kishor uponnash",
|
|
@@ -199,7 +214,8 @@
|
|
| 199 |
"mixed_script",
|
| 200 |
"author_variation",
|
| 201 |
"theme"
|
| 202 |
-
]
|
|
|
|
| 203 |
},
|
| 204 |
{
|
| 205 |
"query": "Bibhutibhushan er প্রকৃতি বিষয়ক লেখা",
|
|
@@ -211,7 +227,8 @@
|
|
| 211 |
"mixed_script",
|
| 212 |
"author_variation",
|
| 213 |
"theme"
|
| 214 |
-
]
|
|
|
|
| 215 |
},
|
| 216 |
{
|
| 217 |
"query": "Sarat Chandra র সামাজিক উপন্যাস",
|
|
@@ -224,7 +241,8 @@
|
|
| 224 |
"mixed_script",
|
| 225 |
"author_variation",
|
| 226 |
"genre"
|
| 227 |
-
]
|
|
|
|
| 228 |
},
|
| 229 |
{
|
| 230 |
"query": "Humayoon Ahmed er lekha jonopriyo boi somuho",
|
|
@@ -239,7 +257,8 @@
|
|
| 239 |
"author_variation",
|
| 240 |
"banglish",
|
| 241 |
"bibliography"
|
| 242 |
-
]
|
|
|
|
| 243 |
},
|
| 244 |
{
|
| 245 |
"query": "Zafor Iqbal science fiction boi",
|
|
@@ -252,7 +271,8 @@
|
|
| 252 |
"author_variation",
|
| 253 |
"banglish",
|
| 254 |
"genre"
|
| 255 |
-
]
|
|
|
|
| 256 |
},
|
| 257 |
{
|
| 258 |
"query": "Bongkim Chandra historical novels",
|
|
@@ -265,7 +285,8 @@
|
|
| 265 |
"author_variation",
|
| 266 |
"english_to_bangla",
|
| 267 |
"genre"
|
| 268 |
-
]
|
|
|
|
| 269 |
},
|
| 270 |
{
|
| 271 |
"query": "Shordindu Bandopadhyay Byomkesh series",
|
|
@@ -276,7 +297,8 @@
|
|
| 276 |
"author_variation",
|
| 277 |
"banglish",
|
| 278 |
"series"
|
| 279 |
-
]
|
|
|
|
| 280 |
},
|
| 281 |
{
|
| 282 |
"query": "nondito noroke",
|
|
@@ -286,7 +308,8 @@
|
|
| 286 |
"categories": [
|
| 287 |
"typo",
|
| 288 |
"title_variation"
|
| 289 |
-
]
|
|
|
|
| 290 |
},
|
| 291 |
{
|
| 292 |
"query": "nondito norokey",
|
|
@@ -296,7 +319,8 @@
|
|
| 296 |
"categories": [
|
| 297 |
"typo",
|
| 298 |
"title_variation"
|
| 299 |
-
]
|
|
|
|
| 300 |
},
|
| 301 |
{
|
| 302 |
"query": "debdash bangla uponnash",
|
|
@@ -306,7 +330,8 @@
|
|
| 306 |
"categories": [
|
| 307 |
"typo",
|
| 308 |
"title_variation"
|
| 309 |
-
]
|
|
|
|
| 310 |
},
|
| 311 |
{
|
| 312 |
"query": "pother panchaly bivutibhushon",
|
|
@@ -317,7 +342,8 @@
|
|
| 317 |
"typo",
|
| 318 |
"title_variation",
|
| 319 |
"author_variation"
|
| 320 |
-
]
|
|
|
|
| 321 |
},
|
| 322 |
{
|
| 323 |
"query": "thriller boi suggestion",
|
|
@@ -332,7 +358,8 @@
|
|
| 332 |
"intent_based",
|
| 333 |
"recommendation",
|
| 334 |
"genre"
|
| 335 |
-
]
|
|
|
|
| 336 |
},
|
| 337 |
{
|
| 338 |
"query": "science fiction bangla boi",
|
|
@@ -347,7 +374,8 @@
|
|
| 347 |
"intent_based",
|
| 348 |
"recommendation",
|
| 349 |
"genre"
|
| 350 |
-
]
|
|
|
|
| 351 |
},
|
| 352 |
{
|
| 353 |
"query": "romantic bangla novel recommendation",
|
|
@@ -361,7 +389,8 @@
|
|
| 361 |
"intent_based",
|
| 362 |
"recommendation",
|
| 363 |
"genre"
|
| 364 |
-
]
|
|
|
|
| 365 |
},
|
| 366 |
{
|
| 367 |
"query": "bengali poetry books for beginners",
|
|
@@ -375,6 +404,7 @@
|
|
| 375 |
"intent_based",
|
| 376 |
"recommendation",
|
| 377 |
"genre"
|
| 378 |
-
]
|
|
|
|
| 379 |
}
|
| 380 |
-
]
|
|
|
|
| 10 |
"pure_bangla",
|
| 11 |
"factual",
|
| 12 |
"literature_history"
|
| 13 |
+
],
|
| 14 |
+
"normalized_query": "বাংলা সাহিত্যের প্রথম ঐতিহাসিক উপন্যাস কোনটি?"
|
| 15 |
},
|
| 16 |
{
|
| 17 |
"query": "রবীন্দ্রনাথ ঠাকুরের বিখ্যাত কাব্যগ্রন্থ কী কী?",
|
|
|
|
| 23 |
"pure_bangla",
|
| 24 |
"author_specific",
|
| 25 |
"poetry"
|
| 26 |
+
],
|
| 27 |
+
"normalized_query": "রবীন্দ্রনাথ ঠাকুরের বিখ্যাত কাব্যগ্রন্থ কী কী?"
|
| 28 |
},
|
| 29 |
{
|
| 30 |
"query": "মুক্তিযুদ্ধ নিয়ে লেখা ভালো উপন্যাস",
|
|
|
|
| 39 |
"pure_bangla",
|
| 40 |
"intent_based",
|
| 41 |
"theme"
|
| 42 |
+
],
|
| 43 |
+
"normalized_query": "মুক্তিযুদ্ধ নিয়ে লেখা ভালো উপন্যাস"
|
| 44 |
},
|
| 45 |
{
|
| 46 |
"query": "শরৎচন্দ্র চট্টোপাধ্যায়ের প্রেমের উপন্যাস",
|
|
|
|
| 53 |
"pure_bangla",
|
| 54 |
"author_specific",
|
| 55 |
"genre"
|
| 56 |
+
],
|
| 57 |
+
"normalized_query": "শরৎচন্দ্র চট্টোপাধ্যায়ের প্রেমের উপন্যাস"
|
| 58 |
},
|
| 59 |
{
|
| 60 |
"query": "ভালো গোয়েন্দা কাহিনী সুপারিশ করুন",
|
|
|
|
| 68 |
"pure_bangla",
|
| 69 |
"recommendation",
|
| 70 |
"genre"
|
| 71 |
+
],
|
| 72 |
+
"normalized_query": "ভালো গোয়েন্দা কাহিনী সুপারিশ করুন"
|
| 73 |
},
|
| 74 |
{
|
| 75 |
"query": "misir ali series er shera boi gula ki ki",
|
|
|
|
| 82 |
"banglish",
|
| 83 |
"series",
|
| 84 |
"recommendation"
|
| 85 |
+
],
|
| 86 |
+
"normalized_query": "মিসির আলি সিরিজের সেরা বই গুলো কি কি"
|
| 87 |
},
|
| 88 |
{
|
| 89 |
"query": "rabindranath er kabbo grontho gula",
|
|
|
|
| 95 |
"banglish",
|
| 96 |
"author_variation",
|
| 97 |
"poetry"
|
| 98 |
+
],
|
| 99 |
+
"normalized_query": "রবীন্দ্রনাথের কাব্যগ্রন্থ গুলো"
|
| 100 |
},
|
| 101 |
{
|
| 102 |
"query": "himu series porte chai kon boi diye shuru korbo",
|
|
|
|
| 108 |
"banglish",
|
| 109 |
"series",
|
| 110 |
"recommendation"
|
| 111 |
+
],
|
| 112 |
+
"normalized_query": "হিমু সিরিজ পড়তে চাই কোন বই দিয়ে শুরু করবো"
|
| 113 |
},
|
| 114 |
{
|
| 115 |
"query": "sunil gangopadhyay er itihas bhittik uponnash",
|
|
|
|
| 122 |
"banglish",
|
| 123 |
"author_variation",
|
| 124 |
"genre"
|
| 125 |
+
],
|
| 126 |
+
"normalized_query": "সুনীল গঙ্গোপাধ্যায়ের ইতিহাস ভিত্তিক উপন্যাস"
|
| 127 |
},
|
| 128 |
{
|
| 129 |
"query": "muktijudder upore lekha kishor uponnash",
|
|
|
|
| 136 |
"banglish",
|
| 137 |
"theme",
|
| 138 |
"recommendation"
|
| 139 |
+
],
|
| 140 |
+
"normalized_query": "মুক্তিযুদ্ধের উপরে লেখা কিশোর উপন্যাস"
|
| 141 |
},
|
| 142 |
{
|
| 143 |
"query": "What is considered the first novel in Bengali literature?",
|
|
|
|
| 148 |
"categories": [
|
| 149 |
"english_to_bangla",
|
| 150 |
"factual"
|
| 151 |
+
],
|
| 152 |
+
"normalized_query": "বাংলা সাহিত্যের প্রথম উপন্যাস কোনটি বলে বিবেচিত হয়?"
|
| 153 |
},
|
| 154 |
{
|
| 155 |
"query": "Best science fiction books by Muhammed Zafar Iqbal",
|
|
|
|
| 163 |
"english_to_bangla",
|
| 164 |
"author_specific",
|
| 165 |
"genre"
|
| 166 |
+
],
|
| 167 |
+
"normalized_query": "মুহম্মদ জাফর ইকবালের সেরা বিজ্ঞান কল্পকাহিনী বই"
|
| 168 |
},
|
| 169 |
{
|
| 170 |
"query": "Autobiography written by Taslima Nasrin",
|
|
|
|
| 176 |
"english_to_bangla",
|
| 177 |
"author_specific",
|
| 178 |
"genre"
|
| 179 |
+
],
|
| 180 |
+
"normalized_query": "তসলিমা নাসরিনের লেখা আত্মজীবনী"
|
| 181 |
},
|
| 182 |
{
|
| 183 |
"query": "Detective novels featuring Feluda",
|
|
|
|
| 188 |
"english_to_bangla",
|
| 189 |
"character_specific",
|
| 190 |
"genre"
|
| 191 |
+
],
|
| 192 |
+
"normalized_query": "ফেলুদা চরিত্রের গোয়েন্দা উপন্যাস"
|
| 193 |
},
|
| 194 |
{
|
| 195 |
"query": "Rabindranath er nationalism niye lekha boi",
|
|
|
|
| 201 |
"mixed_script",
|
| 202 |
"theme",
|
| 203 |
"author_variation"
|
| 204 |
+
],
|
| 205 |
+
"normalized_query": "রবীন্দ্রনাথের জাতীয়তাবাদ নিয়ে লেখা বই"
|
| 206 |
},
|
| 207 |
{
|
| 208 |
"query": "Zafar Iqbal er muktijuddho niye kishor uponnash",
|
|
|
|
| 214 |
"mixed_script",
|
| 215 |
"author_variation",
|
| 216 |
"theme"
|
| 217 |
+
],
|
| 218 |
+
"normalized_query": "জাফর ইকবালের মুক্তিযুদ্ধ নিয়ে কিশোর উপন্যাস"
|
| 219 |
},
|
| 220 |
{
|
| 221 |
"query": "Bibhutibhushan er প্রকৃতি বিষয়ক লেখা",
|
|
|
|
| 227 |
"mixed_script",
|
| 228 |
"author_variation",
|
| 229 |
"theme"
|
| 230 |
+
],
|
| 231 |
+
"normalized_query": "বিভূতিভূষণের প্রকৃতি বিষয়ক লেখা"
|
| 232 |
},
|
| 233 |
{
|
| 234 |
"query": "Sarat Chandra র সামাজিক উপন্যাস",
|
|
|
|
| 241 |
"mixed_script",
|
| 242 |
"author_variation",
|
| 243 |
"genre"
|
| 244 |
+
],
|
| 245 |
+
"normalized_query": "শরৎচন্দ্রের সামাজিক উপন্যাস"
|
| 246 |
},
|
| 247 |
{
|
| 248 |
"query": "Humayoon Ahmed er lekha jonopriyo boi somuho",
|
|
|
|
| 257 |
"author_variation",
|
| 258 |
"banglish",
|
| 259 |
"bibliography"
|
| 260 |
+
],
|
| 261 |
+
"normalized_query": "হুমায়ূন আহমেদের লেখা জনপ্রিয় বই সমূহ"
|
| 262 |
},
|
| 263 |
{
|
| 264 |
"query": "Zafor Iqbal science fiction boi",
|
|
|
|
| 271 |
"author_variation",
|
| 272 |
"banglish",
|
| 273 |
"genre"
|
| 274 |
+
],
|
| 275 |
+
"normalized_query": "জাফর ইকবালের বিজ্ঞান কল্পকাহিনী বই"
|
| 276 |
},
|
| 277 |
{
|
| 278 |
"query": "Bongkim Chandra historical novels",
|
|
|
|
| 285 |
"author_variation",
|
| 286 |
"english_to_bangla",
|
| 287 |
"genre"
|
| 288 |
+
],
|
| 289 |
+
"normalized_query": "বঙ্কিমচন্দ্রের ঐতিহাসিক উপন্যাস"
|
| 290 |
},
|
| 291 |
{
|
| 292 |
"query": "Shordindu Bandopadhyay Byomkesh series",
|
|
|
|
| 297 |
"author_variation",
|
| 298 |
"banglish",
|
| 299 |
"series"
|
| 300 |
+
],
|
| 301 |
+
"normalized_query": "শরদিন্দু বন্দ্যোপাধ্যায়ের ব্যোমকেশ সিরিজ"
|
| 302 |
},
|
| 303 |
{
|
| 304 |
"query": "nondito noroke",
|
|
|
|
| 308 |
"categories": [
|
| 309 |
"typo",
|
| 310 |
"title_variation"
|
| 311 |
+
],
|
| 312 |
+
"normalized_query": "নন্দিত নরকে"
|
| 313 |
},
|
| 314 |
{
|
| 315 |
"query": "nondito norokey",
|
|
|
|
| 319 |
"categories": [
|
| 320 |
"typo",
|
| 321 |
"title_variation"
|
| 322 |
+
],
|
| 323 |
+
"normalized_query": "নন্দিত নরকে"
|
| 324 |
},
|
| 325 |
{
|
| 326 |
"query": "debdash bangla uponnash",
|
|
|
|
| 330 |
"categories": [
|
| 331 |
"typo",
|
| 332 |
"title_variation"
|
| 333 |
+
],
|
| 334 |
+
"normalized_query": "দেবদাস বাংলা উপন্যাস"
|
| 335 |
},
|
| 336 |
{
|
| 337 |
"query": "pother panchaly bivutibhushon",
|
|
|
|
| 342 |
"typo",
|
| 343 |
"title_variation",
|
| 344 |
"author_variation"
|
| 345 |
+
],
|
| 346 |
+
"normalized_query": "পথের পাঁচালী বিভূতিভূষণ"
|
| 347 |
},
|
| 348 |
{
|
| 349 |
"query": "thriller boi suggestion",
|
|
|
|
| 358 |
"intent_based",
|
| 359 |
"recommendation",
|
| 360 |
"genre"
|
| 361 |
+
],
|
| 362 |
+
"normalized_query": "থ্রিলার বই সাজেশন"
|
| 363 |
},
|
| 364 |
{
|
| 365 |
"query": "science fiction bangla boi",
|
|
|
|
| 374 |
"intent_based",
|
| 375 |
"recommendation",
|
| 376 |
"genre"
|
| 377 |
+
],
|
| 378 |
+
"normalized_query": "বিজ্ঞান কল্পকাহিনী বাংলা বই"
|
| 379 |
},
|
| 380 |
{
|
| 381 |
"query": "romantic bangla novel recommendation",
|
|
|
|
| 389 |
"intent_based",
|
| 390 |
"recommendation",
|
| 391 |
"genre"
|
| 392 |
+
],
|
| 393 |
+
"normalized_query": "রোমান্টিক বাংলা উপন্যাস সুপারিশ"
|
| 394 |
},
|
| 395 |
{
|
| 396 |
"query": "bengali poetry books for beginners",
|
|
|
|
| 404 |
"intent_based",
|
| 405 |
"recommendation",
|
| 406 |
"genre"
|
| 407 |
+
],
|
| 408 |
+
"normalized_query": "শুরুর জন্য বাংলা কবিতার বই"
|
| 409 |
}
|
| 410 |
+
]
|
run_benchmark.py
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""CLI entry point for the embedding-model retrieval benchmark.
|
| 3 |
+
|
| 4 |
+
Usage:
|
| 5 |
+
python run_benchmark.py # run the full model set from benchmark/config.py
|
| 6 |
+
python run_benchmark.py --models kazalbrur/bangla-embed-e5-small-banglish
|
| 7 |
+
python run_benchmark.py --query-modes raw # skip the normalized-query experiment
|
| 8 |
+
"""
|
| 9 |
+
|
| 10 |
+
import argparse
|
| 11 |
+
|
| 12 |
+
from benchmark import config, report
|
| 13 |
+
from benchmark.runner import run_benchmark
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def parse_args():
|
| 17 |
+
parser = argparse.ArgumentParser(description=__doc__)
|
| 18 |
+
parser.add_argument(
|
| 19 |
+
"--models",
|
| 20 |
+
nargs="+",
|
| 21 |
+
default=None,
|
| 22 |
+
help="HF model names to benchmark (default: full set in benchmark/config.py)",
|
| 23 |
+
)
|
| 24 |
+
parser.add_argument(
|
| 25 |
+
"--query-modes",
|
| 26 |
+
nargs="+",
|
| 27 |
+
choices=["raw", "normalized"],
|
| 28 |
+
default=None,
|
| 29 |
+
help="Which query variants to evaluate (default: both)",
|
| 30 |
+
)
|
| 31 |
+
parser.add_argument(
|
| 32 |
+
"--retrieval-modes",
|
| 33 |
+
nargs="+",
|
| 34 |
+
choices=["dense", "hybrid"],
|
| 35 |
+
default=None,
|
| 36 |
+
help="dense (embedding-only) and/or hybrid (BM25 + dense RRF) (default: both)",
|
| 37 |
+
)
|
| 38 |
+
parser.add_argument("--top-k", type=int, default=None)
|
| 39 |
+
parser.add_argument("--results-dir", default=config.RESULTS_DIR)
|
| 40 |
+
return parser.parse_args()
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def main():
|
| 44 |
+
args = parse_args()
|
| 45 |
+
|
| 46 |
+
models = None
|
| 47 |
+
if args.models:
|
| 48 |
+
by_name = {m["name"]: m for m in config.MODELS}
|
| 49 |
+
models = []
|
| 50 |
+
for name in args.models:
|
| 51 |
+
if name not in by_name:
|
| 52 |
+
raise SystemExit(
|
| 53 |
+
f"Unknown model '{name}'. Add it to benchmark/config.py MODELS first "
|
| 54 |
+
f"(with its query/passage prefix)."
|
| 55 |
+
)
|
| 56 |
+
models.append(by_name[name])
|
| 57 |
+
|
| 58 |
+
results = run_benchmark(
|
| 59 |
+
models=models,
|
| 60 |
+
query_modes=args.query_modes,
|
| 61 |
+
retrieval_modes=args.retrieval_modes,
|
| 62 |
+
top_k=args.top_k,
|
| 63 |
+
)
|
| 64 |
+
|
| 65 |
+
paths = report.save_results(results, args.results_dir)
|
| 66 |
+
|
| 67 |
+
print("\n" + "=" * 100)
|
| 68 |
+
print("SUMMARY")
|
| 69 |
+
print("=" * 100)
|
| 70 |
+
report.print_summary_table(results)
|
| 71 |
+
|
| 72 |
+
print("\nSaved:")
|
| 73 |
+
for label, path in paths.items():
|
| 74 |
+
print(f" {label}: {path}")
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
if __name__ == "__main__":
|
| 78 |
+
main()
|