File size: 5,105 Bytes
f8b48da da4731b f8b48da da4731b f8b48da da4731b f8b48da da4731b f8b48da da4731b f8b48da da4731b f8b48da da4731b f8b48da | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 | """
Generate evals/reports/benchmark.md β a human-readable comparison of all experiments.
Run:
python scripts/generate_benchmark.py
"""
from __future__ import annotations
import json
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
RESULTS_DIR = ROOT / "evals" / "results"
REPORT_PATH = ROOT / "evals" / "reports" / "benchmark.md"
EXPERIMENTS = ["vector", "bm25", "hybrid", "final"]
def _load_results(experiment: str, results_dir: str | Path | None = None) -> dict | None:
"""Load the latest results JSON for the given experiment name.
Results are saved to evals/results/experiment_{experiment}_{id}.json
(e.g. experiment_final_a1b2c3d4.json), so we glob for that pattern
directly rather than looking in a subdirectory.
``results_dir`` overrides the default ``evals/results`` location.
"""
root = Path(results_dir) if results_dir else RESULTS_DIR
if not root.exists():
return None
# Find all JSON files matching the experiment pattern and pick the latest
pattern = f"experiment_{experiment}_*.json"
files = sorted(root.glob(pattern), key=lambda p: p.stat().st_mtime)
if not files:
return None
return json.loads(files[-1].read_text(encoding="utf-8"))
def _fmt(v: float | None, decimals: int = 4) -> str:
if v is None:
return "N/A"
return f"{v:.{decimals}f}"
def _section(title: str) -> str:
return f"\n## {title}\n"
def _table_row(values: list[str]) -> str:
return "| " + " | ".join(values) + " |"
def _table_header(cols: list[str]) -> str:
return "| " + " | ".join(cols) + " |\n" + "| " + " | ".join("---" for _ in cols) + " |"
def generate_report(results_dir: str | Path | None = None) -> Path:
"""Render ``evals/reports/benchmark.md`` from the experiment result files.
Returns the path the report was written to. Callers that only need the
artifact (e.g. the Celery task) should use this; ``main()`` is the CLI
wrapper around it.
``results_dir`` overrides the default ``evals/results`` location.
"""
REPORT_PATH.parent.mkdir(parents=True, exist_ok=True)
lines: list[str] = [
"# RAG Evaluation Benchmark",
"",
_section("Experiment Summary"),
_table_header([
"Experiment", "Questions", "Duration (s)",
"Recall@10", "Hit Rate@5", "Citation Correctness",
]),
"",
]
experiments_data: dict[str, dict] = {}
for exp in EXPERIMENTS:
data = _load_results(exp, results_dir)
if data is None:
lines.append(
_table_row([exp, "β", "β", "β", "β", "β", "No results found"])
)
continue
experiments_data[exp] = data
agg = data.get("aggregate_retrieval", {})
cit = data.get("aggregate_citation", {})
lines.append(
_table_row([
exp,
str(data.get("question_count", "?")),
f"{data.get('duration_seconds', 0):.1f}",
_fmt(agg.get("recall_at_k")),
_fmt(agg.get("hit_rate")),
_fmt(cit.get("citation_correctness")),
])
)
lines.append("")
# Per-difficulty breakdown
if experiments_data:
lines.append(_section("Per-Difficulty Breakdown"))
for exp, data in experiments_data.items():
per_diff = data.get("per_difficulty", {})
if not per_diff:
continue
lines.append(f"\n### {exp}\n")
lines.append(_table_header([
"Difficulty", "Count", "Recall@10",
"Hit Rate", "Citation Correctness",
]))
for diff_name in ["easy", "medium", "hard", "unanswerable"]:
dd = per_diff.get(diff_name, {})
if not dd:
continue
lines.append(
_table_row([
diff_name,
str(dd.get("count", "?")),
_fmt(dd.get("recall_at_k")),
_fmt(dd.get("hit_rate")),
_fmt(dd.get("citation_correctness")),
])
)
lines.append("")
# Ragas metrics
ragas_exp = experiments_data.get("final", {})
ragas_metrics = ragas_exp.get("aggregate_ragas", {})
if ragas_metrics:
lines.append(_section("Ragas Metrics (Experiment D β Full Pipeline)"))
lines.append(_table_header(["Metric", "Value"]))
for key, val in sorted(ragas_metrics.items()):
lines.append(_table_row([key, _fmt(val)]))
lines.append("")
REPORT_PATH.write_text("\n".join(lines) + "\n", encoding="utf-8")
return REPORT_PATH
def main() -> int:
report_path = generate_report()
print(f"Benchmark written to {report_path}")
return 0
if __name__ == "__main__":
sys.exit(main())
|