Hemprasad Badgujar
Add bm25 prefilter stage; unify candidates cache; validate reasoning quality
cd5f9d4 Download src/common/paths.py from hembad/reranker: direct link, hf CLI and curl.
- Browser
- Download file 2 kB
-
https://huggingface.co/spaces/hembad/reranker/resolve/main/src/common/paths.py
- Command line
-
hf download hf://spaces/hembad/reranker/src/common/paths.py
-
curl -L -o paths.py https://huggingface.co/spaces/hembad/reranker/resolve/main/src/common/paths.py
2 kB
| """Repo-anchored paths β resolve data files independent of the process working directory. | |
| WHY: Streamlit (and HF Spaces) launch with a working directory that is not the repo root, so a relative | |
| ``data/candidates.jsonl`` fails to resolve. Anchoring to the repo root (derived from this file's location) | |
| makes the same default work from the CLI, the app, and a container alike. | |
| """ | |
| from __future__ import annotations | |
| from pathlib import Path | |
| # src/common/paths.py β parents[2] is the repo root (../../ up from common/). | |
| REPO_ROOT = Path(__file__).resolve().parents[2] | |
| DATA_DIR = REPO_ROOT / "data" | |
| DEFAULT_CANDIDATES = DATA_DIR / "candidates.jsonl" | |
| # Loaded-candidates cache β the RAW load_candidates() output (before validate/materialize/quality) for the | |
| # FULL, unlimited dataset. Written once by ``rank.py --setup`` (bakes into the Docker image, same as model | |
| # weights) or on first use, then read directly on every later call instead of re-parsing the 465MB JSONL | |
| # (~13-20s β ~1-2s). Lives INSIDE load_candidates() itself (not one layer up in run_preprocess) specifically | |
| # so BOTH the CLI (run_preprocess) and Streamlit (which calls load_candidates directly, bypassing | |
| # run_preprocess, for its per-phase explorer) share the same cache. validate/materialize/quality always run | |
| # fresh downstream in both callers β they're cheap (~2-3s combined) AND quality.py appends to the ``text`` | |
| # column, so caching their output and replaying them again would double-append and corrupt it. | |
| LOADED_CANDIDATES_CACHE = DATA_DIR / "artifacts" / "candidates_loaded.parquet" | |
| def resolve_candidates(path: str | Path) -> Path: | |
| """Resolve a candidates path: use it as-given if it exists, else try it relative to the repo root. | |
| Returns the resolved Path if found; otherwise returns the original (so the caller reports not-found). | |
| """ | |
| given = Path(path) | |
| if given.exists(): | |
| return given | |
| anchored = REPO_ROOT / path | |
| return anchored if anchored.exists() else given | |