"""Repo-anchored paths — resolve data files independent of the process working directory. WHY: Streamlit (and HF Spaces) launch with a working directory that is not the repo root, so a relative ``data/candidates.jsonl`` fails to resolve. Anchoring to the repo root (derived from this file's location) makes the same default work from the CLI, the app, and a container alike. """ from __future__ import annotations from pathlib import Path # src/common/paths.py → parents[2] is the repo root (../../ up from common/). REPO_ROOT = Path(__file__).resolve().parents[2] DATA_DIR = REPO_ROOT / "data" DEFAULT_CANDIDATES = DATA_DIR / "candidates.jsonl" # Loaded-candidates cache — the RAW load_candidates() output (before validate/materialize/quality) for the # FULL, unlimited dataset. Written once by ``rank.py --setup`` (bakes into the Docker image, same as model # weights) or on first use, then read directly on every later call instead of re-parsing the 465MB JSONL # (~13-20s → ~1-2s). Lives INSIDE load_candidates() itself (not one layer up in run_preprocess) specifically # so BOTH the CLI (run_preprocess) and Streamlit (which calls load_candidates directly, bypassing # run_preprocess, for its per-phase explorer) share the same cache. validate/materialize/quality always run # fresh downstream in both callers — they're cheap (~2-3s combined) AND quality.py appends to the ``text`` # column, so caching their output and replaying them again would double-append and corrupt it. LOADED_CANDIDATES_CACHE = DATA_DIR / "artifacts" / "candidates_loaded.parquet" def resolve_candidates(path: str | Path) -> Path: """Resolve a candidates path: use it as-given if it exists, else try it relative to the repo root. Returns the resolved Path if found; otherwise returns the original (so the caller reports not-found). """ given = Path(path) if given.exists(): return given anchored = REPO_ROOT / path return anchored if anchored.exists() else given