Hemprasad Badgujar
Fix Docker build OOM; fix missing bm25_prefilter in Streamlit; update docs
6f26961 Download src/cli.py from hembad/reranker: direct link, hf CLI and curl.
- Browser
- Download file 5.47 kB
-
https://huggingface.co/spaces/hembad/reranker/resolve/main/src/cli.py
- Command line
-
hf download hf://spaces/hembad/reranker/src/cli.py
-
curl -L -o cli.py https://huggingface.co/spaces/hembad/reranker/resolve/main/src/cli.py
5.47 kB
| """Command-line entry (Deployment 1). Thin layer over the engine β parse args, call, render. | |
| Run via the repo-root shim: ``uv run python rank.py β¦`` (it puts ``src`` on the path). As the pipeline | |
| lands step-by-step this grows a ``--out`` full-run path; today it runs the preprocess phase (load β validate). | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import sys | |
| import polars as pl | |
| from common.logging import get_logger | |
| from common.paths import DEFAULT_CANDIDATES, resolve_candidates | |
| from common.runtime import configure_compute | |
| from config import DebugSettings, load_settings | |
| from preprocess import run_preprocess | |
| def _render_summary(frame: pl.DataFrame) -> None: | |
| """Print a human summary of the preprocessed candidate frame (CLI display β not engine output).""" | |
| # Polars renders tables with Unicode box chars; ensure the console can encode them. | |
| reconfigure = getattr(sys.stdout, "reconfigure", None) | |
| if reconfigure is not None: | |
| reconfigure(encoding="utf-8") | |
| print(f"rows : {frame.height:,}") | |
| print(f"columns : {frame.width}") | |
| print(f"candidate_id nulls: {frame['candidate_id'].null_count()}") | |
| if "validation_failed" in frame.columns: | |
| failed = int(frame.select(pl.col("validation_failed").sum()).item()) | |
| rate = round(failed / frame.height, 4) if frame.height else 0.0 | |
| print(f"validation_failed : {failed:,} ({rate:.2%})") | |
| preview = frame.select( | |
| "candidate_id", | |
| "current_title", | |
| "years_of_experience", | |
| pl.col("skills").list.len().alias("n_skills"), | |
| ) | |
| print(preview.head(min(frame.height, 10))) | |
| def main(argv: list[str] | None = None) -> int: | |
| """Parse args and run. Returns a process exit code (0 ok, 2 bad input).""" | |
| # Config loading never imports numpy/polars/onnxruntime, so it's safe to read here β before those libraries | |
| # get imported β letting compute.cpu_threads drive the thread-pool env vars from the very first line. | |
| settings = load_settings() | |
| configure_compute(settings.compute.cpu_threads) | |
| parser = argparse.ArgumentParser(prog="rank", description="Redrob JD->candidate ranker.") | |
| parser.add_argument( | |
| "--candidates", default=str(DEFAULT_CANDIDATES), help="path to candidates.jsonl" | |
| ) | |
| # Default 0 (all candidates) deliberately matches the hackathon's own submission_metadata_template.yaml | |
| # reproduce_command, which passes no --limit at all: "python rank.py --candidates ... --out ...". A | |
| # non-zero default here would silently under-rank the real grading run. Pass e.g. --limit 500 yourself for | |
| # a fast dev/smoke iteration. | |
| parser.add_argument("--limit", type=int, default=0, help="max records to load (0 = all, the reproduce default)") | |
| parser.add_argument( | |
| "--out", default=None, help="run the FULL pipeline -> write this submission.csv (else preprocess only)" | |
| ) | |
| parser.add_argument( | |
| "--checkpoints", action="store_true", help="persist each phase's output to Parquet (debug)" | |
| ) | |
| parser.add_argument( | |
| "--setup", action="store_true", | |
| help="pre-download + warm ALL models into the in-repo cache (offline-ready), then exit", | |
| ) | |
| args = parser.parse_args(argv) | |
| log = get_logger("cli") | |
| if args.setup: # build-time / first-run prefetch so ranking runs with no network | |
| from models import ensure_all_models | |
| # Deliberately does NOT also build the candidates Parquet cache here. A real HF Space build was | |
| # OOMKilled (exit 137) partway through preprocess.load when this DID call run_preprocess(limit=None) | |
| # here β Docker BUILD environments are commonly more memory-constrained than the RUNTIME container, | |
| # and loading+flattening 100K nested JSON records into a list[dict] before it becomes a Polars frame | |
| # is exactly the kind of Python-object-heavy step that can spike well past the "logical" data size. | |
| # OOMKilled is an OS-level SIGKILL β unrecoverable from Python, so the only real fix is not doing this | |
| # heavy step during the vulnerable build step at all. The cache still gets built on first real use | |
| # (Streamlit's _warm_candidates at container startup, or the first `rank.py --out`) β in the RUNTIME | |
| # container's own memory budget, not the build environment's β every run after that one is still fast. | |
| ensure_all_models(settings) | |
| log.info("cli.setup_complete") | |
| return 0 | |
| limit = None if args.limit == 0 else args.limit | |
| candidates = resolve_candidates(args.candidates) | |
| if not candidates.exists(): | |
| log.error("cli.candidates_missing", path=str(candidates)) | |
| return 2 | |
| if args.checkpoints: # CLI flag overrides the config toggle for this run | |
| settings = settings.model_copy( | |
| update={"debug": DebugSettings(checkpoints=True, checkpoint_dir=settings.debug.checkpoint_dir)} | |
| ) | |
| if args.out: # full online run: candidates.jsonl β submission.csv | |
| from pipeline import run_full # lazy: pulls the model stack only when actually ranking | |
| submission = run_full(str(candidates), args.out, settings, limit) | |
| log.info("cli.submission_written", path=args.out, rows=submission.height) | |
| return 0 | |
| frame, _as_of = run_preprocess(candidates, limit=limit, settings=settings) | |
| _render_summary(frame) | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit(main()) | |