reranker / src /cli.py
Hemprasad Badgujar
Fix Docker build OOM; fix missing bm25_prefilter in Streamlit; update docs
6f26961
Raw History Blame Contribute Delete
5.47 kB
"""Command-line entry (Deployment 1). Thin layer over the engine β€” parse args, call, render.
Run via the repo-root shim: ``uv run python rank.py …`` (it puts ``src`` on the path). As the pipeline
lands step-by-step this grows a ``--out`` full-run path; today it runs the preprocess phase (load β†’ validate).
"""
from __future__ import annotations
import argparse
import sys
import polars as pl
from common.logging import get_logger
from common.paths import DEFAULT_CANDIDATES, resolve_candidates
from common.runtime import configure_compute
from config import DebugSettings, load_settings
from preprocess import run_preprocess
def _render_summary(frame: pl.DataFrame) -> None:
"""Print a human summary of the preprocessed candidate frame (CLI display β€” not engine output)."""
# Polars renders tables with Unicode box chars; ensure the console can encode them.
reconfigure = getattr(sys.stdout, "reconfigure", None)
if reconfigure is not None:
reconfigure(encoding="utf-8")
print(f"rows : {frame.height:,}")
print(f"columns : {frame.width}")
print(f"candidate_id nulls: {frame['candidate_id'].null_count()}")
if "validation_failed" in frame.columns:
failed = int(frame.select(pl.col("validation_failed").sum()).item())
rate = round(failed / frame.height, 4) if frame.height else 0.0
print(f"validation_failed : {failed:,} ({rate:.2%})")
preview = frame.select(
"candidate_id",
"current_title",
"years_of_experience",
pl.col("skills").list.len().alias("n_skills"),
)
print(preview.head(min(frame.height, 10)))
def main(argv: list[str] | None = None) -> int:
"""Parse args and run. Returns a process exit code (0 ok, 2 bad input)."""
# Config loading never imports numpy/polars/onnxruntime, so it's safe to read here β€” before those libraries
# get imported β€” letting compute.cpu_threads drive the thread-pool env vars from the very first line.
settings = load_settings()
configure_compute(settings.compute.cpu_threads)
parser = argparse.ArgumentParser(prog="rank", description="Redrob JD->candidate ranker.")
parser.add_argument(
"--candidates", default=str(DEFAULT_CANDIDATES), help="path to candidates.jsonl"
)
# Default 0 (all candidates) deliberately matches the hackathon's own submission_metadata_template.yaml
# reproduce_command, which passes no --limit at all: "python rank.py --candidates ... --out ...". A
# non-zero default here would silently under-rank the real grading run. Pass e.g. --limit 500 yourself for
# a fast dev/smoke iteration.
parser.add_argument("--limit", type=int, default=0, help="max records to load (0 = all, the reproduce default)")
parser.add_argument(
"--out", default=None, help="run the FULL pipeline -> write this submission.csv (else preprocess only)"
)
parser.add_argument(
"--checkpoints", action="store_true", help="persist each phase's output to Parquet (debug)"
)
parser.add_argument(
"--setup", action="store_true",
help="pre-download + warm ALL models into the in-repo cache (offline-ready), then exit",
)
args = parser.parse_args(argv)
log = get_logger("cli")
if args.setup: # build-time / first-run prefetch so ranking runs with no network
from models import ensure_all_models
# Deliberately does NOT also build the candidates Parquet cache here. A real HF Space build was
# OOMKilled (exit 137) partway through preprocess.load when this DID call run_preprocess(limit=None)
# here β€” Docker BUILD environments are commonly more memory-constrained than the RUNTIME container,
# and loading+flattening 100K nested JSON records into a list[dict] before it becomes a Polars frame
# is exactly the kind of Python-object-heavy step that can spike well past the "logical" data size.
# OOMKilled is an OS-level SIGKILL β€” unrecoverable from Python, so the only real fix is not doing this
# heavy step during the vulnerable build step at all. The cache still gets built on first real use
# (Streamlit's _warm_candidates at container startup, or the first `rank.py --out`) β€” in the RUNTIME
# container's own memory budget, not the build environment's β€” every run after that one is still fast.
ensure_all_models(settings)
log.info("cli.setup_complete")
return 0
limit = None if args.limit == 0 else args.limit
candidates = resolve_candidates(args.candidates)
if not candidates.exists():
log.error("cli.candidates_missing", path=str(candidates))
return 2
if args.checkpoints: # CLI flag overrides the config toggle for this run
settings = settings.model_copy(
update={"debug": DebugSettings(checkpoints=True, checkpoint_dir=settings.debug.checkpoint_dir)}
)
if args.out: # full online run: candidates.jsonl β†’ submission.csv
from pipeline import run_full # lazy: pulls the model stack only when actually ranking
submission = run_full(str(candidates), args.out, settings, limit)
log.info("cli.submission_written", path=args.out, rows=submission.height)
return 0
frame, _as_of = run_preprocess(candidates, limit=limit, settings=settings)
_render_summary(frame)
return 0
if __name__ == "__main__":
raise SystemExit(main())