"""Command-line entry (Deployment 1). Thin layer over the engine — parse args, call, render. Run via the repo-root shim: ``uv run python rank.py …`` (it puts ``src`` on the path). As the pipeline lands step-by-step this grows a ``--out`` full-run path; today it runs the preprocess phase (load → validate). """ from __future__ import annotations import argparse import sys import polars as pl from common.logging import get_logger from common.paths import DEFAULT_CANDIDATES, resolve_candidates from common.runtime import configure_compute from config import DebugSettings, load_settings from preprocess import run_preprocess def _render_summary(frame: pl.DataFrame) -> None: """Print a human summary of the preprocessed candidate frame (CLI display — not engine output).""" # Polars renders tables with Unicode box chars; ensure the console can encode them. reconfigure = getattr(sys.stdout, "reconfigure", None) if reconfigure is not None: reconfigure(encoding="utf-8") print(f"rows : {frame.height:,}") print(f"columns : {frame.width}") print(f"candidate_id nulls: {frame['candidate_id'].null_count()}") if "validation_failed" in frame.columns: failed = int(frame.select(pl.col("validation_failed").sum()).item()) rate = round(failed / frame.height, 4) if frame.height else 0.0 print(f"validation_failed : {failed:,} ({rate:.2%})") preview = frame.select( "candidate_id", "current_title", "years_of_experience", pl.col("skills").list.len().alias("n_skills"), ) print(preview.head(min(frame.height, 10))) def main(argv: list[str] | None = None) -> int: """Parse args and run. Returns a process exit code (0 ok, 2 bad input).""" # Config loading never imports numpy/polars/onnxruntime, so it's safe to read here — before those libraries # get imported — letting compute.cpu_threads drive the thread-pool env vars from the very first line. settings = load_settings() configure_compute(settings.compute.cpu_threads) parser = argparse.ArgumentParser(prog="rank", description="Redrob JD->candidate ranker.") parser.add_argument( "--candidates", default=str(DEFAULT_CANDIDATES), help="path to candidates.jsonl" ) # Default 0 (all candidates) deliberately matches the hackathon's own submission_metadata_template.yaml # reproduce_command, which passes no --limit at all: "python rank.py --candidates ... --out ...". A # non-zero default here would silently under-rank the real grading run. Pass e.g. --limit 500 yourself for # a fast dev/smoke iteration. parser.add_argument("--limit", type=int, default=0, help="max records to load (0 = all, the reproduce default)") parser.add_argument( "--out", default=None, help="run the FULL pipeline -> write this submission.csv (else preprocess only)" ) parser.add_argument( "--checkpoints", action="store_true", help="persist each phase's output to Parquet (debug)" ) parser.add_argument( "--setup", action="store_true", help="pre-download + warm ALL models into the in-repo cache (offline-ready), then exit", ) args = parser.parse_args(argv) log = get_logger("cli") if args.setup: # build-time / first-run prefetch so ranking runs with no network from models import ensure_all_models # Deliberately does NOT also build the candidates Parquet cache here. A real HF Space build was # OOMKilled (exit 137) partway through preprocess.load when this DID call run_preprocess(limit=None) # here — Docker BUILD environments are commonly more memory-constrained than the RUNTIME container, # and loading+flattening 100K nested JSON records into a list[dict] before it becomes a Polars frame # is exactly the kind of Python-object-heavy step that can spike well past the "logical" data size. # OOMKilled is an OS-level SIGKILL — unrecoverable from Python, so the only real fix is not doing this # heavy step during the vulnerable build step at all. The cache still gets built on first real use # (Streamlit's _warm_candidates at container startup, or the first `rank.py --out`) — in the RUNTIME # container's own memory budget, not the build environment's — every run after that one is still fast. ensure_all_models(settings) log.info("cli.setup_complete") return 0 limit = None if args.limit == 0 else args.limit candidates = resolve_candidates(args.candidates) if not candidates.exists(): log.error("cli.candidates_missing", path=str(candidates)) return 2 if args.checkpoints: # CLI flag overrides the config toggle for this run settings = settings.model_copy( update={"debug": DebugSettings(checkpoints=True, checkpoint_dir=settings.debug.checkpoint_dir)} ) if args.out: # full online run: candidates.jsonl → submission.csv from pipeline import run_full # lazy: pulls the model stack only when actually ranking submission = run_full(str(candidates), args.out, settings, limit) log.info("cli.submission_written", path=args.out, rows=submission.height) return 0 frame, _as_of = run_preprocess(candidates, limit=limit, settings=settings) _render_summary(frame) return 0 if __name__ == "__main__": raise SystemExit(main())