File size: 5,465 Bytes
28c12aa 700e896 28c12aa 0d4730e 66cf40d 700e896 28c12aa 700e896 28c12aa 700e896 28c12aa 700e896 28c12aa a41b023 d68637e 0d4730e b24d6e4 e73e29e d68637e e73e29e 700e896 e264368 6f26961 e264368 28c12aa 6f26961 e264368 6f26961 a41b023 e264368 28c12aa 0d4730e 28c12aa 700e896 e73e29e b24d6e4 e73e29e 700e896 28c12aa | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 | """Command-line entry (Deployment 1). Thin layer over the engine β parse args, call, render.
Run via the repo-root shim: ``uv run python rank.py β¦`` (it puts ``src`` on the path). As the pipeline
lands step-by-step this grows a ``--out`` full-run path; today it runs the preprocess phase (load β validate).
"""
from __future__ import annotations
import argparse
import sys
import polars as pl
from common.logging import get_logger
from common.paths import DEFAULT_CANDIDATES, resolve_candidates
from common.runtime import configure_compute
from config import DebugSettings, load_settings
from preprocess import run_preprocess
def _render_summary(frame: pl.DataFrame) -> None:
"""Print a human summary of the preprocessed candidate frame (CLI display β not engine output)."""
# Polars renders tables with Unicode box chars; ensure the console can encode them.
reconfigure = getattr(sys.stdout, "reconfigure", None)
if reconfigure is not None:
reconfigure(encoding="utf-8")
print(f"rows : {frame.height:,}")
print(f"columns : {frame.width}")
print(f"candidate_id nulls: {frame['candidate_id'].null_count()}")
if "validation_failed" in frame.columns:
failed = int(frame.select(pl.col("validation_failed").sum()).item())
rate = round(failed / frame.height, 4) if frame.height else 0.0
print(f"validation_failed : {failed:,} ({rate:.2%})")
preview = frame.select(
"candidate_id",
"current_title",
"years_of_experience",
pl.col("skills").list.len().alias("n_skills"),
)
print(preview.head(min(frame.height, 10)))
def main(argv: list[str] | None = None) -> int:
"""Parse args and run. Returns a process exit code (0 ok, 2 bad input)."""
# Config loading never imports numpy/polars/onnxruntime, so it's safe to read here β before those libraries
# get imported β letting compute.cpu_threads drive the thread-pool env vars from the very first line.
settings = load_settings()
configure_compute(settings.compute.cpu_threads)
parser = argparse.ArgumentParser(prog="rank", description="Redrob JD->candidate ranker.")
parser.add_argument(
"--candidates", default=str(DEFAULT_CANDIDATES), help="path to candidates.jsonl"
)
# Default 0 (all candidates) deliberately matches the hackathon's own submission_metadata_template.yaml
# reproduce_command, which passes no --limit at all: "python rank.py --candidates ... --out ...". A
# non-zero default here would silently under-rank the real grading run. Pass e.g. --limit 500 yourself for
# a fast dev/smoke iteration.
parser.add_argument("--limit", type=int, default=0, help="max records to load (0 = all, the reproduce default)")
parser.add_argument(
"--out", default=None, help="run the FULL pipeline -> write this submission.csv (else preprocess only)"
)
parser.add_argument(
"--checkpoints", action="store_true", help="persist each phase's output to Parquet (debug)"
)
parser.add_argument(
"--setup", action="store_true",
help="pre-download + warm ALL models into the in-repo cache (offline-ready), then exit",
)
args = parser.parse_args(argv)
log = get_logger("cli")
if args.setup: # build-time / first-run prefetch so ranking runs with no network
from models import ensure_all_models
# Deliberately does NOT also build the candidates Parquet cache here. A real HF Space build was
# OOMKilled (exit 137) partway through preprocess.load when this DID call run_preprocess(limit=None)
# here β Docker BUILD environments are commonly more memory-constrained than the RUNTIME container,
# and loading+flattening 100K nested JSON records into a list[dict] before it becomes a Polars frame
# is exactly the kind of Python-object-heavy step that can spike well past the "logical" data size.
# OOMKilled is an OS-level SIGKILL β unrecoverable from Python, so the only real fix is not doing this
# heavy step during the vulnerable build step at all. The cache still gets built on first real use
# (Streamlit's _warm_candidates at container startup, or the first `rank.py --out`) β in the RUNTIME
# container's own memory budget, not the build environment's β every run after that one is still fast.
ensure_all_models(settings)
log.info("cli.setup_complete")
return 0
limit = None if args.limit == 0 else args.limit
candidates = resolve_candidates(args.candidates)
if not candidates.exists():
log.error("cli.candidates_missing", path=str(candidates))
return 2
if args.checkpoints: # CLI flag overrides the config toggle for this run
settings = settings.model_copy(
update={"debug": DebugSettings(checkpoints=True, checkpoint_dir=settings.debug.checkpoint_dir)}
)
if args.out: # full online run: candidates.jsonl β submission.csv
from pipeline import run_full # lazy: pulls the model stack only when actually ranking
submission = run_full(str(candidates), args.out, settings, limit)
log.info("cli.submission_written", path=args.out, rows=submission.height)
return 0
frame, _as_of = run_preprocess(candidates, limit=limit, settings=settings)
_render_summary(frame)
return 0
if __name__ == "__main__":
raise SystemExit(main())
|