"""Pre-download + warm EVERY runtime model — the build-time ``ranker-setup`` so ranking runs OFFLINE. The reproduce contract is "no network during ranking": models must already be on disk before the timed run. This warms all four into the in-image cache (``data/artifacts/models`` via the ``HF_HOME`` redirect) at Docker **build** time, so the container never touches the network while ranking. Run manually with ``uv run python rank.py --setup``. Model set (LFM2.5-350M, Qwen3.5-2B, Qwen3-4B/0.6B, and Gemma-3-4B/1B were all benchmarked and dropped — Qwen3-1.7B won on speed while still clearing the min_skills recall target with correctly-atomized keywords): * Qwen3-1.7B-GGUF — JD parse + T1 reasoning polish (top 10 only, when reasoning.mode == "llm") * potion-base-32M — Step 05 broad embed + NLG variant selection * granite-r2 ONNX — Step 08 precision embed * granite-reranker — Step 09 cross-encoder """ from __future__ import annotations from common.logging import get_logger from config import Settings, load_settings from .llm import index_gen_client def ensure_all_models(settings: Settings | None = None) -> None: """(Down)load + warm every model the pipeline uses. Safe to re-run (cached loads no-op).""" active = settings or load_settings() log = get_logger("models") # 1) LLM (llama.cpp GGUF) — does JD parse AND the Step-11 outcome notes (T1, top 10 only, mode == "llm"). # LFM2.5-350M was verified useless for reasoning (3/15, verbatim copies) and is no longer loaded/prefetched. index_gen_client(active).ensure() log.info("prefetch.llm_ready", model=active.models.index_gen_llm_repo) # 2) Static + ONNX retrieval models — warm via the same private loaders the pipeline uses. from retrieve.granite import _load_granite from retrieve.potion import _load_potion from retrieve.reranker import _load_reranker # cpu_threads is threaded through so the @lru_cache key matches what the real rank run will request — if it # didn't, a non-default compute.cpu_threads would cache-MISS at rank time and rebuild the ONNX session from # scratch (weights are still local, so not a network issue, but it would waste the whole point of prefetch). threads = active.compute.cpu_threads _load_potion(active.models.embedder_broad) log.info("prefetch.embed_ready", model=active.models.embedder_broad) _load_granite(active.models.embedder_precision, active.precision.onnx_file, active.precision.max_length, threads) log.info("prefetch.embed_ready", model=active.models.embedder_precision) _load_reranker(active.models.reranker, active.rerank.onnx_file, active.rerank.max_length, threads) log.info("prefetch.embed_ready", model=active.models.reranker) log.info("prefetch.complete") if __name__ == "__main__": from common.runtime import configure_compute active = load_settings() configure_compute(active.compute.cpu_threads) ensure_all_models(active)