reranker / src /models /prefetch.py
Hemprasad Badgujar
Switch JD-parse/T1 polish to Qwen3-1.7B
baeef08
Raw History Blame Contribute Delete
2.99 kB
"""Pre-download + warm EVERY runtime model β€” the build-time ``ranker-setup`` so ranking runs OFFLINE.
The reproduce contract is "no network during ranking": models must already be on disk before the timed run.
This warms all four into the in-image cache (``data/artifacts/models`` via the ``HF_HOME`` redirect) at Docker
**build** time, so the container never touches the network while ranking. Run manually with
``uv run python rank.py --setup``.
Model set (LFM2.5-350M, Qwen3.5-2B, Qwen3-4B/0.6B, and Gemma-3-4B/1B were all benchmarked and dropped β€”
Qwen3-1.7B won on speed while still clearing the min_skills recall target with correctly-atomized keywords):
* Qwen3-1.7B-GGUF β€” JD parse + T1 reasoning polish (top 10 only, when reasoning.mode == "llm")
* potion-base-32M β€” Step 05 broad embed + NLG variant selection
* granite-r2 ONNX β€” Step 08 precision embed
* granite-reranker β€” Step 09 cross-encoder
"""
from __future__ import annotations
from common.logging import get_logger
from config import Settings, load_settings
from .llm import index_gen_client
def ensure_all_models(settings: Settings | None = None) -> None:
"""(Down)load + warm every model the pipeline uses. Safe to re-run (cached loads no-op)."""
active = settings or load_settings()
log = get_logger("models")
# 1) LLM (llama.cpp GGUF) β€” does JD parse AND the Step-11 outcome notes (T1, top 10 only, mode == "llm").
# LFM2.5-350M was verified useless for reasoning (3/15, verbatim copies) and is no longer loaded/prefetched.
index_gen_client(active).ensure()
log.info("prefetch.llm_ready", model=active.models.index_gen_llm_repo)
# 2) Static + ONNX retrieval models β€” warm via the same private loaders the pipeline uses.
from retrieve.granite import _load_granite
from retrieve.potion import _load_potion
from retrieve.reranker import _load_reranker
# cpu_threads is threaded through so the @lru_cache key matches what the real rank run will request β€” if it
# didn't, a non-default compute.cpu_threads would cache-MISS at rank time and rebuild the ONNX session from
# scratch (weights are still local, so not a network issue, but it would waste the whole point of prefetch).
threads = active.compute.cpu_threads
_load_potion(active.models.embedder_broad)
log.info("prefetch.embed_ready", model=active.models.embedder_broad)
_load_granite(active.models.embedder_precision, active.precision.onnx_file, active.precision.max_length, threads)
log.info("prefetch.embed_ready", model=active.models.embedder_precision)
_load_reranker(active.models.reranker, active.rerank.onnx_file, active.rerank.max_length, threads)
log.info("prefetch.embed_ready", model=active.models.reranker)
log.info("prefetch.complete")
if __name__ == "__main__":
from common.runtime import configure_compute
active = load_settings()
configure_compute(active.compute.cpu_threads)
ensure_all_models(active)