Download src/models/prefetch.py from hembad/reranker: direct link, hf CLI and curl.
- Browser
- Download file 2.99 kB
-
https://huggingface.co/spaces/hembad/reranker/resolve/main/src/models/prefetch.py
- Command line
-
hf download hf://spaces/hembad/reranker/src/models/prefetch.py
-
curl -L -o prefetch.py https://huggingface.co/spaces/hembad/reranker/resolve/main/src/models/prefetch.py
2.99 kB
| """Pre-download + warm EVERY runtime model β the build-time ``ranker-setup`` so ranking runs OFFLINE. | |
| The reproduce contract is "no network during ranking": models must already be on disk before the timed run. | |
| This warms all four into the in-image cache (``data/artifacts/models`` via the ``HF_HOME`` redirect) at Docker | |
| **build** time, so the container never touches the network while ranking. Run manually with | |
| ``uv run python rank.py --setup``. | |
| Model set (LFM2.5-350M, Qwen3.5-2B, Qwen3-4B/0.6B, and Gemma-3-4B/1B were all benchmarked and dropped β | |
| Qwen3-1.7B won on speed while still clearing the min_skills recall target with correctly-atomized keywords): | |
| * Qwen3-1.7B-GGUF β JD parse + T1 reasoning polish (top 10 only, when reasoning.mode == "llm") | |
| * potion-base-32M β Step 05 broad embed + NLG variant selection | |
| * granite-r2 ONNX β Step 08 precision embed | |
| * granite-reranker β Step 09 cross-encoder | |
| """ | |
| from __future__ import annotations | |
| from common.logging import get_logger | |
| from config import Settings, load_settings | |
| from .llm import index_gen_client | |
| def ensure_all_models(settings: Settings | None = None) -> None: | |
| """(Down)load + warm every model the pipeline uses. Safe to re-run (cached loads no-op).""" | |
| active = settings or load_settings() | |
| log = get_logger("models") | |
| # 1) LLM (llama.cpp GGUF) β does JD parse AND the Step-11 outcome notes (T1, top 10 only, mode == "llm"). | |
| # LFM2.5-350M was verified useless for reasoning (3/15, verbatim copies) and is no longer loaded/prefetched. | |
| index_gen_client(active).ensure() | |
| log.info("prefetch.llm_ready", model=active.models.index_gen_llm_repo) | |
| # 2) Static + ONNX retrieval models β warm via the same private loaders the pipeline uses. | |
| from retrieve.granite import _load_granite | |
| from retrieve.potion import _load_potion | |
| from retrieve.reranker import _load_reranker | |
| # cpu_threads is threaded through so the @lru_cache key matches what the real rank run will request β if it | |
| # didn't, a non-default compute.cpu_threads would cache-MISS at rank time and rebuild the ONNX session from | |
| # scratch (weights are still local, so not a network issue, but it would waste the whole point of prefetch). | |
| threads = active.compute.cpu_threads | |
| _load_potion(active.models.embedder_broad) | |
| log.info("prefetch.embed_ready", model=active.models.embedder_broad) | |
| _load_granite(active.models.embedder_precision, active.precision.onnx_file, active.precision.max_length, threads) | |
| log.info("prefetch.embed_ready", model=active.models.embedder_precision) | |
| _load_reranker(active.models.reranker, active.rerank.onnx_file, active.rerank.max_length, threads) | |
| log.info("prefetch.embed_ready", model=active.models.reranker) | |
| log.info("prefetch.complete") | |
| if __name__ == "__main__": | |
| from common.runtime import configure_compute | |
| active = load_settings() | |
| configure_compute(active.compute.cpu_threads) | |
| ensure_all_models(active) | |