File size: 2,994 Bytes
e264368
 
 
d68637e
e264368
 
 
baeef08
 
 
e264368
 
 
 
 
 
 
 
 
 
64b3fbe
e264368
 
 
 
 
 
 
baeef08
a41b023
baeef08
 
e264368
 
 
 
 
 
d68637e
 
 
 
e264368
 
d68637e
e264368
d68637e
e264368
 
 
 
 
 
 
a41b023
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
"""Pre-download + warm EVERY runtime model β€” the build-time ``ranker-setup`` so ranking runs OFFLINE.

The reproduce contract is "no network during ranking": models must already be on disk before the timed run.
This warms all four into the in-image cache (``data/artifacts/models`` via the ``HF_HOME`` redirect) at Docker
**build** time, so the container never touches the network while ranking. Run manually with
``uv run python rank.py --setup``.

Model set (LFM2.5-350M, Qwen3.5-2B, Qwen3-4B/0.6B, and Gemma-3-4B/1B were all benchmarked and dropped β€”
Qwen3-1.7B won on speed while still clearing the min_skills recall target with correctly-atomized keywords):
  * Qwen3-1.7B-GGUF   β€” JD parse + T1 reasoning polish (top 10 only, when reasoning.mode == "llm")
  * potion-base-32M   β€” Step 05 broad embed + NLG variant selection
  * granite-r2 ONNX   β€” Step 08 precision embed
  * granite-reranker  β€” Step 09 cross-encoder
"""

from __future__ import annotations

from common.logging import get_logger
from config import Settings, load_settings

from .llm import index_gen_client


def ensure_all_models(settings: Settings | None = None) -> None:
    """(Down)load + warm every model the pipeline uses. Safe to re-run (cached loads no-op)."""
    active = settings or load_settings()
    log = get_logger("models")

    # 1) LLM (llama.cpp GGUF) β€” does JD parse AND the Step-11 outcome notes (T1, top 10 only, mode == "llm").
    # LFM2.5-350M was verified useless for reasoning (3/15, verbatim copies) and is no longer loaded/prefetched.
    index_gen_client(active).ensure()
    log.info("prefetch.llm_ready", model=active.models.index_gen_llm_repo)

    # 2) Static + ONNX retrieval models β€” warm via the same private loaders the pipeline uses.
    from retrieve.granite import _load_granite
    from retrieve.potion import _load_potion
    from retrieve.reranker import _load_reranker

    # cpu_threads is threaded through so the @lru_cache key matches what the real rank run will request β€” if it
    # didn't, a non-default compute.cpu_threads would cache-MISS at rank time and rebuild the ONNX session from
    # scratch (weights are still local, so not a network issue, but it would waste the whole point of prefetch).
    threads = active.compute.cpu_threads
    _load_potion(active.models.embedder_broad)
    log.info("prefetch.embed_ready", model=active.models.embedder_broad)
    _load_granite(active.models.embedder_precision, active.precision.onnx_file, active.precision.max_length, threads)
    log.info("prefetch.embed_ready", model=active.models.embedder_precision)
    _load_reranker(active.models.reranker, active.rerank.onnx_file, active.rerank.max_length, threads)
    log.info("prefetch.embed_ready", model=active.models.reranker)
    log.info("prefetch.complete")


if __name__ == "__main__":
    from common.runtime import configure_compute

    active = load_settings()
    configure_compute(active.compute.cpu_threads)
    ensure_all_models(active)