File size: 5,779 Bytes
da5cba1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
"""Models on HF Inference Providers, what they cost, and which make good judges.

The agent needs tool calling, so only models with a live tool-capable provider are offered. Each
call is pinned to one provider (`model:provider`) so the price shown is the price paid.
"""

from __future__ import annotations

import threading
import time

import httpx

from . import config

_cache: dict = {"at": 0.0, "models": []}
_lock = threading.Lock()

# Frontier agentic models, shown first. Everything else with tool calling is still selectable.
FEATURED = ["moonshotai/Kimi-K3", "zai-org/GLM-5.3", "deepseek-ai/DeepSeek-V4-Pro", "Qwen/Qwen3.8-2.4T-A95B",
            "thinkingmachines/Inkling", "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "MiniMaxAI/MiniMax-M3",
            "Qwen/Qwen3.5-397B-A17B", "deepseek-ai/DeepSeek-V4.1-Flash", "zai-org/GLM-5.3-Flash"]
DEFAULT_AGENT = "zai-org/GLM-5.3"

# Judges measured on the real General verifier prompts, English and Chinese. The verifier needs each reply keyed by
# the check's id; a reply keyed otherwise (gpt-oss-120b sometimes echoes the prompt's placeholder "检查点id") is no
# verdict, and after one retry the whole rollout goes unscored. Reasoning-first models were dropped: they often spend
# the judge's 4k-token budget thinking and return nothing.
TEXT_JUDGES = [
    {"id": "thinkingmachines/Inkling", "note": "most reliable in our tests (14/14 usable verdicts), fast"},
    {"id": "moonshotai/Kimi-K3", "note": "14/14 usable verdicts, frontier"},
    {"id": "deepseek-ai/DeepSeek-V4-Pro", "note": "14/14 usable verdicts, more lenient"},
    {"id": "zai-org/GLM-5.3-Flash", "note": "13/14 usable verdicts, fast"},
    {"id": "openai/gpt-oss-120b", "note": "fastest, but 12/14: some rollouts go unscored"},
]
# Measured on a real webdev render with Xiaomi's vision rubric (45 vision models on the router, two passes).
# These returned a usable verdict both times and scored near the median (~0.70). Judges disagree a lot
# (0.34-0.87 on the same page) and the rubric runs at temperature 1.0, so compare scores per judge.
VISION_JUDGES = [
    {"id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "note": "fastest (~4s), near the median"},
    {"id": "moonshotai/Kimi-K2.7-Code", "note": "fast (~7s), near the median"},
    {"id": "Qwen/Qwen3.8-27B", "note": "fastest (~4s), slightly strict"},
    {"id": "deepseek-ai/DeepSeek-V4.1-Flash", "note": "fast, slightly strict"},
    {"id": "moonshotai/Kimi-K3", "note": "frontier, slow (~1 min)"},
]


def _p(x):
    """Provider prices arrive as floats with noise (5.999999999999999); keep four significant digits."""
    return None if x is None else float(f"{float(x):.4g}")


def _fetch() -> list[dict]:
    r = httpx.get(f"{config.ROUTER}/models", timeout=20)
    r.raise_for_status()
    out = []
    for m in r.json()["data"]:
        live = [p for p in m.get("providers", []) if p.get("status") == "live"]
        tools = [p for p in live if p.get("supports_tools")]
        if not live:
            continue
        pick = min(tools or live, key=lambda p: ((p.get("pricing") or {}).get("output") or 1e9))
        price = pick.get("pricing") or {}
        arch = m.get("architecture") or {}
        out.append({
            "id": m["id"],
            "provider": pick["provider"],
            "tools": bool(tools),
            "vision": "image" in (arch.get("input_modalities") or []),
            "input": _p(price.get("input")), "output": _p(price.get("output")),
            "context": pick.get("context_length"), "speed": round(pick.get("throughput") or 0),
            "latency_ms": round(pick.get("first_token_latency_ms") or 0),
            "providers": [{"name": p["provider"], "input": _p((p.get("pricing") or {}).get("input")),
                           "output": _p((p.get("pricing") or {}).get("output")), "tools": bool(p.get("supports_tools")),
                           "speed": round(p.get("throughput") or 0)} for p in live],
            "featured": m["id"] in FEATURED,
        })
    order = {m: i for i, m in enumerate(FEATURED)}
    out.sort(key=lambda m: (order.get(m["id"], 999), -(m["output"] or 0)))
    return out


def all_models() -> list[dict]:
    with _lock:
        if time.time() - _cache["at"] > 900 or not _cache["models"]:
            try:
                _cache["models"] = _fetch()
                _cache["at"] = time.time()
            except Exception:
                if not _cache["models"]:
                    raise
        return _cache["models"]


def get(model_id: str) -> dict | None:
    return next((m for m in all_models() if m["id"] == model_id), None)


def price_of(model_id: str, provider: str | None = None) -> tuple[float, float]:
    """$ per 1M (input, output) tokens for this model on this provider."""
    m = get(model_id) or {}
    p = next((x for x in m.get("providers", []) if x["name"] == (provider or m.get("provider"))), None) or m
    return float(p.get("input") or 0), float(p.get("output") or 0)


def cost(model_id: str, provider: str | None, tokens: dict) -> float:
    pin, pout = price_of(model_id, provider)
    inp = tokens.get("input", 0) + tokens.get("cache_read", 0) + tokens.get("cache_write", 0)
    out = tokens.get("output", 0) + tokens.get("reasoning", 0)   # reasoning tokens bill as output
    return (inp * pin + out * pout) / 1e6


def catalog() -> dict:
    ms = all_models()
    ids = {m["id"] for m in ms}
    return {
        "agents": [m for m in ms if m["tools"]],
        "text_judges": [{**j, **(get(j["id"]) or {})} for j in TEXT_JUDGES if j["id"] in ids],
        "vision_judges": [{**j, **(get(j["id"]) or {})} for j in VISION_JUDGES if j["id"] in ids],
        "default_agent": DEFAULT_AGENT,
        "sandbox_price_per_hour": config.FLAVOR_PRICE_PER_HOUR,
    }