Spaces:
Running on CPU Upgrade
Running on CPU Upgrade
File size: 5,779 Bytes
da5cba1 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 | """Models on HF Inference Providers, what they cost, and which make good judges.
The agent needs tool calling, so only models with a live tool-capable provider are offered. Each
call is pinned to one provider (`model:provider`) so the price shown is the price paid.
"""
from __future__ import annotations
import threading
import time
import httpx
from . import config
_cache: dict = {"at": 0.0, "models": []}
_lock = threading.Lock()
# Frontier agentic models, shown first. Everything else with tool calling is still selectable.
FEATURED = ["moonshotai/Kimi-K3", "zai-org/GLM-5.3", "deepseek-ai/DeepSeek-V4-Pro", "Qwen/Qwen3.8-2.4T-A95B",
"thinkingmachines/Inkling", "nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", "MiniMaxAI/MiniMax-M3",
"Qwen/Qwen3.5-397B-A17B", "deepseek-ai/DeepSeek-V4.1-Flash", "zai-org/GLM-5.3-Flash"]
DEFAULT_AGENT = "zai-org/GLM-5.3"
# Judges measured on the real General verifier prompts, English and Chinese. The verifier needs each reply keyed by
# the check's id; a reply keyed otherwise (gpt-oss-120b sometimes echoes the prompt's placeholder "检查点id") is no
# verdict, and after one retry the whole rollout goes unscored. Reasoning-first models were dropped: they often spend
# the judge's 4k-token budget thinking and return nothing.
TEXT_JUDGES = [
{"id": "thinkingmachines/Inkling", "note": "most reliable in our tests (14/14 usable verdicts), fast"},
{"id": "moonshotai/Kimi-K3", "note": "14/14 usable verdicts, frontier"},
{"id": "deepseek-ai/DeepSeek-V4-Pro", "note": "14/14 usable verdicts, more lenient"},
{"id": "zai-org/GLM-5.3-Flash", "note": "13/14 usable verdicts, fast"},
{"id": "openai/gpt-oss-120b", "note": "fastest, but 12/14: some rollouts go unscored"},
]
# Measured on a real webdev render with Xiaomi's vision rubric (45 vision models on the router, two passes).
# These returned a usable verdict both times and scored near the median (~0.70). Judges disagree a lot
# (0.34-0.87 on the same page) and the rubric runs at temperature 1.0, so compare scores per judge.
VISION_JUDGES = [
{"id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct-FP8", "note": "fastest (~4s), near the median"},
{"id": "moonshotai/Kimi-K2.7-Code", "note": "fast (~7s), near the median"},
{"id": "Qwen/Qwen3.8-27B", "note": "fastest (~4s), slightly strict"},
{"id": "deepseek-ai/DeepSeek-V4.1-Flash", "note": "fast, slightly strict"},
{"id": "moonshotai/Kimi-K3", "note": "frontier, slow (~1 min)"},
]
def _p(x):
"""Provider prices arrive as floats with noise (5.999999999999999); keep four significant digits."""
return None if x is None else float(f"{float(x):.4g}")
def _fetch() -> list[dict]:
r = httpx.get(f"{config.ROUTER}/models", timeout=20)
r.raise_for_status()
out = []
for m in r.json()["data"]:
live = [p for p in m.get("providers", []) if p.get("status") == "live"]
tools = [p for p in live if p.get("supports_tools")]
if not live:
continue
pick = min(tools or live, key=lambda p: ((p.get("pricing") or {}).get("output") or 1e9))
price = pick.get("pricing") or {}
arch = m.get("architecture") or {}
out.append({
"id": m["id"],
"provider": pick["provider"],
"tools": bool(tools),
"vision": "image" in (arch.get("input_modalities") or []),
"input": _p(price.get("input")), "output": _p(price.get("output")),
"context": pick.get("context_length"), "speed": round(pick.get("throughput") or 0),
"latency_ms": round(pick.get("first_token_latency_ms") or 0),
"providers": [{"name": p["provider"], "input": _p((p.get("pricing") or {}).get("input")),
"output": _p((p.get("pricing") or {}).get("output")), "tools": bool(p.get("supports_tools")),
"speed": round(p.get("throughput") or 0)} for p in live],
"featured": m["id"] in FEATURED,
})
order = {m: i for i, m in enumerate(FEATURED)}
out.sort(key=lambda m: (order.get(m["id"], 999), -(m["output"] or 0)))
return out
def all_models() -> list[dict]:
with _lock:
if time.time() - _cache["at"] > 900 or not _cache["models"]:
try:
_cache["models"] = _fetch()
_cache["at"] = time.time()
except Exception:
if not _cache["models"]:
raise
return _cache["models"]
def get(model_id: str) -> dict | None:
return next((m for m in all_models() if m["id"] == model_id), None)
def price_of(model_id: str, provider: str | None = None) -> tuple[float, float]:
"""$ per 1M (input, output) tokens for this model on this provider."""
m = get(model_id) or {}
p = next((x for x in m.get("providers", []) if x["name"] == (provider or m.get("provider"))), None) or m
return float(p.get("input") or 0), float(p.get("output") or 0)
def cost(model_id: str, provider: str | None, tokens: dict) -> float:
pin, pout = price_of(model_id, provider)
inp = tokens.get("input", 0) + tokens.get("cache_read", 0) + tokens.get("cache_write", 0)
out = tokens.get("output", 0) + tokens.get("reasoning", 0) # reasoning tokens bill as output
return (inp * pin + out * pout) / 1e6
def catalog() -> dict:
ms = all_models()
ids = {m["id"] for m in ms}
return {
"agents": [m for m in ms if m["tools"]],
"text_judges": [{**j, **(get(j["id"]) or {})} for j in TEXT_JUDGES if j["id"] in ids],
"vision_judges": [{**j, **(get(j["id"]) or {})} for j in VISION_JUDGES if j["id"] in ids],
"default_agent": DEFAULT_AGENT,
"sandbox_price_per_hour": config.FLAVOR_PRICE_PER_HOUR,
}
|