File size: 11,306 Bytes
dfb775d | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 | """automindXtrain FastAPI app.
Exposes:
GET / β coach UI (mindXtrain Coach)
GET /health β liveness check
POST /v1/chat/completions β OpenAI-compatible chat
POST /v1/agentic β mindX-native agentic dispatch (Day 5+)
/v1/training/jobs/* β public training-jobs API (mindX agents,
external clients). Bearer auth via
MINDXTRAIN_API_KEY when set.
GET /coach/* β Coach UI + API (recipes, autotune, cost)
The production deployment lives at https://mindx.pythai.net β the Coach UI
is at /coach/ and the public training-jobs API is at /v1/training/jobs.
"""
from __future__ import annotations
import logging
import os
from contextlib import asynccontextmanager
from pathlib import Path
from typing import TYPE_CHECKING, Any
import httpx
from fastapi import FastAPI, HTTPException
if TYPE_CHECKING:
from collections.abc import AsyncIterator
from fastapi.responses import RedirectResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel
from mindxtrain import __version__
from mindxtrain.models.registry import ChatRequest, ChatResponse, build_backend
from mindxtrain.operator.coach import router as coach_router
from mindxtrain.operator.training_api import router as training_router
# ---- backend resolution --------------------------------------------------
def _ollama_reachable(timeout_s: float = 1.0) -> bool:
"""Probe ollama at MINDXTRAIN_OLLAMA_BASE_URL.
Used by auto-detect to pick `ollama` as the default backend on hosts
where ollama is the only thing running (e.g., the laptop dev
environment). Strips `/v1` from the configured base URL because
ollama's health-style endpoint is `/api/tags`, not OpenAI-shaped.
"""
base = os.environ.get("MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1")
probe_url = base.rstrip("/").removesuffix("/v1") + "/api/tags"
try:
with httpx.Client(timeout=timeout_s) as client:
return client.get(probe_url).status_code == 200
except (httpx.HTTPError, OSError):
return False
def _vllm_reachable(timeout_s: float = 1.0) -> bool:
"""Probe vLLM at MINDXTRAIN_VLLM_BASE_URL.
Hits `/v1/models` β the OpenAI-compatible models listing endpoint
vLLM always exposes. This is what flips the production Coach chat
card on the MI300X droplet from "(no backend configured)" to
live, and what `/health` consults so a load balancer knows when
inference is actually warm.
"""
base = os.environ.get(
"MINDXTRAIN_VLLM_BASE_URL",
os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
)
probe_url = base.rstrip("/") + "/models"
try:
with httpx.Client(timeout=timeout_s) as client:
return client.get(probe_url).status_code == 200
except (httpx.HTTPError, OSError):
return False
def _vllm_first_model() -> str | None:
"""Return the first model id vLLM lists, or None on failure.
Lets `/coach/api/health` render "vllm (Qwen/Qwen3-8B) ready" in
prod the same way ollama does on the laptop. Best-effort: probe
failure β None and the UI degrades to just the backend name.
"""
base = os.environ.get(
"MINDXTRAIN_VLLM_BASE_URL",
os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
)
probe_url = base.rstrip("/") + "/models"
try:
with httpx.Client(timeout=1.0) as client:
resp = client.get(probe_url)
if resp.status_code != 200:
return None
body = resp.json()
models = body.get("data", [])
if not models:
return None
first = models[0]
return first.get("id") if isinstance(first, dict) else None
except (httpx.HTTPError, OSError, ValueError, IndexError):
return None
def backend_reachable(name: str) -> bool:
"""Live probe for a backend by name. Used by both /health and /coach health."""
if name == "ollama":
return _ollama_reachable()
if name == "vllm":
return _vllm_reachable()
# openai_compat and unknown backends: we don't have a generic probe,
# so the chat-completions failure path remains the authoritative signal.
return False
def backend_first_model(name: str) -> str | None:
"""Best-effort first-model lookup; None when the backend doesn't list one."""
if name == "ollama":
return ollama_first_model()
if name == "vllm":
return _vllm_first_model()
return None
def resolve_backend_name() -> str:
"""Pick the active backend.
Resolution order:
1. Explicit `MINDXTRAIN_BACKEND` env var (canonical).
2. Legacy `AUTOMINDX_BACKEND` (back-compat with the pre-rename code).
3. Auto-detect: ollama if reachable on localhost:11434, else vllm.
"""
explicit = (
os.environ.get("MINDXTRAIN_BACKEND")
or os.environ.get("AUTOMINDX_BACKEND")
)
if explicit:
return explicit
if _ollama_reachable():
return "ollama"
return "vllm"
def ollama_first_model() -> str | None:
"""Return the name of the first model ollama lists, or None on failure.
Used by the Coach health endpoint to render
`ollama (qwen3:0.6b) ready` instead of just `ollama ready`. Best-effort:
a timeout / parse failure returns None, the UI still shows the backend
name without a model qualifier.
"""
base = os.environ.get("MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1")
probe_url = base.rstrip("/").removesuffix("/v1") + "/api/tags"
try:
with httpx.Client(timeout=1.0) as client:
resp = client.get(probe_url)
if resp.status_code != 200:
return None
body = resp.json()
models = body.get("models", [])
# Prefer local (non-cloud) models first; the user's qwen3:0.6b
# ranks ahead of glm-5.1:cloud, deepseek-v4-pro:cloud, etc.
local = [m for m in models if ":cloud" not in (m.get("name") or "")]
chosen = (local or models)[0] if (local or models) else None
return chosen.get("name") if chosen else None
except (httpx.HTTPError, OSError, ValueError, IndexError):
return None
@asynccontextmanager
async def _lifespan(_app: FastAPI) -> AsyncIterator[None]:
"""Operator startup β optionally auto-launch a hands-free CPU training run.
When `MINDXTRAIN_AUTOSTART` is set the operator kicks off a CPU
training run the moment uvicorn boots, so the Coach UI shows a live
session without anyone pressing "Run training". Autostart is off by
default so `TestClient` lifespans and CI never spawn a trainer.
Failures are swallowed β a bad autostart must never block boot.
"""
from mindxtrain.operator.coach.api import autostart_cpu_training
try:
await autostart_cpu_training()
except Exception: # boot must survive any autostart fault
logging.getLogger("mindxtrain.operator").exception(
"autostart raised β Coach UI still available, launch manually",
)
yield
app = FastAPI(
title="automindXtrain",
version=__version__,
description="Pluggable LLM cognitive runtime for the mindXtrain pipeline.",
lifespan=_lifespan,
)
# --- coach UI -------------------------------------------------------------
class _NoCacheStaticFiles(StaticFiles):
"""StaticFiles that forces browser revalidation.
The Coach JS/CSS change frequently; without this, browsers serve a stale
`coach.js` against fresh `index.html` (visible controls that don't wire up).
`no-cache` still allows efficient 304s via ETag β it just never serves stale.
"""
async def get_response(self, path: str, scope: Any) -> Any:
response = await super().get_response(path, scope)
response.headers["Cache-Control"] = "no-cache, must-revalidate"
return response
_COACH_STATIC = Path(__file__).parent / "coach" / "static"
app.mount("/coach/static", _NoCacheStaticFiles(directory=_COACH_STATIC), name="coach-static")
app.include_router(coach_router)
app.include_router(training_router)
@app.get("/", include_in_schema=False)
async def root() -> RedirectResponse:
"""Land on the Coach UI."""
return RedirectResponse(url="/coach/")
class HealthResponse(BaseModel):
status: str
version: str
backend: str
backend_ready: bool
backend_model: str = ""
coach_url: str
@app.get("/health", response_model=HealthResponse)
async def health() -> HealthResponse:
"""Liveness β always 200.
`status` is "ok" even when the backend is unreachable so simple
load-balancer health checks don't take the operator out of
rotation just because vLLM is still warming. The structured
`backend_ready` field is what an inference-aware probe should
consult; `/readyz` enforces it as the HTTP status.
"""
backend = resolve_backend_name()
ready = backend_reachable(backend)
return HealthResponse(
status="ok",
version=__version__,
backend=backend,
backend_ready=ready,
backend_model=(backend_first_model(backend) or "") if ready else "",
coach_url="/coach/",
)
@app.get("/readyz", include_in_schema=False)
async def readyz() -> dict[str, object]:
"""Readiness gate β 503 when the resolved backend is unreachable.
Use this when you want a probe that *fails* until inference is
actually warm (e.g., k8s readiness probe, uptime monitor that
pages on inference outage rather than process death).
"""
backend = resolve_backend_name()
if not backend_reachable(backend):
raise HTTPException(
status_code=503,
detail={"backend": backend, "reachable": False},
)
return {"backend": backend, "reachable": True}
@app.post("/v1/chat/completions", response_model=ChatResponse)
async def chat_completions(request: ChatRequest) -> ChatResponse:
backend_name = resolve_backend_name()
backend_kwargs: dict[str, object] = {}
if backend_name == "vllm":
backend_kwargs["base_url"] = os.environ.get(
"MINDXTRAIN_VLLM_BASE_URL",
os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
)
elif backend_name == "ollama":
backend_kwargs["base_url"] = os.environ.get(
"MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1",
)
elif backend_name == "openai_compat":
backend_kwargs["base_url"] = os.environ["MINDXTRAIN_OPENAI_BASE_URL"]
backend_kwargs["api_key"] = os.environ.get("MINDXTRAIN_OPENAI_API_KEY", "")
try:
backend = build_backend(backend_name, **backend_kwargs)
return await backend.chat(request)
except NotImplementedError as exc:
raise HTTPException(status_code=501, detail=str(exc)) from exc
except KeyError as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
@app.post("/v1/agentic")
async def agentic() -> dict[str, str]:
"""mindX-native agentic endpoint (Day 5+)."""
raise HTTPException(status_code=501, detail="TODO Day 5: wire mindX MASTERMIND dispatch")
|