File size: 11,306 Bytes
dfb775d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
"""automindXtrain FastAPI app.

Exposes:
    GET  /                        β€” coach UI (mindXtrain Coach)
    GET  /health                  β€” liveness check
    POST /v1/chat/completions     β€” OpenAI-compatible chat
    POST /v1/agentic              β€” mindX-native agentic dispatch (Day 5+)
    /v1/training/jobs/*           β€” public training-jobs API (mindX agents,
                                    external clients). Bearer auth via
                                    MINDXTRAIN_API_KEY when set.
    GET  /coach/*                 β€” Coach UI + API (recipes, autotune, cost)

The production deployment lives at https://mindx.pythai.net β€” the Coach UI
is at /coach/ and the public training-jobs API is at /v1/training/jobs.
"""

from __future__ import annotations

import logging
import os
from contextlib import asynccontextmanager
from pathlib import Path
from typing import TYPE_CHECKING, Any

import httpx
from fastapi import FastAPI, HTTPException

if TYPE_CHECKING:
    from collections.abc import AsyncIterator
from fastapi.responses import RedirectResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel

from mindxtrain import __version__
from mindxtrain.models.registry import ChatRequest, ChatResponse, build_backend
from mindxtrain.operator.coach import router as coach_router
from mindxtrain.operator.training_api import router as training_router

# ---- backend resolution --------------------------------------------------


def _ollama_reachable(timeout_s: float = 1.0) -> bool:
    """Probe ollama at MINDXTRAIN_OLLAMA_BASE_URL.

    Used by auto-detect to pick `ollama` as the default backend on hosts
    where ollama is the only thing running (e.g., the laptop dev
    environment). Strips `/v1` from the configured base URL because
    ollama's health-style endpoint is `/api/tags`, not OpenAI-shaped.
    """
    base = os.environ.get("MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1")
    probe_url = base.rstrip("/").removesuffix("/v1") + "/api/tags"
    try:
        with httpx.Client(timeout=timeout_s) as client:
            return client.get(probe_url).status_code == 200
    except (httpx.HTTPError, OSError):
        return False


def _vllm_reachable(timeout_s: float = 1.0) -> bool:
    """Probe vLLM at MINDXTRAIN_VLLM_BASE_URL.

    Hits `/v1/models` β€” the OpenAI-compatible models listing endpoint
    vLLM always exposes. This is what flips the production Coach chat
    card on the MI300X droplet from "(no backend configured)" to
    live, and what `/health` consults so a load balancer knows when
    inference is actually warm.
    """
    base = os.environ.get(
        "MINDXTRAIN_VLLM_BASE_URL",
        os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
    )
    probe_url = base.rstrip("/") + "/models"
    try:
        with httpx.Client(timeout=timeout_s) as client:
            return client.get(probe_url).status_code == 200
    except (httpx.HTTPError, OSError):
        return False


def _vllm_first_model() -> str | None:
    """Return the first model id vLLM lists, or None on failure.

    Lets `/coach/api/health` render "vllm (Qwen/Qwen3-8B) ready" in
    prod the same way ollama does on the laptop. Best-effort: probe
    failure β†’ None and the UI degrades to just the backend name.
    """
    base = os.environ.get(
        "MINDXTRAIN_VLLM_BASE_URL",
        os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
    )
    probe_url = base.rstrip("/") + "/models"
    try:
        with httpx.Client(timeout=1.0) as client:
            resp = client.get(probe_url)
            if resp.status_code != 200:
                return None
            body = resp.json()
        models = body.get("data", [])
        if not models:
            return None
        first = models[0]
        return first.get("id") if isinstance(first, dict) else None
    except (httpx.HTTPError, OSError, ValueError, IndexError):
        return None


def backend_reachable(name: str) -> bool:
    """Live probe for a backend by name. Used by both /health and /coach health."""
    if name == "ollama":
        return _ollama_reachable()
    if name == "vllm":
        return _vllm_reachable()
    # openai_compat and unknown backends: we don't have a generic probe,
    # so the chat-completions failure path remains the authoritative signal.
    return False


def backend_first_model(name: str) -> str | None:
    """Best-effort first-model lookup; None when the backend doesn't list one."""
    if name == "ollama":
        return ollama_first_model()
    if name == "vllm":
        return _vllm_first_model()
    return None


def resolve_backend_name() -> str:
    """Pick the active backend.

    Resolution order:
    1. Explicit `MINDXTRAIN_BACKEND` env var (canonical).
    2. Legacy `AUTOMINDX_BACKEND` (back-compat with the pre-rename code).
    3. Auto-detect: ollama if reachable on localhost:11434, else vllm.
    """
    explicit = (
        os.environ.get("MINDXTRAIN_BACKEND")
        or os.environ.get("AUTOMINDX_BACKEND")
    )
    if explicit:
        return explicit
    if _ollama_reachable():
        return "ollama"
    return "vllm"


def ollama_first_model() -> str | None:
    """Return the name of the first model ollama lists, or None on failure.

    Used by the Coach health endpoint to render
    `ollama (qwen3:0.6b) ready` instead of just `ollama ready`. Best-effort:
    a timeout / parse failure returns None, the UI still shows the backend
    name without a model qualifier.
    """
    base = os.environ.get("MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1")
    probe_url = base.rstrip("/").removesuffix("/v1") + "/api/tags"
    try:
        with httpx.Client(timeout=1.0) as client:
            resp = client.get(probe_url)
            if resp.status_code != 200:
                return None
            body = resp.json()
        models = body.get("models", [])
        # Prefer local (non-cloud) models first; the user's qwen3:0.6b
        # ranks ahead of glm-5.1:cloud, deepseek-v4-pro:cloud, etc.
        local = [m for m in models if ":cloud" not in (m.get("name") or "")]
        chosen = (local or models)[0] if (local or models) else None
        return chosen.get("name") if chosen else None
    except (httpx.HTTPError, OSError, ValueError, IndexError):
        return None

@asynccontextmanager
async def _lifespan(_app: FastAPI) -> AsyncIterator[None]:
    """Operator startup β€” optionally auto-launch a hands-free CPU training run.

    When `MINDXTRAIN_AUTOSTART` is set the operator kicks off a CPU
    training run the moment uvicorn boots, so the Coach UI shows a live
    session without anyone pressing "Run training". Autostart is off by
    default so `TestClient` lifespans and CI never spawn a trainer.
    Failures are swallowed β€” a bad autostart must never block boot.
    """
    from mindxtrain.operator.coach.api import autostart_cpu_training

    try:
        await autostart_cpu_training()
    except Exception:  # boot must survive any autostart fault
        logging.getLogger("mindxtrain.operator").exception(
            "autostart raised β€” Coach UI still available, launch manually",
        )
    yield


app = FastAPI(
    title="automindXtrain",
    version=__version__,
    description="Pluggable LLM cognitive runtime for the mindXtrain pipeline.",
    lifespan=_lifespan,
)

# --- coach UI -------------------------------------------------------------


class _NoCacheStaticFiles(StaticFiles):
    """StaticFiles that forces browser revalidation.

    The Coach JS/CSS change frequently; without this, browsers serve a stale
    `coach.js` against fresh `index.html` (visible controls that don't wire up).
    `no-cache` still allows efficient 304s via ETag β€” it just never serves stale.
    """

    async def get_response(self, path: str, scope: Any) -> Any:
        response = await super().get_response(path, scope)
        response.headers["Cache-Control"] = "no-cache, must-revalidate"
        return response


_COACH_STATIC = Path(__file__).parent / "coach" / "static"
app.mount("/coach/static", _NoCacheStaticFiles(directory=_COACH_STATIC), name="coach-static")
app.include_router(coach_router)
app.include_router(training_router)


@app.get("/", include_in_schema=False)
async def root() -> RedirectResponse:
    """Land on the Coach UI."""
    return RedirectResponse(url="/coach/")


class HealthResponse(BaseModel):
    status: str
    version: str
    backend: str
    backend_ready: bool
    backend_model: str = ""
    coach_url: str


@app.get("/health", response_model=HealthResponse)
async def health() -> HealthResponse:
    """Liveness β€” always 200.

    `status` is "ok" even when the backend is unreachable so simple
    load-balancer health checks don't take the operator out of
    rotation just because vLLM is still warming. The structured
    `backend_ready` field is what an inference-aware probe should
    consult; `/readyz` enforces it as the HTTP status.
    """
    backend = resolve_backend_name()
    ready = backend_reachable(backend)
    return HealthResponse(
        status="ok",
        version=__version__,
        backend=backend,
        backend_ready=ready,
        backend_model=(backend_first_model(backend) or "") if ready else "",
        coach_url="/coach/",
    )


@app.get("/readyz", include_in_schema=False)
async def readyz() -> dict[str, object]:
    """Readiness gate β€” 503 when the resolved backend is unreachable.

    Use this when you want a probe that *fails* until inference is
    actually warm (e.g., k8s readiness probe, uptime monitor that
    pages on inference outage rather than process death).
    """
    backend = resolve_backend_name()
    if not backend_reachable(backend):
        raise HTTPException(
            status_code=503,
            detail={"backend": backend, "reachable": False},
        )
    return {"backend": backend, "reachable": True}


@app.post("/v1/chat/completions", response_model=ChatResponse)
async def chat_completions(request: ChatRequest) -> ChatResponse:
    backend_name = resolve_backend_name()
    backend_kwargs: dict[str, object] = {}
    if backend_name == "vllm":
        backend_kwargs["base_url"] = os.environ.get(
            "MINDXTRAIN_VLLM_BASE_URL",
            os.environ.get("AUTOMINDX_VLLM_BASE_URL", "http://localhost:8000/v1"),
        )
    elif backend_name == "ollama":
        backend_kwargs["base_url"] = os.environ.get(
            "MINDXTRAIN_OLLAMA_BASE_URL", "http://localhost:11434/v1",
        )
    elif backend_name == "openai_compat":
        backend_kwargs["base_url"] = os.environ["MINDXTRAIN_OPENAI_BASE_URL"]
        backend_kwargs["api_key"] = os.environ.get("MINDXTRAIN_OPENAI_API_KEY", "")

    try:
        backend = build_backend(backend_name, **backend_kwargs)
        return await backend.chat(request)
    except NotImplementedError as exc:
        raise HTTPException(status_code=501, detail=str(exc)) from exc
    except KeyError as exc:
        raise HTTPException(status_code=400, detail=str(exc)) from exc


@app.post("/v1/agentic")
async def agentic() -> dict[str, str]:
    """mindX-native agentic endpoint (Day 5+)."""
    raise HTTPException(status_code=501, detail="TODO Day 5: wire mindX MASTERMIND dispatch")