Trifecta-Lab / src /lmstudio_model.py
Brettapps's picture
Upload folder using huggingface_hub (part 21)
e23172f verified
Raw History Blame Contribute Delete
3.19 kB
"""LM Studio model integration for Trifecta-Bro.
Model id: Brettapps/trifecta-bro/v1
This module is the bridge between Trifecta-Bro and LM Studio's `lms` server
(OpenAI-compatible API on http://localhost:1234/v1). It is installed here now;
the Obsidian-vault backend wiring is added later once the model weights / prompt
pipeline are ready.
Responsibilities today:
- Resolve the LM Studio server base URL.
- Expose `Brettapps/trifecta-bro/v1` as the canonical model identifier.
- Provide a health check so the rest of the app can detect when the model
is unavailable and fall back to the open-source `TrifectaPredictor`.
Responsibilities later (vault backend):
- Load the *Fields & Form* note from the Obsidian vault.
- Build the Hermes trifecta prompt.
- Call the LM Studio model and parse the response back into predictions.
"""
from __future__ import annotations
import os
import urllib.request
from dataclasses import dataclass
# Canonical model identifier used by Trifecta-Bro against LM Studio.
MODEL_ID = "Brettapps/trifecta-bro/v1"
# LM Studio serves an OpenAI-compatible API by default on this port.
DEFAULT_LMSTUDIO_BASE_URL = "http://localhost:1234/v1"
# The `lms` server only accepts a chat/completions call for a model that is
# loaded. We expose the id here so the server can be primed with:
# lms load Brettapps/trifecta-bro/v1
# (or by importing the GGUF into LM Studio's model library).
@dataclass
class LMStudioStatus:
reachable: bool
base_url: str
model_id: str
detail: str = ""
def resolve_base_url() -> str:
"""Return the LM Studio base URL, allowing override via env."""
return os.environ.get("LMSTUDIO_BASE_URL", DEFAULT_LMSTUDIO_BASE_URL).rstrip("/")
def get_model_id() -> str:
"""Canonical model identifier for Trifecta-Bro v1."""
return MODEL_ID
def health_check(base_url: str | None = None, timeout: float = 3.0) -> LMStudioStatus:
"""Check whether the LM Studio server is up and lists our model.
This is a lightweight probe of the /v1/models endpoint. It does not require
the model to be loaded to report reachability.
"""
url = (base_url or resolve_base_url()) + "/models"
try:
req = urllib.request.Request(url, headers={"Accept": "application/json"})
with urllib.request.urlopen(req, timeout=timeout) as resp:
body = resp.read().decode("utf-8", "ignore")
reachable = resp.status == 200 # type: ignore[attr-defined]
detail = "ok" if reachable else f"status {getattr(resp, 'status', '?')}"
# Peek whether our model id is present (non-fatal if not).
if reachable and MODEL_ID not in body:
detail = "server up; model not loaded"
return LMStudioStatus(
reachable=reachable,
base_url=(base_url or resolve_base_url()),
model_id=MODEL_ID,
detail=detail,
)
except Exception as exc: # noqa: BLE001 - network probe, report failure
return LMStudioStatus(
reachable=False,
base_url=(base_url or resolve_base_url()),
model_id=MODEL_ID,
detail=str(exc) or "unreachable",
)