bot_host / admin_stats.py
ItsBounvy's picture
Upload 27 files
f598ecc verified
Raw History Blame Contribute Delete
8.59 kB
"""Admin-panel metrics: host load + per-user / per-bot breakdown.
Centralised here (instead of inline in routes.py) so:
* The HTML ``/admin/stats`` page and the JSON ``/api/admin/stats``
endpoint read from the same source — guaranteed parity.
* Tests can exercise the collector directly without spinning up
a full FastAPI client + DB session for what is fundamentally
a data-shape test.
Layout of the returned dict
---------------------------
::
{
"host": {cpu_*, ram_*, disk_*, boot_time, uptime_sec, ok},
"totals": {bots, running, cpu_live, ram_live_mb, bots_by_status: {...}},
"users": [ {user, bots_total, bots_running, cpu_live_percent,
ram_live_mb, cpu_alloc_cores, ram_alloc_mb,
storage_used_mb, bots_by_status: {...}} ... ],
"ts": float, # unix time at collection
}
Failure modes
-------------
``psutil`` is best-effort: if it's missing or the host denies the
read, host fields become ``None`` and ``host.ok`` flips to ``False``.
The rest of the dashboard still renders — a blank host card beats
a 500 on the whole page.
"""
from __future__ import annotations
import time
from datetime import datetime
from typing import Any
from sqlalchemy import select, func
from sqlalchemy.ext.asyncio import AsyncSession
try:
import psutil # type: ignore
_HAS_PSUTIL = True
except Exception: # noqa: BLE001
_HAS_PSUTIL = False
# Prime psutil's CPU% baseline once at import so the very first
# ``cpu_percent(interval=None)`` call returns a real number instead
# of the always-zero first sample. Without this priming the
# dashboard would show "0%" for one render after every restart.
if _HAS_PSUTIL:
try:
psutil.cpu_percent(interval=None)
except Exception: # noqa: BLE001
pass
# ---------------------------------------------------------------------------
# Host metrics
# ---------------------------------------------------------------------------
def _collect_host() -> dict[str, Any]:
"""Snapshot of the host the panel itself runs on."""
out: dict[str, Any] = {
"ok": False,
"cpu_percent": None,
"cpu_count": None,
"ram_used_mb": None,
"ram_total_mb": None,
"ram_percent": None,
"disk_used_mb": None,
"disk_total_mb": None,
"disk_percent": None,
"boot_time": None,
"uptime_sec": None,
"error": None,
}
if not _HAS_PSUTIL:
out["error"] = "psutil unavailable"
return out
try:
out["cpu_percent"] = float(psutil.cpu_percent(interval=0.15))
out["cpu_count"] = psutil.cpu_count() or 1
vm = psutil.virtual_memory()
out["ram_total_mb"] = round(vm.total / (1024 * 1024), 1)
out["ram_used_mb"] = round((vm.total - vm.available) / (1024 * 1024), 1)
out["ram_percent"] = float(vm.percent)
try:
# The panel runs on Render / HF Spaces — data_dir is the
# most relevant mount, not necessarily "/".
from config import PANEL_CONFIG # type: ignore
du = psutil.disk_usage(str(PANEL_CONFIG.data_dir))
except Exception: # noqa: BLE001
du = psutil.disk_usage("/")
out["disk_total_mb"] = round(du.total / (1024 * 1024), 1)
out["disk_used_mb"] = round(du.used / (1024 * 1024), 1)
out["disk_percent"] = float(du.percent)
boot = psutil.boot_time()
out["boot_time"] = float(boot)
out["uptime_sec"] = int(time.time() - boot)
out["ok"] = True
return out
except Exception as exc: # noqa: BLE001
out["error"] = str(exc)
return out
# ---------------------------------------------------------------------------
# Bot-population metrics
# ---------------------------------------------------------------------------
async def _bot_status_counts(db: AsyncSession) -> dict[str, int]:
"""Count BotInstance rows grouped by status. Returns all known
statuses even when the count is 0, so the dashboard always renders
the same row order."""
# Imported lazily so this module can be imported standalone
# (e.g. by tests) without pulling the whole models graph.
from models import BotInstance, BotStatus
counts: dict[str, int] = {s.value: 0 for s in BotStatus}
rows = (await db.execute(
select(BotInstance.status, func.count(BotInstance.id))
.group_by(BotInstance.status)
)).all()
for status, count in rows:
key = status.value if hasattr(status, "value") else str(status)
counts[key] = int(count)
return counts
async def collect_admin_stats(db: AsyncSession) -> dict[str, Any]:
"""Build the full admin-stats snapshot.
The HTML page (``/admin/stats``) and the JSON endpoint
(``/api/admin/stats``) both call this — they only differ in how
they render the result.
"""
from models import (
BotInstance,
BotStatus,
DeploymentMode,
User,
)
host = _collect_host()
bots_by_status = await _bot_status_counts(db)
# One query, group by owner + status, so the per-user rollup
# is a single DB round-trip instead of N+1.
rows = (await db.execute(
select(
BotInstance.owner_id,
BotInstance.status,
func.count(BotInstance.id),
).group_by(BotInstance.owner_id, BotInstance.status)
)).all()
# owner_id -> {status -> count}
per_owner_status: dict[int, dict[str, int]] = {}
for owner_id, status, count in rows:
per_owner_status.setdefault(int(owner_id), {})[
status.value if hasattr(status, "value") else str(status)
] = int(count)
users_result = await db.execute(select(User).order_by(User.username.asc()))
users = list(users_result.scalars().all())
# Single query for all bots; per-user rows are O(1) lookups after.
all_bots = list((await db.execute(select(BotInstance))).scalars().all())
bots_by_owner: dict[int, list[BotInstance]] = {}
for b in all_bots:
bots_by_owner.setdefault(b.owner_id, []).append(b)
user_rows: list[dict[str, Any]] = []
totals = {
"bots": 0,
"running": 0,
"cpu_live": 0.0,
"ram_live_mb": 0.0,
"bots_by_status": dict(bots_by_status),
}
for u in users:
owned = bots_by_owner.get(u.id, [])
running_bots = [b for b in owned if b.status == BotStatus.RUNNING]
# Live process metrics only meaningful for MULTITENANT bots;
# LEGACY_SPACE bots run on their own HF Space and aren't
# visible to this process.
cpu_live = 0.0
ram_live_mb = 0.0
for b in owned:
if b.deployment_mode == DeploymentMode.MULTITENANT:
try:
import bot_runner # local import to avoid heavy cost on tests
st = bot_runner.bot_status(b)
if st.get("running"):
cpu_live += float(st.get("cpu_percent", 0.0))
ram_live_mb += float(st.get("rss_mb", 0.0))
except Exception: # noqa: BLE001
# If bot_runner can't probe (process exited, no
# permissions, …) we just don't add to the live
# totals — better than crashing the whole page.
pass
ram_alloc_mb = sum((b.ram_mb or 0) for b in running_bots)
cpu_alloc_cores = sum((b.cpu_cores or 0.0) for b in running_bots)
storage_mb = sum((b.storage_used_mb or 0) for b in owned)
row = {
"user": u,
"username": u.username,
"bots_total": len(owned),
"bots_running": len(running_bots),
"bots_by_status": dict(per_owner_status.get(u.id, {})),
"cpu_live_percent": round(cpu_live, 1),
"ram_live_mb": round(ram_live_mb, 1),
"cpu_alloc_cores": round(cpu_alloc_cores, 2),
"ram_alloc_mb": ram_alloc_mb,
"storage_used_mb": storage_mb,
}
user_rows.append(row)
totals["bots"] += len(owned)
totals["running"] += len(running_bots)
totals["cpu_live"] += cpu_live
totals["ram_live_mb"] += ram_live_mb
return {
"host": host,
"totals": totals,
"users": user_rows,
"ts": time.time(),
"ts_iso": datetime.utcnow().isoformat(timespec="seconds") + "Z",
}
__all__ = ["collect_admin_stats", "_collect_host", "_bot_status_counts"]