"""Admin-panel metrics: host load + per-user / per-bot breakdown. Centralised here (instead of inline in routes.py) so: * The HTML ``/admin/stats`` page and the JSON ``/api/admin/stats`` endpoint read from the same source — guaranteed parity. * Tests can exercise the collector directly without spinning up a full FastAPI client + DB session for what is fundamentally a data-shape test. Layout of the returned dict --------------------------- :: { "host": {cpu_*, ram_*, disk_*, boot_time, uptime_sec, ok}, "totals": {bots, running, cpu_live, ram_live_mb, bots_by_status: {...}}, "users": [ {user, bots_total, bots_running, cpu_live_percent, ram_live_mb, cpu_alloc_cores, ram_alloc_mb, storage_used_mb, bots_by_status: {...}} ... ], "ts": float, # unix time at collection } Failure modes ------------- ``psutil`` is best-effort: if it's missing or the host denies the read, host fields become ``None`` and ``host.ok`` flips to ``False``. The rest of the dashboard still renders — a blank host card beats a 500 on the whole page. """ from __future__ import annotations import time from datetime import datetime from typing import Any from sqlalchemy import select, func from sqlalchemy.ext.asyncio import AsyncSession try: import psutil # type: ignore _HAS_PSUTIL = True except Exception: # noqa: BLE001 _HAS_PSUTIL = False # Prime psutil's CPU% baseline once at import so the very first # ``cpu_percent(interval=None)`` call returns a real number instead # of the always-zero first sample. Without this priming the # dashboard would show "0%" for one render after every restart. if _HAS_PSUTIL: try: psutil.cpu_percent(interval=None) except Exception: # noqa: BLE001 pass # --------------------------------------------------------------------------- # Host metrics # --------------------------------------------------------------------------- def _collect_host() -> dict[str, Any]: """Snapshot of the host the panel itself runs on.""" out: dict[str, Any] = { "ok": False, "cpu_percent": None, "cpu_count": None, "ram_used_mb": None, "ram_total_mb": None, "ram_percent": None, "disk_used_mb": None, "disk_total_mb": None, "disk_percent": None, "boot_time": None, "uptime_sec": None, "error": None, } if not _HAS_PSUTIL: out["error"] = "psutil unavailable" return out try: out["cpu_percent"] = float(psutil.cpu_percent(interval=0.15)) out["cpu_count"] = psutil.cpu_count() or 1 vm = psutil.virtual_memory() out["ram_total_mb"] = round(vm.total / (1024 * 1024), 1) out["ram_used_mb"] = round((vm.total - vm.available) / (1024 * 1024), 1) out["ram_percent"] = float(vm.percent) try: # The panel runs on Render / HF Spaces — data_dir is the # most relevant mount, not necessarily "/". from config import PANEL_CONFIG # type: ignore du = psutil.disk_usage(str(PANEL_CONFIG.data_dir)) except Exception: # noqa: BLE001 du = psutil.disk_usage("/") out["disk_total_mb"] = round(du.total / (1024 * 1024), 1) out["disk_used_mb"] = round(du.used / (1024 * 1024), 1) out["disk_percent"] = float(du.percent) boot = psutil.boot_time() out["boot_time"] = float(boot) out["uptime_sec"] = int(time.time() - boot) out["ok"] = True return out except Exception as exc: # noqa: BLE001 out["error"] = str(exc) return out # --------------------------------------------------------------------------- # Bot-population metrics # --------------------------------------------------------------------------- async def _bot_status_counts(db: AsyncSession) -> dict[str, int]: """Count BotInstance rows grouped by status. Returns all known statuses even when the count is 0, so the dashboard always renders the same row order.""" # Imported lazily so this module can be imported standalone # (e.g. by tests) without pulling the whole models graph. from models import BotInstance, BotStatus counts: dict[str, int] = {s.value: 0 for s in BotStatus} rows = (await db.execute( select(BotInstance.status, func.count(BotInstance.id)) .group_by(BotInstance.status) )).all() for status, count in rows: key = status.value if hasattr(status, "value") else str(status) counts[key] = int(count) return counts async def collect_admin_stats(db: AsyncSession) -> dict[str, Any]: """Build the full admin-stats snapshot. The HTML page (``/admin/stats``) and the JSON endpoint (``/api/admin/stats``) both call this — they only differ in how they render the result. """ from models import ( BotInstance, BotStatus, DeploymentMode, User, ) host = _collect_host() bots_by_status = await _bot_status_counts(db) # One query, group by owner + status, so the per-user rollup # is a single DB round-trip instead of N+1. rows = (await db.execute( select( BotInstance.owner_id, BotInstance.status, func.count(BotInstance.id), ).group_by(BotInstance.owner_id, BotInstance.status) )).all() # owner_id -> {status -> count} per_owner_status: dict[int, dict[str, int]] = {} for owner_id, status, count in rows: per_owner_status.setdefault(int(owner_id), {})[ status.value if hasattr(status, "value") else str(status) ] = int(count) users_result = await db.execute(select(User).order_by(User.username.asc())) users = list(users_result.scalars().all()) # Single query for all bots; per-user rows are O(1) lookups after. all_bots = list((await db.execute(select(BotInstance))).scalars().all()) bots_by_owner: dict[int, list[BotInstance]] = {} for b in all_bots: bots_by_owner.setdefault(b.owner_id, []).append(b) user_rows: list[dict[str, Any]] = [] totals = { "bots": 0, "running": 0, "cpu_live": 0.0, "ram_live_mb": 0.0, "bots_by_status": dict(bots_by_status), } for u in users: owned = bots_by_owner.get(u.id, []) running_bots = [b for b in owned if b.status == BotStatus.RUNNING] # Live process metrics only meaningful for MULTITENANT bots; # LEGACY_SPACE bots run on their own HF Space and aren't # visible to this process. cpu_live = 0.0 ram_live_mb = 0.0 for b in owned: if b.deployment_mode == DeploymentMode.MULTITENANT: try: import bot_runner # local import to avoid heavy cost on tests st = bot_runner.bot_status(b) if st.get("running"): cpu_live += float(st.get("cpu_percent", 0.0)) ram_live_mb += float(st.get("rss_mb", 0.0)) except Exception: # noqa: BLE001 # If bot_runner can't probe (process exited, no # permissions, …) we just don't add to the live # totals — better than crashing the whole page. pass ram_alloc_mb = sum((b.ram_mb or 0) for b in running_bots) cpu_alloc_cores = sum((b.cpu_cores or 0.0) for b in running_bots) storage_mb = sum((b.storage_used_mb or 0) for b in owned) row = { "user": u, "username": u.username, "bots_total": len(owned), "bots_running": len(running_bots), "bots_by_status": dict(per_owner_status.get(u.id, {})), "cpu_live_percent": round(cpu_live, 1), "ram_live_mb": round(ram_live_mb, 1), "cpu_alloc_cores": round(cpu_alloc_cores, 2), "ram_alloc_mb": ram_alloc_mb, "storage_used_mb": storage_mb, } user_rows.append(row) totals["bots"] += len(owned) totals["running"] += len(running_bots) totals["cpu_live"] += cpu_live totals["ram_live_mb"] += ram_live_mb return { "host": host, "totals": totals, "users": user_rows, "ts": time.time(), "ts_iso": datetime.utcnow().isoformat(timespec="seconds") + "Z", } __all__ = ["collect_admin_stats", "_collect_host", "_bot_status_counts"]