Spaces:
Running on Zero
Running on Zero
File size: 12,916 Bytes
de2772a a29d413 de2772a 3ef32ca 54cef29 11732b0 3ef32ca db08952 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 | """Offline unit tests for GPUUsageTracker ZeroGPU quota telemetry.
Pure logic tests with an injected fake clock; no network, no ZeroGPU
dependency, no writes outside pytest tmp_path.
"""
from __future__ import annotations
import importlib.util
import os
import sys
from typing import Any, Dict, List
import pytest
_REPO = os.path.join(os.path.dirname(__file__), "..", "..")
def _load_app():
"""Import app.py with mocked heavy deps (gradio, spaces, llama_cpp)."""
sys.path.insert(0, _REPO)
import app # noqa: F401 (mock fallbacks make this importable offline)
return app
@pytest.fixture()
def tracker():
app = _load_app()
clock = {"t": 1_000_000.0}
def fake_clock():
return clock["t"]
t = app.GPUUsageTracker(clock=fake_clock)
t._advance = lambda s: clock.__setitem__("t", clock["t"] + s)
return t
def test_lease_roundtrip_records_wall_seconds(tracker):
call_id = tracker.lease_start("chat")
tracker._advance(12.5)
record = tracker.lease_end(call_id, model="Qwen3.8-9B-Q8_0.gguf", kind="chat")
assert record["lease_seconds"] == 12.5
assert record["model"] == "Qwen3.8-9B-Q8_0.gguf"
assert record["error"] == ""
def test_lease_end_unknown_id_returns_empty(tracker):
assert tracker.lease_end("missing-id") == {}
def test_daily_usage_rolls_up_same_day(tracker):
for lease in (10.0, 20.0, 5.0):
cid = tracker.lease_start("chat")
tracker._advance(lease)
tracker.lease_end(cid, kind="chat")
usage = tracker.daily_usage()
assert usage["calls_total"] == 3
assert usage["calls_ok"] == 3
assert usage["gpu_seconds_used"] == 35.0
assert usage["gpu_minutes_used"] == 0.58 # rounded to 2 decimals by tracker
assert usage["daily_quota_minutes"] == 40
assert usage["quota_consumed_percent"] == 1.5 # rounded to 1 decimal by tracker
assert usage["quota_exhausted_observed"] is False
def test_quota_error_marks_exhaustion_state(tracker):
cid = tracker.lease_start("chat")
tracker._advance(3.0)
tracker.lease_end(cid, kind="chat", error="Space app has reached its GPU limit")
usage = tracker.daily_usage()
assert usage["calls_failed"] == 1
assert usage["quota_exhausted_observed"] is True
assert "GPU limit" in usage["last_quota_error"]["detail"]
def test_stale_quota_error_flag_expires_after_ttl(tracker):
"""A quota error must not stick for the whole UTC day (rolling-24h HF
window can free up in the meantime). After the TTL the flag expires and a
real generation attempt arbitrates."""
cid = tracker.lease_start("chat")
tracker._advance(3.0)
tracker.lease_end(cid, kind="chat", error="exceeded your ZeroGPU quota")
assert tracker.daily_usage()["quota_exhausted_observed"] is True
# still fresh just before TTL
tracker._advance(tracker.QUOTA_FLAG_TTL_SECONDS - 10)
assert tracker.daily_usage()["quota_exhausted_observed"] is True
# past TTL: stale flag must be ignored
tracker._advance(11)
usage = tracker.daily_usage()
assert usage["quota_exhausted_observed"] is False
# raw error is still retained for diagnostics
assert usage["last_quota_error"]["detail"].endswith("ZeroGPU quota")
def test_ttl_env_override_respected():
app = _load_app()
import subprocess # noqa: F401
# constant must read env override at import time
assert isinstance(app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS, int)
assert app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS > 0
def test_daily_usage_ignores_previous_days(tracker):
cid = tracker.lease_start("chat")
tracker._advance(60.0)
tracker.lease_end(cid, kind="chat")
# advance a full day: 24h + a bit
tracker._advance(24 * 3600 + 5)
cid2 = tracker.lease_start("chat")
tracker._advance(10.0)
tracker.lease_end(cid2, kind="chat")
usage = tracker.daily_usage()
assert usage["calls_total"] == 1
assert usage["gpu_seconds_used"] == 10.0
def test_summary_throughput_and_queue(tracker):
# chat call: 600 completion tokens over 30s lease, 2s queue wait
cid = tracker.lease_start("chat")
tracker._advance(30.0)
rec = tracker.lease_end(cid, kind="chat", queue_wait_s=2.0)
rec["completion_tokens"] = 600
profile = tracker.summary()
assert profile["total_calls"] == 1
assert profile["avg_queue_wait_seconds"] == 2.0
assert profile["chat_throughput_tokens_per_gpu_second"] == pytest.approx(20.0)
def test_summary_ignores_failed_calls_for_throughput(tracker):
cid = tracker.lease_start("chat")
tracker._advance(5.0)
tracker.lease_end(cid, kind="chat", error="boom")
profile = tracker.summary()
assert profile["total_calls"] == 1
assert profile["chat_throughput_tokens_per_gpu_second"] == 0.0
def test_usage_snapshot_shape(tracker):
cid = tracker.lease_start("chat")
tracker._advance(1.0)
tracker.lease_end(cid, kind="chat")
snap: Dict[str, Any] = tracker.usage_snapshot()
assert set(snap.keys()) == {"daily", "profile"}
assert set(snap["daily"].keys()) >= {"date", "gpu_seconds_used", "quota_consumed_percent"}
assert set(snap["profile"].keys()) >= {"total_calls", "avg_queue_wait_seconds"}
@pytest.fixture()
def fresh_usage():
app = _load_app()
app.GPU_USAGE = app.GPUUsageTracker()
yield app
app.GPU_USAGE = app.GPUUsageTracker() # restore module-level state
def test_gpu_tracked_call_records_usage(fresh_usage):
"""Regression: gpu_tracked_call must open a lease BEFORE invoking the
GPU body, otherwise lease_end finds no active call and usage stays 0
(caught live: successful chat calls were not recorded)."""
app = fresh_usage
result = app.gpu_tracked_call(
"chat",
lambda *a, **k: {"choices": [{"message": {"content": "ok"}}]},
[],
model="test.gguf",
)
assert result["choices"][0]["message"]["content"] == "ok"
snap = app.GPU_USAGE.usage_snapshot()
assert snap["daily"]["calls_total"] == 1
assert snap["daily"]["calls_ok"] == 1
assert snap["profile"]["total_calls"] == 1
def test_gpu_tracked_call_cold_start_worker_side(fresh_usage):
"""ZeroGPU runs @spaces.GPU bodies in a worker process: get_model sets
_llm/_loaded_file there and those globals never reach the web process.
Cold-start detection must therefore run inside the body (returns DO
propagate back). Regression for the live finding where cold_start was
always 0.0 (the -1.0 sentinel was silently dropped by summary())."""
app = fresh_usage
app._llm = None
app._loaded_file = None
def loader(msgs, model_file, *a, **k):
# mimic get_model side effect inside the worker
app._llm = object()
app._loaded_file = model_file
return {"choices": [{"message": {"content": "ok"}}]}
model = "Qwen3.8-9B-Q8_0.gguf"
app.gpu_tracked_call("chat", loader, [], model, 0.7, 30, model=model)
cold_rec = app.GPU_USAGE._calls[-1]
assert cold_rec["cold_start_seconds"] >= 0.0
assert cold_rec["lease_seconds"] >= cold_rec["cold_start_seconds"]
def warm(msgs, model_file, *a, **k):
return {"choices": [{"message": {"content": "ok"}}]}
app.gpu_tracked_call("chat", warm, [], model, 0.7, 30, model=model)
warm_rec = app.GPU_USAGE._calls[-1]
assert warm_rec["cold_start_seconds"] == 0.0
def test_gpu_tracked_call_completion_tokens(fresh_usage):
"""Usage.completion_tokens from the chat result feed the throughput
metric (tokens per GPU-second); absent/None must degrade to 0."""
app = fresh_usage
res = {"choices": [{"message": {"content": "ok"}}],
"usage": {"completion_tokens": 42}}
app.gpu_tracked_call("chat", lambda *a, **k: res, [], "m.gguf", model="m.gguf")
rec = app.GPU_USAGE._calls[-1]
assert rec["completion_tokens"] == 42
prof = app.GPU_USAGE.summary()
assert prof["chat_throughput_tokens_per_gpu_second"] > 0
app.gpu_tracked_call("chat", lambda *a, **k: {"choices": []}, [], "m.gguf", model="m.gguf")
rec2 = app.GPU_USAGE._calls[-1]
assert rec2["completion_tokens"] == 0
def test_quota_error_marks_exhaustion(tracker):
cid = tracker.lease_start("chat")
tracker.lease_end(cid, kind="chat", error="ZeroGPU quota exceeded: limit reached")
daily = tracker.daily_usage()
assert daily["calls_failed"] == 1
assert daily["quota_exhausted_observed"] is True
assert daily["last_quota_error"] is not None
assert "limit" in daily["last_quota_error"]["detail"]
def test_gpu_tracked_call_records_error_path(fresh_usage):
app = fresh_usage
with pytest.raises(RuntimeError):
app.gpu_tracked_call(
"chat",
lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom")),
[],
model="test.gguf",
)
snap = app.GPU_USAGE.usage_snapshot()
assert snap["daily"]["calls_total"] == 1
assert snap["daily"]["calls_failed"] == 1
# ββ Restart-honesty regression: the ledger must survive a process bounce ββββββ
# Before this, the counter lived only in memory, so every Space reload reported a
# full 40 minutes while Hugging Face was already refusing leases ("31/40 left"
# on screen next to a 429 "Space app has reached its GPU limit").
def test_ledger_persists_usage_across_restart(tmp_path):
app = _load_app()
ledger = str(tmp_path / "zerogpu_quota_ledger.json")
clock = {"t": 2_000_000.0}
def fake_clock():
return clock["t"]
first = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
cid = first.lease_start("chat")
clock["t"] += 540.0 # 9 minutes of billed GPU time
first.lease_end(cid, kind="chat", model="Qwen3.8-9B-Q8_0.gguf")
assert first.remaining_seconds() == 40 * 60 - 540.0
# Simulate the Space restarting: a brand-new tracker over the same ledger.
second = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
assert second.remaining_seconds() == 40 * 60 - 540.0, (
"a restart must not restore phantom GPU minutes"
)
assert second.used_seconds_today() == 540.0
def test_ledger_keeps_quota_error_and_expires_at_window_end(tmp_path):
app = _load_app()
ledger = str(tmp_path / "zerogpu_quota_ledger.json")
clock = {"t": 3_000_000.0}
def fake_clock():
return clock["t"]
t = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
cid = t.lease_start("chat")
clock["t"] += 4.0
t.lease_end(cid, kind="chat", error="Space app has reached its GPU limit")
assert t.daily_usage()["quota_exhausted_observed"] is True
bounced = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
assert bounced.daily_usage()["quota_exhausted_observed"] is True, (
"a restart must not hide an observed platform refusal"
)
# Past the documented 24h window the allowance is genuinely fresh again.
clock["t"] += app.GPUUsageTracker.WINDOW_SECONDS + 1
reset = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
assert reset.remaining_seconds() == 40 * 60
assert reset.daily_usage()["quota_exhausted_observed"] is False
def test_daily_usage_reports_window_reset_time(tmp_path):
app = _load_app()
clock = {"t": 4_000_000.0}
def fake_clock():
return clock["t"]
t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json"))
assert t.daily_usage()["quota_window_resets_at"] is None
t.lease_start("chat")
usage = t.daily_usage()
assert usage["quota_window_resets_at"] == 4_000_000 + app.GPUUsageTracker.WINDOW_SECONDS
assert usage["quota_window_seconds_remaining"] == app.GPUUsageTracker.WINDOW_SECONDS
assert usage["ledger_persisted"] is True
def test_observed_refusal_reports_zero_remaining_minutes(tmp_path):
"""A platform refusal must not coexist with a positive minutes-left readout.
This is the exact reported bug: the UI showed ~31/40 minutes while every
request came back 429 "Space app has reached its GPU limit".
"""
app = _load_app()
clock = {"t": 5_000_000.0}
def fake_clock():
return clock["t"]
t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json"))
cid = t.lease_start("chat")
clock["t"] += 60.0
t.lease_end(cid, kind="chat")
usage = t.daily_usage()
assert usage["quota_remaining_seconds"] == 39 * 60
assert usage["quota_remaining_minutes"] == 39.0
assert usage["quota_headroom_known"] is True
refused = t.lease_start("chat")
clock["t"] += 0.2
t.lease_end(
refused,
kind="chat",
error="Space app has reached its GPU limit. Try re-running outside of examples",
)
usage = t.daily_usage()
assert usage["quota_exhausted_observed"] is True
assert usage["quota_remaining_seconds"] == 0.0
assert usage["quota_remaining_minutes"] == 0.0
assert usage["quota_headroom_known"] is False
|