Spaces:
Running on Zero
Running on Zero
Download tests/unit/test_gpu_usage_tracker.py from abalanescu/flow2: direct link, hf CLI and curl.
- Browser
- Download file 12.9 kB
-
https://huggingface.co/spaces/abalanescu/flow2/resolve/main/tests/unit/test_gpu_usage_tracker.py
- Command line
-
hf download hf://spaces/abalanescu/flow2/tests/unit/test_gpu_usage_tracker.py
-
curl -L -o test_gpu_usage_tracker.py https://huggingface.co/spaces/abalanescu/flow2/resolve/main/tests/unit/test_gpu_usage_tracker.py
12.9 kB
| """Offline unit tests for GPUUsageTracker ZeroGPU quota telemetry. | |
| Pure logic tests with an injected fake clock; no network, no ZeroGPU | |
| dependency, no writes outside pytest tmp_path. | |
| """ | |
| from __future__ import annotations | |
| import importlib.util | |
| import os | |
| import sys | |
| from typing import Any, Dict, List | |
| import pytest | |
| _REPO = os.path.join(os.path.dirname(__file__), "..", "..") | |
| def _load_app(): | |
| """Import app.py with mocked heavy deps (gradio, spaces, llama_cpp).""" | |
| sys.path.insert(0, _REPO) | |
| import app # noqa: F401 (mock fallbacks make this importable offline) | |
| return app | |
| def tracker(): | |
| app = _load_app() | |
| clock = {"t": 1_000_000.0} | |
| def fake_clock(): | |
| return clock["t"] | |
| t = app.GPUUsageTracker(clock=fake_clock) | |
| t._advance = lambda s: clock.__setitem__("t", clock["t"] + s) | |
| return t | |
| def test_lease_roundtrip_records_wall_seconds(tracker): | |
| call_id = tracker.lease_start("chat") | |
| tracker._advance(12.5) | |
| record = tracker.lease_end(call_id, model="Qwen3.8-9B-Q8_0.gguf", kind="chat") | |
| assert record["lease_seconds"] == 12.5 | |
| assert record["model"] == "Qwen3.8-9B-Q8_0.gguf" | |
| assert record["error"] == "" | |
| def test_lease_end_unknown_id_returns_empty(tracker): | |
| assert tracker.lease_end("missing-id") == {} | |
| def test_daily_usage_rolls_up_same_day(tracker): | |
| for lease in (10.0, 20.0, 5.0): | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(lease) | |
| tracker.lease_end(cid, kind="chat") | |
| usage = tracker.daily_usage() | |
| assert usage["calls_total"] == 3 | |
| assert usage["calls_ok"] == 3 | |
| assert usage["gpu_seconds_used"] == 35.0 | |
| assert usage["gpu_minutes_used"] == 0.58 # rounded to 2 decimals by tracker | |
| assert usage["daily_quota_minutes"] == 40 | |
| assert usage["quota_consumed_percent"] == 1.5 # rounded to 1 decimal by tracker | |
| assert usage["quota_exhausted_observed"] is False | |
| def test_quota_error_marks_exhaustion_state(tracker): | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(3.0) | |
| tracker.lease_end(cid, kind="chat", error="Space app has reached its GPU limit") | |
| usage = tracker.daily_usage() | |
| assert usage["calls_failed"] == 1 | |
| assert usage["quota_exhausted_observed"] is True | |
| assert "GPU limit" in usage["last_quota_error"]["detail"] | |
| def test_stale_quota_error_flag_expires_after_ttl(tracker): | |
| """A quota error must not stick for the whole UTC day (rolling-24h HF | |
| window can free up in the meantime). After the TTL the flag expires and a | |
| real generation attempt arbitrates.""" | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(3.0) | |
| tracker.lease_end(cid, kind="chat", error="exceeded your ZeroGPU quota") | |
| assert tracker.daily_usage()["quota_exhausted_observed"] is True | |
| # still fresh just before TTL | |
| tracker._advance(tracker.QUOTA_FLAG_TTL_SECONDS - 10) | |
| assert tracker.daily_usage()["quota_exhausted_observed"] is True | |
| # past TTL: stale flag must be ignored | |
| tracker._advance(11) | |
| usage = tracker.daily_usage() | |
| assert usage["quota_exhausted_observed"] is False | |
| # raw error is still retained for diagnostics | |
| assert usage["last_quota_error"]["detail"].endswith("ZeroGPU quota") | |
| def test_ttl_env_override_respected(): | |
| app = _load_app() | |
| import subprocess # noqa: F401 | |
| # constant must read env override at import time | |
| assert isinstance(app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS, int) | |
| assert app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS > 0 | |
| def test_daily_usage_ignores_previous_days(tracker): | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(60.0) | |
| tracker.lease_end(cid, kind="chat") | |
| # advance a full day: 24h + a bit | |
| tracker._advance(24 * 3600 + 5) | |
| cid2 = tracker.lease_start("chat") | |
| tracker._advance(10.0) | |
| tracker.lease_end(cid2, kind="chat") | |
| usage = tracker.daily_usage() | |
| assert usage["calls_total"] == 1 | |
| assert usage["gpu_seconds_used"] == 10.0 | |
| def test_summary_throughput_and_queue(tracker): | |
| # chat call: 600 completion tokens over 30s lease, 2s queue wait | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(30.0) | |
| rec = tracker.lease_end(cid, kind="chat", queue_wait_s=2.0) | |
| rec["completion_tokens"] = 600 | |
| profile = tracker.summary() | |
| assert profile["total_calls"] == 1 | |
| assert profile["avg_queue_wait_seconds"] == 2.0 | |
| assert profile["chat_throughput_tokens_per_gpu_second"] == pytest.approx(20.0) | |
| def test_summary_ignores_failed_calls_for_throughput(tracker): | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(5.0) | |
| tracker.lease_end(cid, kind="chat", error="boom") | |
| profile = tracker.summary() | |
| assert profile["total_calls"] == 1 | |
| assert profile["chat_throughput_tokens_per_gpu_second"] == 0.0 | |
| def test_usage_snapshot_shape(tracker): | |
| cid = tracker.lease_start("chat") | |
| tracker._advance(1.0) | |
| tracker.lease_end(cid, kind="chat") | |
| snap: Dict[str, Any] = tracker.usage_snapshot() | |
| assert set(snap.keys()) == {"daily", "profile"} | |
| assert set(snap["daily"].keys()) >= {"date", "gpu_seconds_used", "quota_consumed_percent"} | |
| assert set(snap["profile"].keys()) >= {"total_calls", "avg_queue_wait_seconds"} | |
| def fresh_usage(): | |
| app = _load_app() | |
| app.GPU_USAGE = app.GPUUsageTracker() | |
| yield app | |
| app.GPU_USAGE = app.GPUUsageTracker() # restore module-level state | |
| def test_gpu_tracked_call_records_usage(fresh_usage): | |
| """Regression: gpu_tracked_call must open a lease BEFORE invoking the | |
| GPU body, otherwise lease_end finds no active call and usage stays 0 | |
| (caught live: successful chat calls were not recorded).""" | |
| app = fresh_usage | |
| result = app.gpu_tracked_call( | |
| "chat", | |
| lambda *a, **k: {"choices": [{"message": {"content": "ok"}}]}, | |
| [], | |
| model="test.gguf", | |
| ) | |
| assert result["choices"][0]["message"]["content"] == "ok" | |
| snap = app.GPU_USAGE.usage_snapshot() | |
| assert snap["daily"]["calls_total"] == 1 | |
| assert snap["daily"]["calls_ok"] == 1 | |
| assert snap["profile"]["total_calls"] == 1 | |
| def test_gpu_tracked_call_cold_start_worker_side(fresh_usage): | |
| """ZeroGPU runs @spaces.GPU bodies in a worker process: get_model sets | |
| _llm/_loaded_file there and those globals never reach the web process. | |
| Cold-start detection must therefore run inside the body (returns DO | |
| propagate back). Regression for the live finding where cold_start was | |
| always 0.0 (the -1.0 sentinel was silently dropped by summary()).""" | |
| app = fresh_usage | |
| app._llm = None | |
| app._loaded_file = None | |
| def loader(msgs, model_file, *a, **k): | |
| # mimic get_model side effect inside the worker | |
| app._llm = object() | |
| app._loaded_file = model_file | |
| return {"choices": [{"message": {"content": "ok"}}]} | |
| model = "Qwen3.8-9B-Q8_0.gguf" | |
| app.gpu_tracked_call("chat", loader, [], model, 0.7, 30, model=model) | |
| cold_rec = app.GPU_USAGE._calls[-1] | |
| assert cold_rec["cold_start_seconds"] >= 0.0 | |
| assert cold_rec["lease_seconds"] >= cold_rec["cold_start_seconds"] | |
| def warm(msgs, model_file, *a, **k): | |
| return {"choices": [{"message": {"content": "ok"}}]} | |
| app.gpu_tracked_call("chat", warm, [], model, 0.7, 30, model=model) | |
| warm_rec = app.GPU_USAGE._calls[-1] | |
| assert warm_rec["cold_start_seconds"] == 0.0 | |
| def test_gpu_tracked_call_completion_tokens(fresh_usage): | |
| """Usage.completion_tokens from the chat result feed the throughput | |
| metric (tokens per GPU-second); absent/None must degrade to 0.""" | |
| app = fresh_usage | |
| res = {"choices": [{"message": {"content": "ok"}}], | |
| "usage": {"completion_tokens": 42}} | |
| app.gpu_tracked_call("chat", lambda *a, **k: res, [], "m.gguf", model="m.gguf") | |
| rec = app.GPU_USAGE._calls[-1] | |
| assert rec["completion_tokens"] == 42 | |
| prof = app.GPU_USAGE.summary() | |
| assert prof["chat_throughput_tokens_per_gpu_second"] > 0 | |
| app.gpu_tracked_call("chat", lambda *a, **k: {"choices": []}, [], "m.gguf", model="m.gguf") | |
| rec2 = app.GPU_USAGE._calls[-1] | |
| assert rec2["completion_tokens"] == 0 | |
| def test_quota_error_marks_exhaustion(tracker): | |
| cid = tracker.lease_start("chat") | |
| tracker.lease_end(cid, kind="chat", error="ZeroGPU quota exceeded: limit reached") | |
| daily = tracker.daily_usage() | |
| assert daily["calls_failed"] == 1 | |
| assert daily["quota_exhausted_observed"] is True | |
| assert daily["last_quota_error"] is not None | |
| assert "limit" in daily["last_quota_error"]["detail"] | |
| def test_gpu_tracked_call_records_error_path(fresh_usage): | |
| app = fresh_usage | |
| with pytest.raises(RuntimeError): | |
| app.gpu_tracked_call( | |
| "chat", | |
| lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom")), | |
| [], | |
| model="test.gguf", | |
| ) | |
| snap = app.GPU_USAGE.usage_snapshot() | |
| assert snap["daily"]["calls_total"] == 1 | |
| assert snap["daily"]["calls_failed"] == 1 | |
| # ββ Restart-honesty regression: the ledger must survive a process bounce ββββββ | |
| # Before this, the counter lived only in memory, so every Space reload reported a | |
| # full 40 minutes while Hugging Face was already refusing leases ("31/40 left" | |
| # on screen next to a 429 "Space app has reached its GPU limit"). | |
| def test_ledger_persists_usage_across_restart(tmp_path): | |
| app = _load_app() | |
| ledger = str(tmp_path / "zerogpu_quota_ledger.json") | |
| clock = {"t": 2_000_000.0} | |
| def fake_clock(): | |
| return clock["t"] | |
| first = app.GPUUsageTracker(clock=fake_clock, state_path=ledger) | |
| cid = first.lease_start("chat") | |
| clock["t"] += 540.0 # 9 minutes of billed GPU time | |
| first.lease_end(cid, kind="chat", model="Qwen3.8-9B-Q8_0.gguf") | |
| assert first.remaining_seconds() == 40 * 60 - 540.0 | |
| # Simulate the Space restarting: a brand-new tracker over the same ledger. | |
| second = app.GPUUsageTracker(clock=fake_clock, state_path=ledger) | |
| assert second.remaining_seconds() == 40 * 60 - 540.0, ( | |
| "a restart must not restore phantom GPU minutes" | |
| ) | |
| assert second.used_seconds_today() == 540.0 | |
| def test_ledger_keeps_quota_error_and_expires_at_window_end(tmp_path): | |
| app = _load_app() | |
| ledger = str(tmp_path / "zerogpu_quota_ledger.json") | |
| clock = {"t": 3_000_000.0} | |
| def fake_clock(): | |
| return clock["t"] | |
| t = app.GPUUsageTracker(clock=fake_clock, state_path=ledger) | |
| cid = t.lease_start("chat") | |
| clock["t"] += 4.0 | |
| t.lease_end(cid, kind="chat", error="Space app has reached its GPU limit") | |
| assert t.daily_usage()["quota_exhausted_observed"] is True | |
| bounced = app.GPUUsageTracker(clock=fake_clock, state_path=ledger) | |
| assert bounced.daily_usage()["quota_exhausted_observed"] is True, ( | |
| "a restart must not hide an observed platform refusal" | |
| ) | |
| # Past the documented 24h window the allowance is genuinely fresh again. | |
| clock["t"] += app.GPUUsageTracker.WINDOW_SECONDS + 1 | |
| reset = app.GPUUsageTracker(clock=fake_clock, state_path=ledger) | |
| assert reset.remaining_seconds() == 40 * 60 | |
| assert reset.daily_usage()["quota_exhausted_observed"] is False | |
| def test_daily_usage_reports_window_reset_time(tmp_path): | |
| app = _load_app() | |
| clock = {"t": 4_000_000.0} | |
| def fake_clock(): | |
| return clock["t"] | |
| t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json")) | |
| assert t.daily_usage()["quota_window_resets_at"] is None | |
| t.lease_start("chat") | |
| usage = t.daily_usage() | |
| assert usage["quota_window_resets_at"] == 4_000_000 + app.GPUUsageTracker.WINDOW_SECONDS | |
| assert usage["quota_window_seconds_remaining"] == app.GPUUsageTracker.WINDOW_SECONDS | |
| assert usage["ledger_persisted"] is True | |
| def test_observed_refusal_reports_zero_remaining_minutes(tmp_path): | |
| """A platform refusal must not coexist with a positive minutes-left readout. | |
| This is the exact reported bug: the UI showed ~31/40 minutes while every | |
| request came back 429 "Space app has reached its GPU limit". | |
| """ | |
| app = _load_app() | |
| clock = {"t": 5_000_000.0} | |
| def fake_clock(): | |
| return clock["t"] | |
| t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json")) | |
| cid = t.lease_start("chat") | |
| clock["t"] += 60.0 | |
| t.lease_end(cid, kind="chat") | |
| usage = t.daily_usage() | |
| assert usage["quota_remaining_seconds"] == 39 * 60 | |
| assert usage["quota_remaining_minutes"] == 39.0 | |
| assert usage["quota_headroom_known"] is True | |
| refused = t.lease_start("chat") | |
| clock["t"] += 0.2 | |
| t.lease_end( | |
| refused, | |
| kind="chat", | |
| error="Space app has reached its GPU limit. Try re-running outside of examples", | |
| ) | |
| usage = t.daily_usage() | |
| assert usage["quota_exhausted_observed"] is True | |
| assert usage["quota_remaining_seconds"] == 0.0 | |
| assert usage["quota_remaining_minutes"] == 0.0 | |
| assert usage["quota_headroom_known"] is False | |