File size: 12,916 Bytes
de2772a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a29d413
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
de2772a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3ef32ca
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
54cef29
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11732b0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3ef32ca
 
 
 
 
 
 
 
 
 
 
db08952
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
"""Offline unit tests for GPUUsageTracker ZeroGPU quota telemetry.

Pure logic tests with an injected fake clock; no network, no ZeroGPU
dependency, no writes outside pytest tmp_path.
"""

from __future__ import annotations

import importlib.util
import os
import sys
from typing import Any, Dict, List

import pytest

_REPO = os.path.join(os.path.dirname(__file__), "..", "..")


def _load_app():
    """Import app.py with mocked heavy deps (gradio, spaces, llama_cpp)."""
    sys.path.insert(0, _REPO)
    import app  # noqa: F401  (mock fallbacks make this importable offline)
    return app


@pytest.fixture()
def tracker():
    app = _load_app()
    clock = {"t": 1_000_000.0}

    def fake_clock():
        return clock["t"]

    t = app.GPUUsageTracker(clock=fake_clock)
    t._advance = lambda s: clock.__setitem__("t", clock["t"] + s)
    return t


def test_lease_roundtrip_records_wall_seconds(tracker):
    call_id = tracker.lease_start("chat")
    tracker._advance(12.5)
    record = tracker.lease_end(call_id, model="Qwen3.8-9B-Q8_0.gguf", kind="chat")
    assert record["lease_seconds"] == 12.5
    assert record["model"] == "Qwen3.8-9B-Q8_0.gguf"
    assert record["error"] == ""


def test_lease_end_unknown_id_returns_empty(tracker):
    assert tracker.lease_end("missing-id") == {}


def test_daily_usage_rolls_up_same_day(tracker):
    for lease in (10.0, 20.0, 5.0):
        cid = tracker.lease_start("chat")
        tracker._advance(lease)
        tracker.lease_end(cid, kind="chat")
    usage = tracker.daily_usage()
    assert usage["calls_total"] == 3
    assert usage["calls_ok"] == 3
    assert usage["gpu_seconds_used"] == 35.0
    assert usage["gpu_minutes_used"] == 0.58  # rounded to 2 decimals by tracker
    assert usage["daily_quota_minutes"] == 40
    assert usage["quota_consumed_percent"] == 1.5  # rounded to 1 decimal by tracker
    assert usage["quota_exhausted_observed"] is False


def test_quota_error_marks_exhaustion_state(tracker):
    cid = tracker.lease_start("chat")
    tracker._advance(3.0)
    tracker.lease_end(cid, kind="chat", error="Space app has reached its GPU limit")
    usage = tracker.daily_usage()
    assert usage["calls_failed"] == 1
    assert usage["quota_exhausted_observed"] is True
    assert "GPU limit" in usage["last_quota_error"]["detail"]


def test_stale_quota_error_flag_expires_after_ttl(tracker):
    """A quota error must not stick for the whole UTC day (rolling-24h HF
    window can free up in the meantime). After the TTL the flag expires and a
    real generation attempt arbitrates."""
    cid = tracker.lease_start("chat")
    tracker._advance(3.0)
    tracker.lease_end(cid, kind="chat", error="exceeded your ZeroGPU quota")
    assert tracker.daily_usage()["quota_exhausted_observed"] is True

    # still fresh just before TTL
    tracker._advance(tracker.QUOTA_FLAG_TTL_SECONDS - 10)
    assert tracker.daily_usage()["quota_exhausted_observed"] is True

    # past TTL: stale flag must be ignored
    tracker._advance(11)
    usage = tracker.daily_usage()
    assert usage["quota_exhausted_observed"] is False
    # raw error is still retained for diagnostics
    assert usage["last_quota_error"]["detail"].endswith("ZeroGPU quota")


def test_ttl_env_override_respected():
    app = _load_app()
    import subprocess  # noqa: F401
    # constant must read env override at import time
    assert isinstance(app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS, int)
    assert app.GPUUsageTracker.QUOTA_FLAG_TTL_SECONDS > 0


def test_daily_usage_ignores_previous_days(tracker):
    cid = tracker.lease_start("chat")
    tracker._advance(60.0)
    tracker.lease_end(cid, kind="chat")
    # advance a full day: 24h + a bit
    tracker._advance(24 * 3600 + 5)
    cid2 = tracker.lease_start("chat")
    tracker._advance(10.0)
    tracker.lease_end(cid2, kind="chat")
    usage = tracker.daily_usage()
    assert usage["calls_total"] == 1
    assert usage["gpu_seconds_used"] == 10.0


def test_summary_throughput_and_queue(tracker):
    # chat call: 600 completion tokens over 30s lease, 2s queue wait
    cid = tracker.lease_start("chat")
    tracker._advance(30.0)
    rec = tracker.lease_end(cid, kind="chat", queue_wait_s=2.0)
    rec["completion_tokens"] = 600
    profile = tracker.summary()
    assert profile["total_calls"] == 1
    assert profile["avg_queue_wait_seconds"] == 2.0
    assert profile["chat_throughput_tokens_per_gpu_second"] == pytest.approx(20.0)


def test_summary_ignores_failed_calls_for_throughput(tracker):
    cid = tracker.lease_start("chat")
    tracker._advance(5.0)
    tracker.lease_end(cid, kind="chat", error="boom")
    profile = tracker.summary()
    assert profile["total_calls"] == 1
    assert profile["chat_throughput_tokens_per_gpu_second"] == 0.0


def test_usage_snapshot_shape(tracker):
    cid = tracker.lease_start("chat")
    tracker._advance(1.0)
    tracker.lease_end(cid, kind="chat")
    snap: Dict[str, Any] = tracker.usage_snapshot()
    assert set(snap.keys()) == {"daily", "profile"}
    assert set(snap["daily"].keys()) >= {"date", "gpu_seconds_used", "quota_consumed_percent"}
    assert set(snap["profile"].keys()) >= {"total_calls", "avg_queue_wait_seconds"}


@pytest.fixture()
def fresh_usage():
    app = _load_app()
    app.GPU_USAGE = app.GPUUsageTracker()
    yield app
    app.GPU_USAGE = app.GPUUsageTracker()  # restore module-level state


def test_gpu_tracked_call_records_usage(fresh_usage):
    """Regression: gpu_tracked_call must open a lease BEFORE invoking the
    GPU body, otherwise lease_end finds no active call and usage stays 0
    (caught live: successful chat calls were not recorded)."""
    app = fresh_usage
    result = app.gpu_tracked_call(
        "chat",
        lambda *a, **k: {"choices": [{"message": {"content": "ok"}}]},
        [],
        model="test.gguf",
    )
    assert result["choices"][0]["message"]["content"] == "ok"
    snap = app.GPU_USAGE.usage_snapshot()
    assert snap["daily"]["calls_total"] == 1
    assert snap["daily"]["calls_ok"] == 1
    assert snap["profile"]["total_calls"] == 1


def test_gpu_tracked_call_cold_start_worker_side(fresh_usage):
    """ZeroGPU runs @spaces.GPU bodies in a worker process: get_model sets
    _llm/_loaded_file there and those globals never reach the web process.
    Cold-start detection must therefore run inside the body (returns DO
    propagate back). Regression for the live finding where cold_start was
    always 0.0 (the -1.0 sentinel was silently dropped by summary())."""
    app = fresh_usage
    app._llm = None
    app._loaded_file = None

    def loader(msgs, model_file, *a, **k):
        # mimic get_model side effect inside the worker
        app._llm = object()
        app._loaded_file = model_file
        return {"choices": [{"message": {"content": "ok"}}]}

    model = "Qwen3.8-9B-Q8_0.gguf"
    app.gpu_tracked_call("chat", loader, [], model, 0.7, 30, model=model)
    cold_rec = app.GPU_USAGE._calls[-1]
    assert cold_rec["cold_start_seconds"] >= 0.0
    assert cold_rec["lease_seconds"] >= cold_rec["cold_start_seconds"]

    def warm(msgs, model_file, *a, **k):
        return {"choices": [{"message": {"content": "ok"}}]}

    app.gpu_tracked_call("chat", warm, [], model, 0.7, 30, model=model)
    warm_rec = app.GPU_USAGE._calls[-1]
    assert warm_rec["cold_start_seconds"] == 0.0


def test_gpu_tracked_call_completion_tokens(fresh_usage):
    """Usage.completion_tokens from the chat result feed the throughput
    metric (tokens per GPU-second); absent/None must degrade to 0."""
    app = fresh_usage
    res = {"choices": [{"message": {"content": "ok"}}],
           "usage": {"completion_tokens": 42}}
    app.gpu_tracked_call("chat", lambda *a, **k: res, [], "m.gguf", model="m.gguf")
    rec = app.GPU_USAGE._calls[-1]
    assert rec["completion_tokens"] == 42
    prof = app.GPU_USAGE.summary()
    assert prof["chat_throughput_tokens_per_gpu_second"] > 0

    app.gpu_tracked_call("chat", lambda *a, **k: {"choices": []}, [], "m.gguf", model="m.gguf")
    rec2 = app.GPU_USAGE._calls[-1]
    assert rec2["completion_tokens"] == 0


def test_quota_error_marks_exhaustion(tracker):
    cid = tracker.lease_start("chat")
    tracker.lease_end(cid, kind="chat", error="ZeroGPU quota exceeded: limit reached")
    daily = tracker.daily_usage()
    assert daily["calls_failed"] == 1
    assert daily["quota_exhausted_observed"] is True
    assert daily["last_quota_error"] is not None
    assert "limit" in daily["last_quota_error"]["detail"]


def test_gpu_tracked_call_records_error_path(fresh_usage):
    app = fresh_usage
    with pytest.raises(RuntimeError):
        app.gpu_tracked_call(
            "chat",
            lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom")),
            [],
            model="test.gguf",
        )
    snap = app.GPU_USAGE.usage_snapshot()
    assert snap["daily"]["calls_total"] == 1
    assert snap["daily"]["calls_failed"] == 1

# ── Restart-honesty regression: the ledger must survive a process bounce ──────
# Before this, the counter lived only in memory, so every Space reload reported a
# full 40 minutes while Hugging Face was already refusing leases ("31/40 left"
# on screen next to a 429 "Space app has reached its GPU limit").

def test_ledger_persists_usage_across_restart(tmp_path):
    app = _load_app()
    ledger = str(tmp_path / "zerogpu_quota_ledger.json")
    clock = {"t": 2_000_000.0}

    def fake_clock():
        return clock["t"]

    first = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
    cid = first.lease_start("chat")
    clock["t"] += 540.0  # 9 minutes of billed GPU time
    first.lease_end(cid, kind="chat", model="Qwen3.8-9B-Q8_0.gguf")
    assert first.remaining_seconds() == 40 * 60 - 540.0

    # Simulate the Space restarting: a brand-new tracker over the same ledger.
    second = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
    assert second.remaining_seconds() == 40 * 60 - 540.0, (
        "a restart must not restore phantom GPU minutes"
    )
    assert second.used_seconds_today() == 540.0


def test_ledger_keeps_quota_error_and_expires_at_window_end(tmp_path):
    app = _load_app()
    ledger = str(tmp_path / "zerogpu_quota_ledger.json")
    clock = {"t": 3_000_000.0}

    def fake_clock():
        return clock["t"]

    t = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
    cid = t.lease_start("chat")
    clock["t"] += 4.0
    t.lease_end(cid, kind="chat", error="Space app has reached its GPU limit")
    assert t.daily_usage()["quota_exhausted_observed"] is True

    bounced = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
    assert bounced.daily_usage()["quota_exhausted_observed"] is True, (
        "a restart must not hide an observed platform refusal"
    )

    # Past the documented 24h window the allowance is genuinely fresh again.
    clock["t"] += app.GPUUsageTracker.WINDOW_SECONDS + 1
    reset = app.GPUUsageTracker(clock=fake_clock, state_path=ledger)
    assert reset.remaining_seconds() == 40 * 60
    assert reset.daily_usage()["quota_exhausted_observed"] is False


def test_daily_usage_reports_window_reset_time(tmp_path):
    app = _load_app()
    clock = {"t": 4_000_000.0}

    def fake_clock():
        return clock["t"]

    t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json"))
    assert t.daily_usage()["quota_window_resets_at"] is None
    t.lease_start("chat")
    usage = t.daily_usage()
    assert usage["quota_window_resets_at"] == 4_000_000 + app.GPUUsageTracker.WINDOW_SECONDS
    assert usage["quota_window_seconds_remaining"] == app.GPUUsageTracker.WINDOW_SECONDS
    assert usage["ledger_persisted"] is True


def test_observed_refusal_reports_zero_remaining_minutes(tmp_path):
    """A platform refusal must not coexist with a positive minutes-left readout.

    This is the exact reported bug: the UI showed ~31/40 minutes while every
    request came back 429 "Space app has reached its GPU limit".
    """
    app = _load_app()
    clock = {"t": 5_000_000.0}

    def fake_clock():
        return clock["t"]

    t = app.GPUUsageTracker(clock=fake_clock, state_path=str(tmp_path / "l.json"))
    cid = t.lease_start("chat")
    clock["t"] += 60.0
    t.lease_end(cid, kind="chat")
    usage = t.daily_usage()
    assert usage["quota_remaining_seconds"] == 39 * 60
    assert usage["quota_remaining_minutes"] == 39.0
    assert usage["quota_headroom_known"] is True

    refused = t.lease_start("chat")
    clock["t"] += 0.2
    t.lease_end(
        refused,
        kind="chat",
        error="Space app has reached its GPU limit. Try re-running outside of examples",
    )
    usage = t.daily_usage()
    assert usage["quota_exhausted_observed"] is True
    assert usage["quota_remaining_seconds"] == 0.0
    assert usage["quota_remaining_minutes"] == 0.0
    assert usage["quota_headroom_known"] is False