Spaces:
Running on Zero
Running on Zero
AndrianBalanescu commited on
Commit ·
3ef32ca
1
Parent(s): de2772a
fix(telemetry): gpu_tracked_call must open lease before GPU body
Browse filesCaught live: /v1/gpu/status showed calls_total=0 after successful chat
calls because gpu_tracked_call never called lease_start, so lease_end
found no active call and dropped the record. Adds offline regression
tests for the success and error paths with module-state isolation.
- app.py +3 -5
- tests/unit/test_gpu_usage_tracker.py +41 -1
app.py
CHANGED
|
@@ -403,6 +403,7 @@ def gpu_tracked_call(kind: str, fn, *args, model: str = "", **kwargs):
|
|
| 403 |
is not observable; the recorded queue_wait_seconds is an upper bound.
|
| 404 |
"""
|
| 405 |
t_submit = time.time()
|
|
|
|
| 406 |
|
| 407 |
@spaces.GPU(duration=120)
|
| 408 |
def _tracked_inner():
|
|
@@ -418,14 +419,11 @@ def gpu_tracked_call(kind: str, fn, *args, model: str = "", **kwargs):
|
|
| 418 |
result, error, t_body_start = _tracked_inner()
|
| 419 |
except Exception as exc:
|
| 420 |
# Lease-level failure (e.g. 429 quota exhaustion): body never ran.
|
| 421 |
-
GPU_USAGE.lease_end(
|
| 422 |
-
f"{kind}-{int(t_submit * 1000)}", model=model, kind=kind,
|
| 423 |
-
error=str(exc),
|
| 424 |
-
)
|
| 425 |
raise
|
| 426 |
queue_wait = max(0.0, t_body_start - t_submit)
|
| 427 |
GPU_USAGE.lease_end(
|
| 428 |
-
|
| 429 |
queue_wait_s=queue_wait, error=error,
|
| 430 |
)
|
| 431 |
if error:
|
|
|
|
| 403 |
is not observable; the recorded queue_wait_seconds is an upper bound.
|
| 404 |
"""
|
| 405 |
t_submit = time.time()
|
| 406 |
+
call_id = GPU_USAGE.lease_start(kind)
|
| 407 |
|
| 408 |
@spaces.GPU(duration=120)
|
| 409 |
def _tracked_inner():
|
|
|
|
| 419 |
result, error, t_body_start = _tracked_inner()
|
| 420 |
except Exception as exc:
|
| 421 |
# Lease-level failure (e.g. 429 quota exhaustion): body never ran.
|
| 422 |
+
GPU_USAGE.lease_end(call_id, model=model, kind=kind, error=str(exc))
|
|
|
|
|
|
|
|
|
|
| 423 |
raise
|
| 424 |
queue_wait = max(0.0, t_body_start - t_submit)
|
| 425 |
GPU_USAGE.lease_end(
|
| 426 |
+
call_id, model=model, kind=kind,
|
| 427 |
queue_wait_s=queue_wait, error=error,
|
| 428 |
)
|
| 429 |
if error:
|
tests/unit/test_gpu_usage_tracker.py
CHANGED
|
@@ -116,4 +116,44 @@ def test_usage_snapshot_shape(tracker):
|
|
| 116 |
snap: Dict[str, Any] = tracker.usage_snapshot()
|
| 117 |
assert set(snap.keys()) == {"daily", "profile"}
|
| 118 |
assert set(snap["daily"].keys()) >= {"date", "gpu_seconds_used", "quota_consumed_percent"}
|
| 119 |
-
assert set(snap["profile"].keys()) >= {"total_calls", "avg_queue_wait_seconds"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 116 |
snap: Dict[str, Any] = tracker.usage_snapshot()
|
| 117 |
assert set(snap.keys()) == {"daily", "profile"}
|
| 118 |
assert set(snap["daily"].keys()) >= {"date", "gpu_seconds_used", "quota_consumed_percent"}
|
| 119 |
+
assert set(snap["profile"].keys()) >= {"total_calls", "avg_queue_wait_seconds"}
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
@pytest.fixture()
|
| 123 |
+
def fresh_usage():
|
| 124 |
+
app = _load_app()
|
| 125 |
+
app.GPU_USAGE = app.GPUUsageTracker()
|
| 126 |
+
yield app
|
| 127 |
+
app.GPU_USAGE = app.GPUUsageTracker() # restore module-level state
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
def test_gpu_tracked_call_records_usage(fresh_usage):
|
| 131 |
+
"""Regression: gpu_tracked_call must open a lease BEFORE invoking the
|
| 132 |
+
GPU body, otherwise lease_end finds no active call and usage stays 0
|
| 133 |
+
(caught live: successful chat calls were not recorded)."""
|
| 134 |
+
app = fresh_usage
|
| 135 |
+
result = app.gpu_tracked_call(
|
| 136 |
+
"chat",
|
| 137 |
+
lambda *a, **k: {"choices": [{"message": {"content": "ok"}}]},
|
| 138 |
+
[],
|
| 139 |
+
model="test.gguf",
|
| 140 |
+
)
|
| 141 |
+
assert result["choices"][0]["message"]["content"] == "ok"
|
| 142 |
+
snap = app.GPU_USAGE.usage_snapshot()
|
| 143 |
+
assert snap["daily"]["calls_total"] == 1
|
| 144 |
+
assert snap["daily"]["calls_ok"] == 1
|
| 145 |
+
assert snap["profile"]["total_calls"] == 1
|
| 146 |
+
|
| 147 |
+
|
| 148 |
+
def test_gpu_tracked_call_records_error_path(fresh_usage):
|
| 149 |
+
app = fresh_usage
|
| 150 |
+
with pytest.raises(RuntimeError):
|
| 151 |
+
app.gpu_tracked_call(
|
| 152 |
+
"chat",
|
| 153 |
+
lambda *a, **k: (_ for _ in ()).throw(RuntimeError("boom")),
|
| 154 |
+
[],
|
| 155 |
+
model="test.gguf",
|
| 156 |
+
)
|
| 157 |
+
snap = app.GPU_USAGE.usage_snapshot()
|
| 158 |
+
assert snap["daily"]["calls_total"] == 1
|
| 159 |
+
assert snap["daily"]["calls_failed"] == 1
|