scorevision: push artifact
Browse files- benchmark_latency.py +76 -0
benchmark_latency.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
Replicates TurboVision's public-model latency gate locally:
|
| 3 |
+
p95 per-frame inference latency must be <= 100 ms in a 2 vCPU environment.
|
| 4 |
+
|
| 5 |
+
Run pinned to 2 CPUs to match the compliance environment:
|
| 6 |
+
|
| 7 |
+
taskset -c 0,1 python benchmark_latency.py [--frames 100] [--imgsz 640]
|
| 8 |
+
|
| 9 |
+
Frames are synthesized at 1280x720 (the validator resizes the long side to
|
| 10 |
+
1280 before calling miners), or loaded from --image-dir if given.
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
import argparse
|
| 14 |
+
import os
|
| 15 |
+
import time
|
| 16 |
+
from pathlib import Path
|
| 17 |
+
|
| 18 |
+
os.environ.setdefault("OMP_NUM_THREADS", "2")
|
| 19 |
+
# The compliance environment has no GPU; hide any local one so torch runs on CPU.
|
| 20 |
+
os.environ["CUDA_VISIBLE_DEVICES"] = ""
|
| 21 |
+
|
| 22 |
+
import numpy as np
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def load_frames(image_dir: str | None, n: int) -> list[np.ndarray]:
|
| 26 |
+
if image_dir:
|
| 27 |
+
import cv2
|
| 28 |
+
|
| 29 |
+
paths = sorted(Path(image_dir).glob("*"))[:n]
|
| 30 |
+
frames = [cv2.imread(str(p)) for p in paths]
|
| 31 |
+
frames = [f for f in frames if f is not None]
|
| 32 |
+
if frames:
|
| 33 |
+
return frames
|
| 34 |
+
rng = np.random.default_rng(0)
|
| 35 |
+
return [
|
| 36 |
+
rng.integers(0, 255, size=(720, 1280, 3), dtype=np.uint8) for _ in range(n)
|
| 37 |
+
]
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def main() -> None:
|
| 41 |
+
parser = argparse.ArgumentParser()
|
| 42 |
+
parser.add_argument("--frames", type=int, default=100)
|
| 43 |
+
parser.add_argument("--imgsz", type=int, default=640)
|
| 44 |
+
parser.add_argument("--image-dir", default=None)
|
| 45 |
+
parser.add_argument("--warmup", type=int, default=5)
|
| 46 |
+
args = parser.parse_args()
|
| 47 |
+
|
| 48 |
+
import torch
|
| 49 |
+
|
| 50 |
+
torch.set_num_threads(2)
|
| 51 |
+
|
| 52 |
+
os.environ["SV_IMG_SIZE"] = str(args.imgsz)
|
| 53 |
+
from miner import Miner
|
| 54 |
+
|
| 55 |
+
miner = Miner(path_hf_repo=Path(__file__).parent)
|
| 56 |
+
frames = load_frames(args.image_dir, args.frames)
|
| 57 |
+
|
| 58 |
+
for frame in frames[: args.warmup]:
|
| 59 |
+
miner.predict_batch([frame], offset=0, n_keypoints=0)
|
| 60 |
+
|
| 61 |
+
latencies_ms = []
|
| 62 |
+
for frame in frames:
|
| 63 |
+
start = time.perf_counter()
|
| 64 |
+
miner.predict_batch([frame], offset=0, n_keypoints=0)
|
| 65 |
+
latencies_ms.append((time.perf_counter() - start) * 1000)
|
| 66 |
+
|
| 67 |
+
latencies_ms.sort()
|
| 68 |
+
p50 = latencies_ms[len(latencies_ms) // 2]
|
| 69 |
+
p95 = latencies_ms[int(len(latencies_ms) * 0.95)]
|
| 70 |
+
print(f"model={miner.model_name} imgsz={args.imgsz} frames={len(frames)}")
|
| 71 |
+
print(f"p50={p50:.1f} ms p95={p95:.1f} ms mean={sum(latencies_ms)/len(latencies_ms):.1f} ms")
|
| 72 |
+
print("PASS (<=100 ms p95)" if p95 <= 100 else "FAIL (>100 ms p95)")
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
if __name__ == "__main__":
|
| 76 |
+
main()
|