File size: 2,412 Bytes
8c4be59
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
"""
Replicates TurboVision's public-model latency gate locally:
p95 per-frame inference latency must be <= 100 ms in a 2 vCPU environment.

Run pinned to 2 CPUs to match the compliance environment:

    taskset -c 0,1 python benchmark_latency.py [--frames 100] [--imgsz 640]

Frames are synthesized at 1280x720 (the validator resizes the long side to
1280 before calling miners), or loaded from --image-dir if given.
"""

import argparse
import os
import time
from pathlib import Path

os.environ.setdefault("OMP_NUM_THREADS", "2")
# The compliance environment has no GPU; hide any local one so torch runs on CPU.
os.environ["CUDA_VISIBLE_DEVICES"] = ""

import numpy as np


def load_frames(image_dir: str | None, n: int) -> list[np.ndarray]:
    if image_dir:
        import cv2

        paths = sorted(Path(image_dir).glob("*"))[:n]
        frames = [cv2.imread(str(p)) for p in paths]
        frames = [f for f in frames if f is not None]
        if frames:
            return frames
    rng = np.random.default_rng(0)
    return [
        rng.integers(0, 255, size=(720, 1280, 3), dtype=np.uint8) for _ in range(n)
    ]


def main() -> None:
    parser = argparse.ArgumentParser()
    parser.add_argument("--frames", type=int, default=100)
    parser.add_argument("--imgsz", type=int, default=640)
    parser.add_argument("--image-dir", default=None)
    parser.add_argument("--warmup", type=int, default=5)
    args = parser.parse_args()

    import torch

    torch.set_num_threads(2)

    os.environ["SV_IMG_SIZE"] = str(args.imgsz)
    from miner import Miner

    miner = Miner(path_hf_repo=Path(__file__).parent)
    frames = load_frames(args.image_dir, args.frames)

    for frame in frames[: args.warmup]:
        miner.predict_batch([frame], offset=0, n_keypoints=0)

    latencies_ms = []
    for frame in frames:
        start = time.perf_counter()
        miner.predict_batch([frame], offset=0, n_keypoints=0)
        latencies_ms.append((time.perf_counter() - start) * 1000)

    latencies_ms.sort()
    p50 = latencies_ms[len(latencies_ms) // 2]
    p95 = latencies_ms[int(len(latencies_ms) * 0.95)]
    print(f"model={miner.model_name} imgsz={args.imgsz} frames={len(frames)}")
    print(f"p50={p50:.1f} ms  p95={p95:.1f} ms  mean={sum(latencies_ms)/len(latencies_ms):.1f} ms")
    print("PASS (<=100 ms p95)" if p95 <= 100 else "FAIL (>100 ms p95)")


if __name__ == "__main__":
    main()