superpoint-p150 / code /models /tests /test_api_warmup_device.py
changh95's picture
Python API: warm-up so the first real call is fast (warmup_variants, model.warmup()), quiet logs, install extras
9350a1f verified
Raw History Blame Contribute Delete
6.14 kB
# SPDX-License-Identifier: Apache-2.0
"""Device test of the warm-up: the first real call after ``from_pretrained`` is fast.
cd code && python -m pytest -s -q models/tests/test_api_warmup_device.py
One chip, opened by the API. Real inputs: crops and mirror images of the demo photo (each call
gets an image it has not seen). For each variant, call 1 (a new image) is compared with the
steady latency of the same image (median of 5 later calls, each one after a call of the
previous variant, as call 1 was): the ratio must be at most
``SP_FIRST_CALL_MAX`` (default 1.25, the target is 1.10; the margin is for a shared, loaded
host) or within 0.3 ms. Variants: the default ``warmup_variants`` (1600x900 RGB array, 1920x1080
RGB array, 480x640 plane, JPEG file, ``max_keypoints=-1``, ``return_descriptors=False``,
``nms_radius=0``) and two variants that ``model.warmup(...)`` adds later (``nms_radius=3``,
a 1024x768 image). Then the outputs of the warmed model are compared with a model made with
``warmup_variants="minimal"``: bit-identical.
A fresh process gives the most exact numbers (``bench_first_call.py``): pytest has already
imported Pillow and others before the model loads.
"""
from __future__ import annotations
import os
import statistics
import time
from pathlib import Path
import numpy as np
import pytest
import torch
from PIL import Image, ImageOps
from tt_superpoint import SuperPoint
CODE = Path(__file__).resolve().parents[2]
SAMPLE = CODE / "sample_data" / "house_in_field_1080p.jpg"
DEVICE_ID = int(os.environ.get("TT_DEVICE_ID", "0"))
MAX_RATIO = float(os.environ.get("SP_FIRST_CALL_MAX", "1.25"))
ABS_MS = 0.3
def _real_images(n):
"""n different real images (crops / mirror images of the demo photo), RGB PIL."""
with Image.open(SAMPLE) as im:
base = im.convert("RGB")
out = []
for i in range(n):
x0, y0 = 40 * (i % 5), 30 * (i // 5)
c = base.crop((x0, y0, x0 + 1400, y0 + 790))
out.append(ImageOps.mirror(c) if i % 2 else c)
return out
def _variants(tmp_path):
imgs = _real_images(8)
jpgs = []
for i, im in enumerate(imgs):
p = tmp_path / f"real_{i}.jpg"
im.resize((1600, 900), Image.BILINEAR).save(p, quality=92)
jpgs.append(str(p))
rgb = lambda w, h: [np.asarray(im.resize((w, h), Image.BILINEAR)) for im in imgs] # noqa: E731
plane = [np.ascontiguousarray(a[..., 0]) for a in rgb(640, 480)]
r1600 = rgb(1600, 900)
# name -> (inputs, call kwargs, warm-up added later with model.warmup or None)
return {
"rgb_1600x900": (r1600, {}, None),
"rgb_1920x1080": (rgb(1920, 1080), {}, None),
"plane_480x640": (plane, {}, None),
"jpeg_1600x900": (jpgs, {}, None),
"all_keypoints": (r1600, {"max_keypoints": -1}, None),
"no_descriptors": (r1600, {"return_descriptors": False}, None),
"host_nms_r0": (r1600, {"nms_radius": 0}, None),
"nms_r3_added": (r1600, {"nms_radius": 3}, {"nms_radius": 3}),
"size_1024x768_added": (rgb(1024, 768), {}, {"size": (1024, 768)}),
}
def _ms(f):
t0 = time.perf_counter()
out = f()
return (time.perf_counter() - t0) * 1e3, out
def test_first_call_fast_and_outputs_unchanged(tmp_path):
variants = _variants(tmp_path)
t0 = time.perf_counter()
model = SuperPoint.from_pretrained(device_id=DEVICE_ID)
t_fp = time.perf_counter() - t0
print(f"\nfrom_pretrained (default warmup_variants): {t_fp:.2f} s, {model.config['warmup_s']}")
rows, outs, fails, prev = [], {}, [], None
try:
for name, (inputs, kw, later) in variants.items():
if later is not None:
t0 = time.perf_counter()
spent = model.warmup(**later)
t_w = time.perf_counter() - t0
assert spent, (name, "warm-up did nothing")
assert model.warmup(**later) == {}, "warmup() must be idempotent"
print(f"model.warmup({later}): {t_w * 1e3:.0f} ms")
first, out = _ms(lambda: model(inputs[0], **kw))
again = []
for _ in range(5):
if prev is not None: # the same switch as before call 1 (resize tables, bucket guess)
model(prev[0], **prev[1])
again.append(_ms(lambda: model(inputs[0], **kw))[0])
steady = statistics.median(again)
prev = (inputs[2], kw) # the last call of this variant (below)
o2 = model(inputs[0], **kw)
assert torch.equal(out.keypoints, o2.keypoints) and torch.equal(out.scores, o2.scores), name
outs[name] = [out] + [model(x, **kw) for x in inputs[1:3]]
ratio = first / steady
rows.append((name, first, steady, ratio))
print(f"{name:20s} call 1 {first:7.2f} ms steady (same image) {steady:7.2f} ms ratio {ratio:.2f}")
if not (ratio <= MAX_RATIO or first - steady <= ABS_MS):
fails.append((name, round(first, 2), round(steady, 2)))
finally:
model.close()
assert not fails, f"first call slower than {MAX_RATIO}x the steady latency: {fails}"
# the warm-up does not change the outputs: same results with the minimal warm-up
with SuperPoint.from_pretrained(device_id=DEVICE_ID, warmup_variants="minimal") as ref:
for name, (inputs, kw, _) in variants.items():
for x, o in zip(inputs[:3], outs[name]):
r = ref(x, **kw)
assert torch.equal(r.keypoints, o.keypoints) and torch.equal(r.scores, o.scores), name
assert (r.descriptors is None) == (o.descriptors is None), name
if r.descriptors is not None:
assert torch.equal(r.descriptors, o.descriptors), name
print("outputs bit-identical to warmup_variants='minimal' for every variant")
@pytest.mark.parametrize("spec", ["minimal"])
def test_warmup_spec_reported(spec):
with SuperPoint.from_pretrained(device_id=DEVICE_ID, warmup_variants=spec) as model:
assert model.config["warmup_variants"]["nms_radii"] == (4,)
assert model.config["warmup_s"]["variants"] < 1.0