File size: 6,143 Bytes
9350a1f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 | # SPDX-License-Identifier: Apache-2.0
"""Device test of the warm-up: the first real call after ``from_pretrained`` is fast.
cd code && python -m pytest -s -q models/tests/test_api_warmup_device.py
One chip, opened by the API. Real inputs: crops and mirror images of the demo photo (each call
gets an image it has not seen). For each variant, call 1 (a new image) is compared with the
steady latency of the same image (median of 5 later calls, each one after a call of the
previous variant, as call 1 was): the ratio must be at most
``SP_FIRST_CALL_MAX`` (default 1.25, the target is 1.10; the margin is for a shared, loaded
host) or within 0.3 ms. Variants: the default ``warmup_variants`` (1600x900 RGB array, 1920x1080
RGB array, 480x640 plane, JPEG file, ``max_keypoints=-1``, ``return_descriptors=False``,
``nms_radius=0``) and two variants that ``model.warmup(...)`` adds later (``nms_radius=3``,
a 1024x768 image). Then the outputs of the warmed model are compared with a model made with
``warmup_variants="minimal"``: bit-identical.
A fresh process gives the most exact numbers (``bench_first_call.py``): pytest has already
imported Pillow and others before the model loads.
"""
from __future__ import annotations
import os
import statistics
import time
from pathlib import Path
import numpy as np
import pytest
import torch
from PIL import Image, ImageOps
from tt_superpoint import SuperPoint
CODE = Path(__file__).resolve().parents[2]
SAMPLE = CODE / "sample_data" / "house_in_field_1080p.jpg"
DEVICE_ID = int(os.environ.get("TT_DEVICE_ID", "0"))
MAX_RATIO = float(os.environ.get("SP_FIRST_CALL_MAX", "1.25"))
ABS_MS = 0.3
def _real_images(n):
"""n different real images (crops / mirror images of the demo photo), RGB PIL."""
with Image.open(SAMPLE) as im:
base = im.convert("RGB")
out = []
for i in range(n):
x0, y0 = 40 * (i % 5), 30 * (i // 5)
c = base.crop((x0, y0, x0 + 1400, y0 + 790))
out.append(ImageOps.mirror(c) if i % 2 else c)
return out
def _variants(tmp_path):
imgs = _real_images(8)
jpgs = []
for i, im in enumerate(imgs):
p = tmp_path / f"real_{i}.jpg"
im.resize((1600, 900), Image.BILINEAR).save(p, quality=92)
jpgs.append(str(p))
rgb = lambda w, h: [np.asarray(im.resize((w, h), Image.BILINEAR)) for im in imgs] # noqa: E731
plane = [np.ascontiguousarray(a[..., 0]) for a in rgb(640, 480)]
r1600 = rgb(1600, 900)
# name -> (inputs, call kwargs, warm-up added later with model.warmup or None)
return {
"rgb_1600x900": (r1600, {}, None),
"rgb_1920x1080": (rgb(1920, 1080), {}, None),
"plane_480x640": (plane, {}, None),
"jpeg_1600x900": (jpgs, {}, None),
"all_keypoints": (r1600, {"max_keypoints": -1}, None),
"no_descriptors": (r1600, {"return_descriptors": False}, None),
"host_nms_r0": (r1600, {"nms_radius": 0}, None),
"nms_r3_added": (r1600, {"nms_radius": 3}, {"nms_radius": 3}),
"size_1024x768_added": (rgb(1024, 768), {}, {"size": (1024, 768)}),
}
def _ms(f):
t0 = time.perf_counter()
out = f()
return (time.perf_counter() - t0) * 1e3, out
def test_first_call_fast_and_outputs_unchanged(tmp_path):
variants = _variants(tmp_path)
t0 = time.perf_counter()
model = SuperPoint.from_pretrained(device_id=DEVICE_ID)
t_fp = time.perf_counter() - t0
print(f"\nfrom_pretrained (default warmup_variants): {t_fp:.2f} s, {model.config['warmup_s']}")
rows, outs, fails, prev = [], {}, [], None
try:
for name, (inputs, kw, later) in variants.items():
if later is not None:
t0 = time.perf_counter()
spent = model.warmup(**later)
t_w = time.perf_counter() - t0
assert spent, (name, "warm-up did nothing")
assert model.warmup(**later) == {}, "warmup() must be idempotent"
print(f"model.warmup({later}): {t_w * 1e3:.0f} ms")
first, out = _ms(lambda: model(inputs[0], **kw))
again = []
for _ in range(5):
if prev is not None: # the same switch as before call 1 (resize tables, bucket guess)
model(prev[0], **prev[1])
again.append(_ms(lambda: model(inputs[0], **kw))[0])
steady = statistics.median(again)
prev = (inputs[2], kw) # the last call of this variant (below)
o2 = model(inputs[0], **kw)
assert torch.equal(out.keypoints, o2.keypoints) and torch.equal(out.scores, o2.scores), name
outs[name] = [out] + [model(x, **kw) for x in inputs[1:3]]
ratio = first / steady
rows.append((name, first, steady, ratio))
print(f"{name:20s} call 1 {first:7.2f} ms steady (same image) {steady:7.2f} ms ratio {ratio:.2f}")
if not (ratio <= MAX_RATIO or first - steady <= ABS_MS):
fails.append((name, round(first, 2), round(steady, 2)))
finally:
model.close()
assert not fails, f"first call slower than {MAX_RATIO}x the steady latency: {fails}"
# the warm-up does not change the outputs: same results with the minimal warm-up
with SuperPoint.from_pretrained(device_id=DEVICE_ID, warmup_variants="minimal") as ref:
for name, (inputs, kw, _) in variants.items():
for x, o in zip(inputs[:3], outs[name]):
r = ref(x, **kw)
assert torch.equal(r.keypoints, o.keypoints) and torch.equal(r.scores, o.scores), name
assert (r.descriptors is None) == (o.descriptors is None), name
if r.descriptors is not None:
assert torch.equal(r.descriptors, o.descriptors), name
print("outputs bit-identical to warmup_variants='minimal' for every variant")
@pytest.mark.parametrize("spec", ["minimal"])
def test_warmup_spec_reported(spec):
with SuperPoint.from_pretrained(device_id=DEVICE_ID, warmup_variants=spec) as model:
assert model.config["warmup_variants"]["nms_radii"] == (4,)
assert model.config["warmup_s"]["variants"] < 1.0
|