superpoint-p150 / code /bench_served.py
changh95's picture
p150 ETH-dispatch compliance (2026-10-05): default ETH dispatch, 1 CQ, 12x10 in Python API and server; numbers re-measured
026da6c verified
Raw History Blame Contribute Delete
7.51 kB
# SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
# SPDX-License-Identifier: Apache-2.0
"""Served-like benchmark: calls the server's own ``predict()`` handler (no HTTP framing) on the
sample 1600x900 JPEG, after the server's own warm-up, and reports the medians / minima of the
response's ``timing_ms`` keys (preprocess = base64 + JPEG decode + resize; device_forward;
postprocess; total) -- the definitions GPU_COMPARISON.md's served-like loop uses.
Run from code/ after `source tools/chipenv.sh <chip> <venv>` (PYTHONPATH must start with code/):
SP_DISPATCH=auto|eth|worker (default auto = ETH, 1 CQ) SP_N_ITER=100 python bench_served.py
Optional request knobs: SP_REQ_NMS_RADIUS (4), SP_REQ_MAX_KP (1024), SP_REQ_THR (0.005).
"""
import base64
import hashlib
import json
import os
import statistics
import time
import torch
import ttnn
from models.reference.superpoint_reference import load_reference_model # noqa: E402
from models.server import app as A # noqa: E402
from models.tt.superpoint_ttnn import TtSuperPoint # noqa: E402
HERE = os.path.dirname(os.path.abspath(__file__))
def main():
n = int(os.environ.get("SP_N_ITER", "100"))
from models.tt.device_open import open_ttnn_device, resolve_dispatch
mode = resolve_dispatch(os.environ.get("SP_DISPATCH", "auto")) # auto: ETH + 1 CQ + 12x10
kw = dict(l1_small_size=A.L1_SMALL_SIZE, trace_region_size=32 << 20)
dev = open_ttnn_device(0, dispatch=mode, **kw)
g = dev.compute_with_storage_grid_size()
print(f"dispatch={mode} num_command_queues=1 compute_grid={g.x}x{g.y}")
close_device = ttnn.close_device
try:
tm = load_reference_model()
A._setup_logging()
torch.set_grad_enabled(False)
model = TtSuperPoint(tm, dev, input_height=A.INPUT_HEIGHT, input_width=A.INPUT_WIDTH, fused=True)
tt_in = model.allocate_input(batch_size=1)
A.STATE.update(
model=model, tt_in=tt_in, fused=True, cfg={"fused": True},
model_config={"border_removal_distance": int(tm.config.border_removal_distance)},
)
dummy = torch.zeros(1, 3, A.INPUT_HEIGHT, A.INPUT_WIDTH)
A._warmup_fused(model, tt_in, dummy, ttnn, dev)
A.STATE["ready"] = True
print("warmup_ms", A.STATE.get("warmup_ms"))
if model.device_resize and os.environ.get("SP_FIRST_USE"):
# first request at a source size without a precompiled resize variant (trace capture)
import io
import numpy as np
from PIL import Image
for wh in ((1000, 700), (1001, 701)):
buf = io.BytesIO()
Image.fromarray(np.random.default_rng(0).integers(0, 256, (wh[1], wh[0], 3), dtype=np.uint8)).save(buf, "PNG")
rq = A.PredictRequest(image=base64.b64encode(buf.getvalue()).decode())
t0 = time.perf_counter(); r1 = A.predict_dict(rq); t1 = time.perf_counter(); r2 = A.predict_dict(rq); t2 = time.perf_counter()
print(f"first use {wh}: device_forward {r1['timing_ms']['device_forward']} ms (second {r2['timing_ms']['device_forward']} ms)")
img = base64.b64encode(open(os.path.join(HERE, "sample_data/house_in_field_1080p.jpg"), "rb").read()).decode()
req = A.PredictRequest(
image=img,
nms_radius=int(os.environ.get("SP_REQ_NMS_RADIUS", "4")),
max_keypoints=int(os.environ.get("SP_REQ_MAX_KP", "1024")),
keypoint_threshold=float(os.environ.get("SP_REQ_THR", "0.005")),
)
for _ in range(10):
r = A.predict_dict(req)
# Exactness vs the previous request path (RGB resize + fp32 /255 -> run_fused -> host
# post-processing from the NMS / descriptor maps), same decoded frame.
im = A._decode_image(img)
from models.tt import postprocess as _post
res = model.run_fused(tt_in, A._preprocess(im))
if req.nms_radius == model.nms_radius_traced:
a = _post.postprocess_from_nms_map(res.nms_map, res.descriptors_nchw, keypoint_threshold=req.keypoint_threshold,
max_keypoints=req.max_keypoints, border_removal_distance=4, with_descriptors=True)[0]
else: # host fold + simple_nms(r) on the traced scores (the previous behaviour for r != 4)
res = model.run_fused(tt_in, A._preprocess(im), nms_radius=req.nms_radius)
a = _post.postprocess_keypoints(res.scores_nchw, res.descriptors_nchw, nms_radius=req.nms_radius,
keypoint_threshold=req.keypoint_threshold, max_keypoints=req.max_keypoints,
border_removal_distance=4, with_descriptors=True)[0]
b = A._infer_fused(model, tt_in, A._preprocess_r8(im), max_keypoints=req.max_keypoints,
keypoint_threshold=req.keypoint_threshold, nms_radius=req.nms_radius,
return_descriptors=True, border=4)
oa, ob = torch.argsort(a[1], descending=True, stable=True), torch.argsort(b[1], descending=True, stable=True)
same = a[0].shape == b[0].shape and torch.equal(a[0][oa], b[0][ob]) and torch.equal(a[1][oa], b[1][ob])
dd = float((a[2][oa] - b[2][ob]).abs().max()) if same and a[0].shape[0] else float("nan")
print(f"host-resize request path vs RGB/run_fused path: kp/scores identical={same} n={b[0].shape[0]} desc max|diff|={dd:.2e}")
if model.device_resize:
# rsz stage: device resize of the full-size R plane vs the host Pillow resize (round-1 server)
c = A._infer_fused(model, tt_in, A._r_plane(im), max_keypoints=req.max_keypoints,
keypoint_threshold=req.keypoint_threshold, nms_radius=req.nms_radius,
return_descriptors=True, border=4)
same_c = all(torch.equal(x, y) for x, y in zip(b[:3], c[:3]))
print(f"device-resize request path vs host-resize request path: kp/scores/desc bit-identical={same_c}")
r_host = A.predict_dict(req)
model.fused_stages = model.fused_stages - {"rsz"}
r_old = A.predict_dict(req)
model.fused_stages = model.fused_stages | {"rsz"}
keys = ("num_keypoints", "keypoints", "scores")
same_r = all(r_host[k] == r_old[k] for k in keys) and r_host["descriptors"]["data"] == r_old["descriptors"]["data"]
print(f"predict() response (device resize) == predict() response (host resize): {same_r}")
keys = ("preprocess", "device_forward", "postprocess", "total")
T = {k: [] for k in keys + ("handler_wall",)}
for _ in range(n):
t0 = time.perf_counter()
r = A.predict_dict(req)
T["handler_wall"].append((time.perf_counter() - t0) * 1e3)
for k in keys:
T[k].append(r["timing_ms"][k])
print(f"mode={mode} n={n} num_keypoints={r['num_keypoints']} serving_path={r.get('serving_path')} "
f"top3={r['keypoints'][:3]} scores={r['scores'][:3]}")
body = {k: r[k] for k in ("num_keypoints", "keypoints", "scores", "descriptors")}
print("response_sha256", hashlib.sha256(json.dumps(body, sort_keys=True).encode()).hexdigest())
for k, v in T.items():
print(f"{k:16s} median={statistics.median(v):8.3f} ms min={min(v):8.3f} ms")
model.release()
finally:
close_device(dev)
if __name__ == "__main__":
main()