Download code/bench_served.py from changh95/superpoint-p150: direct link, hf CLI and curl.
- Browser
- Download file 7.51 kB
-
https://huggingface.co/changh95/superpoint-p150/resolve/main/code/bench_served.py
- Command line
-
hf download hf://changh95/superpoint-p150/code/bench_served.py
-
curl -L -o bench_served.py https://huggingface.co/changh95/superpoint-p150/resolve/main/code/bench_served.py
7.51 kB
| # SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. | |
| # SPDX-License-Identifier: Apache-2.0 | |
| """Served-like benchmark: calls the server's own ``predict()`` handler (no HTTP framing) on the | |
| sample 1600x900 JPEG, after the server's own warm-up, and reports the medians / minima of the | |
| response's ``timing_ms`` keys (preprocess = base64 + JPEG decode + resize; device_forward; | |
| postprocess; total) -- the definitions GPU_COMPARISON.md's served-like loop uses. | |
| Run from code/ after `source tools/chipenv.sh <chip> <venv>` (PYTHONPATH must start with code/): | |
| SP_DISPATCH=auto|eth|worker (default auto = ETH, 1 CQ) SP_N_ITER=100 python bench_served.py | |
| Optional request knobs: SP_REQ_NMS_RADIUS (4), SP_REQ_MAX_KP (1024), SP_REQ_THR (0.005). | |
| """ | |
| import base64 | |
| import hashlib | |
| import json | |
| import os | |
| import statistics | |
| import time | |
| import torch | |
| import ttnn | |
| from models.reference.superpoint_reference import load_reference_model # noqa: E402 | |
| from models.server import app as A # noqa: E402 | |
| from models.tt.superpoint_ttnn import TtSuperPoint # noqa: E402 | |
| HERE = os.path.dirname(os.path.abspath(__file__)) | |
| def main(): | |
| n = int(os.environ.get("SP_N_ITER", "100")) | |
| from models.tt.device_open import open_ttnn_device, resolve_dispatch | |
| mode = resolve_dispatch(os.environ.get("SP_DISPATCH", "auto")) # auto: ETH + 1 CQ + 12x10 | |
| kw = dict(l1_small_size=A.L1_SMALL_SIZE, trace_region_size=32 << 20) | |
| dev = open_ttnn_device(0, dispatch=mode, **kw) | |
| g = dev.compute_with_storage_grid_size() | |
| print(f"dispatch={mode} num_command_queues=1 compute_grid={g.x}x{g.y}") | |
| close_device = ttnn.close_device | |
| try: | |
| tm = load_reference_model() | |
| A._setup_logging() | |
| torch.set_grad_enabled(False) | |
| model = TtSuperPoint(tm, dev, input_height=A.INPUT_HEIGHT, input_width=A.INPUT_WIDTH, fused=True) | |
| tt_in = model.allocate_input(batch_size=1) | |
| A.STATE.update( | |
| model=model, tt_in=tt_in, fused=True, cfg={"fused": True}, | |
| model_config={"border_removal_distance": int(tm.config.border_removal_distance)}, | |
| ) | |
| dummy = torch.zeros(1, 3, A.INPUT_HEIGHT, A.INPUT_WIDTH) | |
| A._warmup_fused(model, tt_in, dummy, ttnn, dev) | |
| A.STATE["ready"] = True | |
| print("warmup_ms", A.STATE.get("warmup_ms")) | |
| if model.device_resize and os.environ.get("SP_FIRST_USE"): | |
| # first request at a source size without a precompiled resize variant (trace capture) | |
| import io | |
| import numpy as np | |
| from PIL import Image | |
| for wh in ((1000, 700), (1001, 701)): | |
| buf = io.BytesIO() | |
| Image.fromarray(np.random.default_rng(0).integers(0, 256, (wh[1], wh[0], 3), dtype=np.uint8)).save(buf, "PNG") | |
| rq = A.PredictRequest(image=base64.b64encode(buf.getvalue()).decode()) | |
| t0 = time.perf_counter(); r1 = A.predict_dict(rq); t1 = time.perf_counter(); r2 = A.predict_dict(rq); t2 = time.perf_counter() | |
| print(f"first use {wh}: device_forward {r1['timing_ms']['device_forward']} ms (second {r2['timing_ms']['device_forward']} ms)") | |
| img = base64.b64encode(open(os.path.join(HERE, "sample_data/house_in_field_1080p.jpg"), "rb").read()).decode() | |
| req = A.PredictRequest( | |
| image=img, | |
| nms_radius=int(os.environ.get("SP_REQ_NMS_RADIUS", "4")), | |
| max_keypoints=int(os.environ.get("SP_REQ_MAX_KP", "1024")), | |
| keypoint_threshold=float(os.environ.get("SP_REQ_THR", "0.005")), | |
| ) | |
| for _ in range(10): | |
| r = A.predict_dict(req) | |
| # Exactness vs the previous request path (RGB resize + fp32 /255 -> run_fused -> host | |
| # post-processing from the NMS / descriptor maps), same decoded frame. | |
| im = A._decode_image(img) | |
| from models.tt import postprocess as _post | |
| res = model.run_fused(tt_in, A._preprocess(im)) | |
| if req.nms_radius == model.nms_radius_traced: | |
| a = _post.postprocess_from_nms_map(res.nms_map, res.descriptors_nchw, keypoint_threshold=req.keypoint_threshold, | |
| max_keypoints=req.max_keypoints, border_removal_distance=4, with_descriptors=True)[0] | |
| else: # host fold + simple_nms(r) on the traced scores (the previous behaviour for r != 4) | |
| res = model.run_fused(tt_in, A._preprocess(im), nms_radius=req.nms_radius) | |
| a = _post.postprocess_keypoints(res.scores_nchw, res.descriptors_nchw, nms_radius=req.nms_radius, | |
| keypoint_threshold=req.keypoint_threshold, max_keypoints=req.max_keypoints, | |
| border_removal_distance=4, with_descriptors=True)[0] | |
| b = A._infer_fused(model, tt_in, A._preprocess_r8(im), max_keypoints=req.max_keypoints, | |
| keypoint_threshold=req.keypoint_threshold, nms_radius=req.nms_radius, | |
| return_descriptors=True, border=4) | |
| oa, ob = torch.argsort(a[1], descending=True, stable=True), torch.argsort(b[1], descending=True, stable=True) | |
| same = a[0].shape == b[0].shape and torch.equal(a[0][oa], b[0][ob]) and torch.equal(a[1][oa], b[1][ob]) | |
| dd = float((a[2][oa] - b[2][ob]).abs().max()) if same and a[0].shape[0] else float("nan") | |
| print(f"host-resize request path vs RGB/run_fused path: kp/scores identical={same} n={b[0].shape[0]} desc max|diff|={dd:.2e}") | |
| if model.device_resize: | |
| # rsz stage: device resize of the full-size R plane vs the host Pillow resize (round-1 server) | |
| c = A._infer_fused(model, tt_in, A._r_plane(im), max_keypoints=req.max_keypoints, | |
| keypoint_threshold=req.keypoint_threshold, nms_radius=req.nms_radius, | |
| return_descriptors=True, border=4) | |
| same_c = all(torch.equal(x, y) for x, y in zip(b[:3], c[:3])) | |
| print(f"device-resize request path vs host-resize request path: kp/scores/desc bit-identical={same_c}") | |
| r_host = A.predict_dict(req) | |
| model.fused_stages = model.fused_stages - {"rsz"} | |
| r_old = A.predict_dict(req) | |
| model.fused_stages = model.fused_stages | {"rsz"} | |
| keys = ("num_keypoints", "keypoints", "scores") | |
| same_r = all(r_host[k] == r_old[k] for k in keys) and r_host["descriptors"]["data"] == r_old["descriptors"]["data"] | |
| print(f"predict() response (device resize) == predict() response (host resize): {same_r}") | |
| keys = ("preprocess", "device_forward", "postprocess", "total") | |
| T = {k: [] for k in keys + ("handler_wall",)} | |
| for _ in range(n): | |
| t0 = time.perf_counter() | |
| r = A.predict_dict(req) | |
| T["handler_wall"].append((time.perf_counter() - t0) * 1e3) | |
| for k in keys: | |
| T[k].append(r["timing_ms"][k]) | |
| print(f"mode={mode} n={n} num_keypoints={r['num_keypoints']} serving_path={r.get('serving_path')} " | |
| f"top3={r['keypoints'][:3]} scores={r['scores'][:3]}") | |
| body = {k: r[k] for k in ("num_keypoints", "keypoints", "scores", "descriptors")} | |
| print("response_sha256", hashlib.sha256(json.dumps(body, sort_keys=True).encode()).hexdigest()) | |
| for k, v in T.items(): | |
| print(f"{k:16s} median={statistics.median(v):8.3f} ms min={min(v):8.3f} ms") | |
| model.release() | |
| finally: | |
| close_device(dev) | |
| if __name__ == "__main__": | |
| main() | |