# SPDX-License-Identifier: Apache-2.0 """C08 I/O: input decoding and output encoding shared by the Python API and the HTTP server of every bundle. One module, two callers: the bundle's ``api`` (Python objects: paths, bytes, numpy arrays) and its ``server.app`` (JSON envelopes with base64 payloads). Both go through the same decoders, so ``model(...)`` and ``POST /predict`` see bit-identical inputs (BUNDLE_CONVENTIONS.md section 7). Rules: - numpy only (PIL is imported lazily for images). No ttnn, no torch, no network: this module is imported by the image's build-time ``verify:`` step and by host tests. - Data parsing only: ``np.load(..., allow_pickle=False)``, no ``eval`` / ``pickle``. Every client mistake raises :class:`InputError` (the server maps it to HTTP 400). Point-cloud envelope (``"points"`` in the request):: {"format": "bin" | "npy" | "npz" | "pcd" | "list", "data": "", # all formats except "list" "values": [[x, y, z, intensity], ...], # "list" only (small clouds) "fields": ["x", "y", "z", "intensity"], # column names: required meaning for "bin" / "list" / plain arrays "dtype": "float32", # "bin" only: float32 | float64 "key": "points", # "npz" only (default: "points", else the first array) "frame_id": "base_link"} Transforms (``T__from_``, 4x4, metres) accept several spellings, see :func:`parse_transform`. """ from __future__ import annotations import base64 import binascii import io import json import math import re from dataclasses import dataclass, field from pathlib import Path from typing import Any, Iterable, Mapping, Optional, Sequence, Union import numpy as np __all__ = [ "InputError", "PointCloud", "CameraImage", "POINT_FORMATS", "DEFAULT_POINT_FIELDS", "MAX_POINTS_DEFAULT", "b64decode", "decode_points", "load_points", "decode_image", "load_image", "decode_cameras", "load_camera", "load_cameras", "decode_rois", "parse_transform", "parse_intrinsics", "resolve_calibration", "check_named_arrays", "decode_named_arrays", "load_named_arrays", "encode_array", "encode_png", "to_jsonable", ] POINT_FORMATS = ("bin", "npy", "npz", "pcd", "list") DEFAULT_POINT_FIELDS = ("x", "y", "z", "intensity") # Autoware's common LiDAR input layout MAX_POINTS_DEFAULT = 2_000_000 # hard cap per cloud (memory guard, not a model limit) class InputError(ValueError): """A client-side input problem. The HTTP server answers 400 with ``str(err)``.""" # ----------------------------------------------------------------------------------------------- base64 _B64_JUNK = re.compile(r"\s+") def b64decode(s: Any, *, field: str = "data", max_bytes: Optional[int] = None) -> bytes: """Decode standard or URL-safe base64; tolerates a ``data:...;base64,`` prefix, whitespace and missing padding.""" if not isinstance(s, str) or not s: raise InputError(f"{field} must be a non-empty base64 string") if s.startswith("data:"): comma = s.find(",") if comma < 0: raise InputError(f"{field}: malformed data: URI") s = s[comma + 1:] s = _B64_JUNK.sub("", s).replace("-", "+").replace("_", "/") s += "=" * (-len(s) % 4) if max_bytes is not None and len(s) * 3 // 4 > max_bytes + 3: raise InputError(f"{field} is larger than {max_bytes} bytes") try: raw = base64.b64decode(s, validate=True) except (binascii.Error, ValueError) as e: raise InputError(f"{field} is not valid base64: {e}") from None if not raw: raise InputError(f"{field} is empty") return raw # ---------------------------------------------------------------------------------------- point clouds @dataclass class PointCloud: """An (N, C) float32 point array plus its column names and coordinate frame. Rows with a non-finite ``x``, ``y`` or ``z`` (the first three columns when the cloud has no such names) are dropped at construction (PCL ``is_dense=false`` clouds); the remaining rows keep their input order.""" points: np.ndarray fields: tuple frame_id: str = "base_link" def __post_init__(self) -> None: self.fields = tuple(self.fields) if self.points.ndim != 2 or self.points.shape[1] != len(self.fields): raise InputError(f"points have shape {tuple(self.points.shape)} but {len(self.fields)} fields " f"{list(self.fields)}") if self.points.dtype != np.float32: self.points = self.points.astype(np.float32) named = [self.fields.index(n) for n in ("x", "y", "z") if n in self.fields] xyz = self.points[:, named] if named else self.points[:, :3] if not np.isfinite(xyz).all(): keep = np.isfinite(xyz).all(axis=1) self.points = np.ascontiguousarray(self.points[keep]) def __len__(self) -> int: return int(self.points.shape[0]) def select(self, names: Sequence[str], *, fill: Optional[Mapping[str, float]] = None) -> np.ndarray: """Columns ``names`` in that order (C-contiguous float32). A missing column is an error unless ``fill`` gives it a constant (e.g. ``{"time_lag": 0.0}`` for a single sweep).""" cols = [] index = {n: i for i, n in enumerate(self.fields)} for n in names: if n in index: cols.append(self.points[:, index[n]]) elif fill is not None and n in fill: cols.append(np.full(len(self), fill[n], np.float32)) else: raise InputError(f"point field {n!r} is required; the cloud has {list(self.fields)}") return np.ascontiguousarray(np.stack(cols, axis=1), dtype=np.float32) _PCD_TYPES = {("F", 4): " bytes: """liblzf decompression (the codec of PCD ``DATA binary_compressed``). Pure Python: fine for test clouds, slow (~1 s per 10 MB) for big ones -- send ``binary`` PCD or ``.npy`` when latency matters.""" out = bytearray(out_len) ip = op = 0 n = len(data) while ip < n: ctrl = data[ip] ip += 1 if ctrl < 32: # literal run of ctrl + 1 bytes ln = ctrl + 1 if ip + ln > n or op + ln > out_len: raise InputError("pcd: corrupt LZF stream (literal run)") out[op:op + ln] = data[ip:ip + ln] ip += ln op += ln else: # back reference ln = ctrl >> 5 ref = op - ((ctrl & 0x1F) << 8) - 1 if ln == 7: if ip >= n: raise InputError("pcd: corrupt LZF stream (length)") ln += data[ip] ip += 1 if ip >= n: raise InputError("pcd: corrupt LZF stream (offset)") ref -= data[ip] ip += 1 ln += 2 if ref < 0 or op + ln > out_len: raise InputError("pcd: corrupt LZF stream (back reference)") if ref + ln <= op: out[op:op + ln] = out[ref:ref + ln] else: # overlapping copy: byte by byte, as liblzf does for k in range(ln): out[op + k] = out[ref + k] op += ln if op != out_len: raise InputError(f"pcd: LZF produced {op} bytes, header says {out_len}") return bytes(out) def _parse_pcd(raw: bytes) -> tuple: """PCD v0.7 (ascii / binary / binary_compressed) -> (float32 (N, C) array, field names).""" header: dict = {} pos = 0 while True: end = raw.find(b"\n", pos) if end < 0: raise InputError("pcd: header has no DATA line") line = raw[pos:end].decode("ascii", "replace").strip() pos = end + 1 if not line or line.startswith("#"): continue key, _, rest = line.partition(" ") header[key.upper()] = rest.split() if key.upper() == "DATA": break try: names = header["FIELDS"] sizes = [int(v) for v in header["SIZE"]] types = [v.upper() for v in header["TYPE"]] counts = [int(v) for v in header.get("COUNT", ["1"] * len(names))] npts = int(header.get("POINTS", [0])[0]) or int(header["WIDTH"][0]) * int(header.get("HEIGHT", ["1"])[0]) mode = header["DATA"][0].lower() except (KeyError, ValueError, IndexError) as e: raise InputError(f"pcd: incomplete header ({e})") from None if not (len(names) == len(sizes) == len(types) == len(counts)): raise InputError("pcd: FIELDS/SIZE/TYPE/COUNT lengths differ") cols, dt = [], [] for n, s, t, c in zip(names, sizes, types, counts): npt = _PCD_TYPES.get((t, s)) if npt is None: raise InputError(f"pcd: unsupported field type {t}{s} for {n!r}") fname = n if n != "_" else f"_pad{len(dt)}" dt.append((fname, npt, (c,)) if c > 1 else (fname, npt)) if n != "_": cols += [n] if c == 1 else [f"{n}_{k}" for k in range(c)] dtype = np.dtype(dt) body = raw[pos:] if mode == "ascii": try: arr = np.loadtxt(io.StringIO(body.decode("ascii")), dtype=np.float64, ndmin=2) except ValueError as e: raise InputError(f"pcd: bad ascii body ({e})") from None flat_names = [n if c == 1 else f"{n}_{k}" for n, c in zip(names, counts) for k in range(c)] keep = [i for i, n in enumerate(flat_names) if not n.startswith("_")] if arr.shape[1] != len(flat_names): raise InputError(f"pcd: ascii rows have {arr.shape[1]} values, header declares {len(flat_names)}") return arr[:, keep].astype(np.float32), tuple(cols) if mode == "binary": need = dtype.itemsize * npts if len(body) < need: raise InputError(f"pcd: binary body has {len(body)} bytes, needs {need}") rec = np.frombuffer(body, dtype=dtype, count=npts) elif mode == "binary_compressed": if len(body) < 8: raise InputError("pcd: binary_compressed body too short") csize, usize = np.frombuffer(body[:8], " tuple: cols, names = [], [] for name in rec.dtype.names: if name.startswith("_pad"): continue v = np.asarray(rec[name], dtype=np.float32) if v.ndim == 1: cols.append(v) names.append(name) else: for k in range(v.shape[1]): cols.append(v[:, k]) names.append(f"{name}_{k}") return np.stack(cols, axis=1) if cols else np.zeros((len(rec), 0), np.float32), tuple(names) def _array_to_cloud(arr: np.ndarray, fields: Optional[Sequence[str]], frame_id: str, default_fields: Sequence[str]) -> PointCloud: if arr.dtype.names: # structured array (npy / npz written from a ROS PointCloud2 or PCL) pts, names = _structured_to_2d(arr.reshape(-1)) return PointCloud(pts, names, frame_id) arr = np.asarray(arr) if arr.ndim != 2: raise InputError(f"points must be a 2-D (N, C) array, got shape {tuple(arr.shape)}") if not np.issubdtype(arr.dtype, np.number): raise InputError(f"points must be numeric, got dtype {arr.dtype}") names = tuple(fields) if fields else tuple(default_fields[: arr.shape[1]]) if len(names) != arr.shape[1]: raise InputError(f"points have {arr.shape[1]} columns; give 'fields' (default {list(default_fields)})") return PointCloud(np.ascontiguousarray(arr, dtype=np.float32), names, frame_id) def _points_from_bytes(raw: bytes, fmt: str, *, fields: Optional[Sequence[str]] = None, dtype: str = "float32", key: Optional[str] = None, frame_id: str = "base_link", default_fields: Sequence[str] = DEFAULT_POINT_FIELDS) -> PointCloud: if fmt == "bin": # raw little-endian rows, KITTI / Autoware dump style: N x len(fields) names = tuple(fields) if fields else tuple(default_fields) if dtype not in ("float32", "float64"): raise InputError("bin dtype must be float32 or float64") item = 4 if dtype == "float32" else 8 row = item * len(names) if len(raw) % row: raise InputError(f"bin payload of {len(raw)} bytes is not a multiple of {len(names)} x {dtype} " f"({row} bytes per point); check 'fields' (got {list(names)})") arr = np.frombuffer(raw, dtype=" PointCloud: """The JSON point-cloud envelope (module docstring) -> :class:`PointCloud`.""" if not isinstance(spec, Mapping): raise InputError("points must be an object {format, data, fields, ...}") fmt = str(spec.get("format") or "bin").lower() frame_id = str(spec.get("frame_id") or "base_link") fields = spec.get("fields") if fields is not None and (not isinstance(fields, (list, tuple)) or not all(isinstance(f, str) for f in fields)): raise InputError("points.fields must be a list of column names") if fmt == "list": values = spec.get("values") if not isinstance(values, list) or not values: raise InputError("points.values must be a non-empty list of rows for format 'list'") try: arr = np.asarray(values, dtype=np.float32) except (TypeError, ValueError) as e: raise InputError(f"points.values: {e}") from None cloud = _array_to_cloud(arr, fields, frame_id, default_fields) else: raw = b64decode(spec.get("data"), field="points.data", max_bytes=max_bytes) cloud = _points_from_bytes(raw, fmt, fields=fields, dtype=str(spec.get("dtype") or "float32"), key=spec.get("key"), frame_id=frame_id, default_fields=default_fields) if len(cloud) > max_points: raise InputError(f"{len(cloud)} points exceed the limit of {max_points}") if len(cloud) == 0: raise InputError("the point cloud is empty") return cloud PointsLike = Union[str, Path, bytes, np.ndarray, Mapping[str, Any], PointCloud] def load_points(source: PointsLike, *, fields: Optional[Sequence[str]] = None, fmt: Optional[str] = None, frame_id: str = "base_link", default_fields: Sequence[str] = DEFAULT_POINT_FIELDS) -> PointCloud: """Python-API input: a path (.bin/.npy/.npz/.pcd), raw file bytes (give ``fmt``), an (N, C) array (or torch tensor), a structured array, the JSON envelope, or a :class:`PointCloud`.""" if isinstance(source, PointCloud): return source if isinstance(source, Mapping): return decode_points(source, default_fields=default_fields) if isinstance(source, (str, Path)): p = Path(source) ext = (fmt or p.suffix.lstrip(".")).lower() if p.name.endswith(".pcd.bin"): # nuScenes naming: raw float32 x5 ext = "bin" return _points_from_bytes(p.read_bytes(), ext, fields=fields, frame_id=frame_id, default_fields=default_fields) if isinstance(source, (bytes, bytearray, memoryview)): if not fmt: raise InputError("raw bytes need fmt= ('bin' | 'npy' | 'npz' | 'pcd')") return _points_from_bytes(bytes(source), fmt, fields=fields, frame_id=frame_id, default_fields=default_fields) if hasattr(source, "detach") and hasattr(source, "cpu"): # torch tensor, without importing torch source = source.detach().cpu().numpy() return _array_to_cloud(np.asarray(source), fields, frame_id, default_fields) # ---------------------------------------------------------------------------------------------- images def _pil_to_rgb(im) -> np.ndarray: if im.mode != "RGB": im = im.convert("RGB") return np.asarray(im, dtype=np.uint8) def decode_image(raw: bytes, *, fmt: str = "auto", max_side: int = 8192) -> np.ndarray: """PNG / JPEG (or .npy uint8 HxWx3) bytes -> RGB uint8 (H, W, 3).""" if fmt == "npy": try: arr = np.load(io.BytesIO(raw), allow_pickle=False) except ValueError as e: raise InputError(f"image npy: {e}") from None return load_image(arr) from PIL import Image, UnidentifiedImageError # lazy: the API may never see an image try: with Image.open(io.BytesIO(raw)) as im: if max(im.size) > max_side: raise InputError(f"image size {im.size} exceeds max side {max_side}") return _pil_to_rgb(im) except (UnidentifiedImageError, OSError) as e: raise InputError(f"image could not be decoded as PNG/JPEG: {e}") from None def load_image(image: Any) -> np.ndarray: """Python-API image input: path, bytes, PIL image, uint8 HxWx3 / 3xHxW array or tensor, float array in [0, 1] -> RGB uint8 (H, W, 3).""" if isinstance(image, (str, Path)): return decode_image(Path(image).read_bytes()) if isinstance(image, (bytes, bytearray, memoryview)): return decode_image(bytes(image)) if hasattr(image, "mode") and hasattr(image, "size") and hasattr(image, "convert"): # PIL return _pil_to_rgb(image) if hasattr(image, "detach") and hasattr(image, "cpu"): image = image.detach().cpu().numpy() a = np.asarray(image) if a.ndim == 3 and a.shape[0] in (1, 3) and a.shape[2] not in (1, 3): a = np.transpose(a, (1, 2, 0)) if a.ndim == 2: a = np.repeat(a[:, :, None], 3, axis=2) if a.ndim != 3 or a.shape[2] not in (1, 3, 4): raise InputError(f"image array must be HxWx3 (or 3xHxW / HxW), got {a.shape}") a = a[:, :, :3] if a.shape[2] != 1 else np.repeat(a, 3, axis=2) if np.issubdtype(a.dtype, np.floating): a = np.clip(np.round(a * 255.0), 0, 255) return np.ascontiguousarray(a, dtype=np.uint8) @dataclass class CameraImage: """One camera of a (multi-)camera request: pixels + calibration, in the order the model expects.""" name: str image: np.ndarray # RGB uint8 (H, W, 3) intrinsics: Optional[np.ndarray] = None # (3, 3) float64, pixels T_ref_from_camera: Optional[np.ndarray] = None # (4, 4) float64: camera -> reference frame (base_link / lidar) distortion: Optional[dict] = None timestamp_s: Optional[float] = None extra: dict = field(default_factory=dict) def decode_cameras(images: Iterable[Mapping[str, Any]], calibration: Optional[Mapping[str, Any]] = None, *, order: Optional[Sequence[str]] = None, require_calibration: bool = True, max_bytes: Optional[int] = None) -> list: """``images[]`` (+ optional ``calibration.cameras``) -> list of :class:`CameraImage` in ``order``. Each image entry: ``{"camera": "CAM_FRONT", "data": , "format": "auto|png|jpeg|npy", "intrinsics": ..., "T_ref_from_camera": ..., "distortion": {...}, "timestamp_s": ...}``. Inline calibration wins over ``calibration.cameras[]``. ``order`` (the model's fixed camera order) reorders and checks that every camera is present exactly once.""" calib = dict((calibration or {}).get("cameras") or {}) out = [] for i, entry in enumerate(images or []): if not isinstance(entry, Mapping): raise InputError(f"images[{i}] must be an object") name = str(entry.get("camera") or (order[i] if order and i < len(order) else f"cam{i}")) raw = b64decode(entry.get("data"), field=f"images[{i}].data", max_bytes=max_bytes) cam_cal = {**dict(calib.get(name) or {}), **{k: v for k, v in entry.items() if k not in ("camera", "data", "format")}} k = cam_cal.get("intrinsics") t = cam_cal.get("T_ref_from_camera", cam_cal.get("extrinsics")) if require_calibration and (k is None or t is None): raise InputError(f"camera {name!r} needs 'intrinsics' and 'T_ref_from_camera' (inline or in calibration)") out.append(CameraImage( name=name, image=decode_image(raw, fmt=str(entry.get("format") or "auto")), intrinsics=None if k is None else parse_intrinsics(k, field=f"{name}.intrinsics"), T_ref_from_camera=None if t is None else parse_transform(t, field=f"{name}.T_ref_from_camera"), distortion=cam_cal.get("distortion"), timestamp_s=cam_cal.get("timestamp_s"))) if order: by_name = {c.name: c for c in out} if len(by_name) != len(out): raise InputError("duplicate camera names in images[]") missing = [n for n in order if n not in by_name] extra = [n for n in by_name if n not in order] if missing or extra: raise InputError(f"cameras must be exactly {list(order)}; missing {missing}, unexpected {extra}") out = [by_name[n] for n in order] return out _CAMERA_KEYS = ("intrinsics", "T_ref_from_camera", "extrinsics", "distortion", "timestamp_s") def _with_calibration(cam: CameraImage, cal: Mapping[str, Any], require: bool) -> CameraImage: """``cam`` with missing intrinsics / extrinsics / distortion / timestamp taken from ``cal`` (its calibration entry), parsed and checked; ``require`` refuses a camera without K and T_ref_from_camera.""" k = cam.intrinsics if cam.intrinsics is not None else cal.get("intrinsics") t = cam.T_ref_from_camera if cam.T_ref_from_camera is not None else cal.get("T_ref_from_camera", cal.get("extrinsics")) if require and (k is None or t is None): raise InputError(f"camera {cam.name!r} needs 'intrinsics' and 'T_ref_from_camera' (inline or in calibration)") return CameraImage( name=cam.name, image=cam.image, intrinsics=None if k is None else parse_intrinsics(k, field=f"{cam.name}.intrinsics"), T_ref_from_camera=None if t is None else parse_transform(t, field=f"{cam.name}.T_ref_from_camera"), distortion=cam.distortion if cam.distortion is not None else cal.get("distortion"), timestamp_s=cam.timestamp_s if cam.timestamp_s is not None else cal.get("timestamp_s"), extra=dict(cam.extra)) def load_camera(source: Any, *, name: Optional[str] = None, calibration: Optional[Mapping[str, Any]] = None, require_calibration: bool = False, max_bytes: Optional[int] = None) -> CameraImage: """One camera of the Python API -> :class:`CameraImage` (RGB uint8 HxWx3, K / T parsed). ``source``: a :class:`CameraImage` (what the server decodes ``/predict`` into), a mapping ``{"camera", "image" | "path", "intrinsics", "T_ref_from_camera", "distortion", "timestamp_s"}`` or the ``/predict`` spelling ``{"camera", "data": , ...}``, or any image :func:`load_image` accepts (then ``name`` names it). Calibration missing inline comes from ``calibration["cameras"][]`` (inline wins, as in :func:`decode_cameras`).""" cams = dict((calibration or {}).get("cameras") or {}) if isinstance(source, CameraImage): cam = source elif isinstance(source, Mapping): cam_name = str(source.get("camera") or name or "") if not cam_name: raise InputError("every camera needs a 'camera' name") if "data" in source: raw = b64decode(source.get("data"), field=f"{cam_name}.data", max_bytes=max_bytes) image = decode_image(raw, fmt=str(source.get("format") or "auto")) else: src = source.get("image", source.get("path")) if src is None: raise InputError(f"camera {cam_name!r}: no 'image', 'path' or 'data'") image = load_image(src) unknown = sorted(set(source) - {"camera", "data", "format", "image", "path", *_CAMERA_KEYS}) if unknown: raise InputError(f"camera {cam_name!r}: unknown keys {unknown}") cam = CameraImage(name=cam_name, image=image, intrinsics=source.get("intrinsics"), T_ref_from_camera=source.get("T_ref_from_camera", source.get("extrinsics")), distortion=source.get("distortion"), timestamp_s=source.get("timestamp_s")) else: if not name: raise InputError("an image without a camera name: pass {'camera': name, 'image': ...} or a " "{name: image} mapping") cam = CameraImage(name=str(name), image=load_image(source)) return _with_calibration(cam, dict(cams.get(cam.name) or {}), require_calibration) def load_cameras(images: Any, calibration: Optional[Mapping[str, Any]] = None, *, order: Optional[Sequence[str]] = None, require_calibration: bool = True, calib_dir: Optional[Path] = None, max_bytes: Optional[int] = None) -> list: """The Python-API twin of :func:`decode_cameras`: ``images`` (a list of :func:`load_camera` sources, or a ``{camera name: image}`` mapping) + ``calibration`` (``{"cameras": {name: {...}}}``, or ``{"preset": name}`` resolved in ``calib_dir``) -> :class:`CameraImage` list, reordered to ``order`` (every camera exactly once) when given, else in input order. ``model(images=..., calibration=...)`` and ``POST /predict`` then see the same cameras: the server passes the CameraImages it decoded, which come back unchanged.""" if images is None: raise InputError("this model needs 'images'" + (f": the cameras {list(order)}" if order else "")) calib = resolve_calibration(calibration, calib_dir) if calibration is not None else None if isinstance(images, Mapping) and not ({"camera", "image", "path", "data"} & set(images)): entries = [(str(k), v) for k, v in images.items()] else: seq = list(images) if isinstance(images, (list, tuple)) else [images] entries = [(None, v) for v in seq] out = [load_camera(v, name=k, calibration=calib, require_calibration=require_calibration, max_bytes=max_bytes) for k, v in entries] names = [c.name for c in out] if len(set(names)) != len(names): raise InputError(f"duplicate camera names in images: {names}") if order: missing = [n for n in order if n not in names] extra = [n for n in names if n not in order] if missing or extra: raise InputError(f"cameras must be exactly {list(order)}; missing {missing}, unexpected {extra}") by_name = {c.name: c for c in out} out = [by_name[n] for n in order] return out def decode_rois(rois: Any, *, cameras: Optional[Sequence[str]] = None, labels: Optional[Sequence[str]] = None, max_rois: int = 4096) -> list: """2-D detections given as an input (PointPainting's ``rois``, BUNDLE_CONVENTIONS.md 7.2): ``[{"camera", "label", "score", "box_xyxy"}]`` -> the same list checked and normalised: ``camera`` a name (one of ``cameras`` when given), ``label`` a name of ``labels`` or an index into it (returned as both ``label`` and ``label_id`` when ``labels`` is given), ``score`` in [0, 1] (default 1.0), ``box_xyxy`` 4 finite pixels with x0 <= x1 and y0 <= y1. ``None`` -> ``[]``.""" if rois is None: return [] if not isinstance(rois, (list, tuple)): raise InputError("rois must be a list of {camera, label, score, box_xyxy}") if len(rois) > max_rois: raise InputError(f"{len(rois)} rois exceed the limit of {max_rois}") out = [] for i, r in enumerate(rois): where = f"rois[{i}]" if not isinstance(r, Mapping): raise InputError(f"{where} must be an object") unknown = sorted(set(r) - {"camera", "label", "label_id", "score", "box_xyxy"}) if unknown: raise InputError(f"{where}: unknown keys {unknown}") cam = r.get("camera") if not isinstance(cam, str) or not cam: raise InputError(f"{where}.camera must be a camera name") if cameras is not None and cam not in cameras: raise InputError(f"{where}.camera {cam!r} is not one of {list(cameras)}") label = r.get("label", r.get("label_id")) if isinstance(label, bool) or not isinstance(label, (str, int)): raise InputError(f"{where}.label must be a class name or index") entry: dict = {"camera": cam} if labels is not None: names = list(labels) if isinstance(label, int): if not 0 <= label < len(names): raise InputError(f"{where}.label {label} is not an index of {names}") entry.update(label=names[label], label_id=label) elif label in names: entry.update(label=label, label_id=names.index(label)) else: raise InputError(f"{where}.label {label!r} is not one of {names}") else: entry["label"] = label score = r.get("score", 1.0) try: score = float(score) except (TypeError, ValueError): raise InputError(f"{where}.score must be a number") from None if isinstance(r.get("score"), bool) or not 0.0 <= score <= 1.0: raise InputError(f"{where}.score must be in [0, 1]") try: box = [float(v) for v in r.get("box_xyxy")] except (TypeError, ValueError): raise InputError(f"{where}.box_xyxy must be 4 numbers") from None if len(box) != 4 or not all(math.isfinite(v) for v in box) or box[0] > box[2] or box[1] > box[3]: raise InputError(f"{where}.box_xyxy must be 4 finite pixels with x0 <= x1 and y0 <= y1") entry.update(score=score, box_xyxy=box) out.append(entry) return out # ----------------------------------------------------------------------------------------- calibration def _rot_from_quat_wxyz(w: float, x: float, y: float, z: float) -> np.ndarray: n = math.sqrt(w * w + x * x + y * y + z * z) if n < 1e-9: raise InputError("zero-length quaternion") w, x, y, z = w / n, x / n, y / n, z / n return np.array([[1 - 2 * (y * y + z * z), 2 * (x * y - z * w), 2 * (x * z + y * w)], [2 * (x * y + z * w), 1 - 2 * (x * x + z * z), 2 * (y * z - x * w)], [2 * (x * z - y * w), 2 * (y * z + x * w), 1 - 2 * (x * x + y * y)]], dtype=np.float64) def _rot_from_rpy(roll: float, pitch: float, yaw: float) -> np.ndarray: """tf2 ``setRPY`` convention used by Autoware's sensor_kit_calibration.yaml: R = Rz(yaw) Ry(pitch) Rx(roll).""" cr, sr, cp, sp, cy, sy = (math.cos(roll), math.sin(roll), math.cos(pitch), math.sin(pitch), math.cos(yaw), math.sin(yaw)) return np.array([[cy * cp, cy * sp * sr - sy * cr, cy * sp * cr + sy * sr], [sy * cp, sy * sp * sr + cy * cr, sy * sp * cr - cy * sr], [-sp, cp * sr, cp * cr]], dtype=np.float64) def parse_transform(obj: Any, *, field: str = "transform") -> np.ndarray: """A rigid transform -> (4, 4) float64. Accepted spellings: - a 4x4 (or 3x4) nested list / array, row-major, translation in the last column; - ``{"matrix": 4x4}``; - ``{"translation": [x, y, z], "rotation_wxyz": [w, x, y, z]}`` (nuScenes / BEVDet sample yaml), or ``rotation_xyzw``, or ``rotation`` (3x3); - ``{"x", "y", "z", "roll", "pitch", "yaw"}`` in metres / radians (Autoware ``sensor_kit_calibration.yaml``). """ try: if isinstance(obj, Mapping): if "matrix" in obj: return parse_transform(obj["matrix"], field=field) if "translation" in obj: t = np.asarray(obj["translation"], dtype=np.float64).reshape(3) if "rotation_wxyz" in obj: r = _rot_from_quat_wxyz(*np.asarray(obj["rotation_wxyz"], dtype=np.float64).reshape(4)) elif "rotation_xyzw" in obj: x, y, z, w = np.asarray(obj["rotation_xyzw"], dtype=np.float64).reshape(4) r = _rot_from_quat_wxyz(w, x, y, z) elif "rotation" in obj: r = np.asarray(obj["rotation"], dtype=np.float64).reshape(3, 3) else: raise InputError(f"{field}: translation needs rotation_wxyz | rotation_xyzw | rotation (3x3)") elif {"x", "y", "z", "roll", "pitch", "yaw"} <= set(obj): t = np.array([obj["x"], obj["y"], obj["z"]], dtype=np.float64) r = _rot_from_rpy(float(obj["roll"]), float(obj["pitch"]), float(obj["yaw"])) else: raise InputError(f"{field}: unrecognised transform keys {sorted(obj)}") m = np.eye(4) m[:3, :3], m[:3, 3] = r, t else: m = np.asarray(obj, dtype=np.float64) if m.shape == (3, 4): m = np.vstack([m, [0.0, 0.0, 0.0, 1.0]]) if m.shape != (4, 4): raise InputError(f"{field}: expected a 4x4 matrix, got shape {m.shape}") except (TypeError, ValueError) as e: if isinstance(e, InputError): raise raise InputError(f"{field}: {e}") from None if not np.isfinite(m).all() or not np.allclose(m[3], [0, 0, 0, 1], atol=1e-6): raise InputError(f"{field}: last row must be [0, 0, 0, 1]") r = m[:3, :3] if not np.allclose(r.T @ r, np.eye(3), atol=1e-3) or np.linalg.det(r) < 0: raise InputError(f"{field}: rotation part is not a proper rotation") return m def parse_intrinsics(obj: Any, *, field: str = "intrinsics") -> np.ndarray: """3x3 K, a 3x4 projection P (its left 3x3), or ``{"fx", "fy", "cx", "cy"[, "skew"]}`` -> (3, 3) float64.""" try: if isinstance(obj, Mapping): k = np.array([[obj["fx"], obj.get("skew", 0.0), obj["cx"]], [0.0, obj["fy"], obj["cy"]], [0.0, 0.0, 1.0]], dtype=np.float64) else: k = np.asarray(obj, dtype=np.float64) if k.shape == (3, 4): k = k[:, :3] if k.shape != (3, 3): raise InputError(f"{field}: expected 3x3, got shape {k.shape}") except (KeyError, TypeError, ValueError) as e: if isinstance(e, InputError): raise raise InputError(f"{field}: {e}") from None if not np.isfinite(k).all() or k[0, 0] <= 0 or k[1, 1] <= 0: raise InputError(f"{field}: fx and fy must be positive") return k _PRESET_NAME = re.compile(r"[A-Za-z0-9_.-]+") def resolve_calibration(calibration: Optional[Mapping[str, Any]], calib_dir: Optional[Path]) -> Optional[dict]: """``{"preset": "", ...}`` -> the preset file ``/.json`` merged with the other keys (request keys win); any other mapping is returned as a dict; ``None`` stays ``None``.""" if calibration is None: return None if "preset" not in calibration: return dict(calibration) name = str(calibration["preset"]) if calib_dir is None or not _PRESET_NAME.fullmatch(name) or not (Path(calib_dir) / f"{name}.json").is_file(): raise InputError(f"unknown calibration preset {name!r}; see /info input.calibration_presets") base = json.loads((Path(calib_dir) / f"{name}.json").read_text()) return {**base, **{k: v for k, v in calibration.items() if k != "preset"}} # ------------------------------------------------------------------------------ named tensors (planner) def check_named_arrays(arrays: Mapping[str, Any], schema: Mapping[str, tuple]) -> dict: """``{name: array}`` checked against ``schema`` (name -> (shape with ``None`` for free dims, dtype)): every listed name is required, unlisted names are rejected, shapes must match and values must convert to the dtype without loss of kind (numbers only; a float into an integer slot must be integral, values must be finite). Returns C-contiguous arrays of the schema dtypes; every problem raises :class:`InputError`.""" if not isinstance(arrays, Mapping): raise InputError("inputs must be an object {name: array}") unknown = sorted(set(arrays) - set(schema)) missing = sorted(set(schema) - set(arrays)) if unknown or missing: raise InputError(f"inputs: missing {missing}, unexpected {unknown}") out = {} for name, (shape, dtype) in schema.items(): try: a = np.asarray(arrays[name]) except (TypeError, ValueError) as e: raise InputError(f"inputs[{name!r}]: {e}") from None if len(a.shape) != len(shape) or any(s is not None and s != d for s, d in zip(shape, a.shape)): raise InputError(f"inputs[{name!r}] has shape {tuple(a.shape)}, expected {tuple(shape)}") if not (np.issubdtype(a.dtype, np.number) or a.dtype == np.bool_): raise InputError(f"inputs[{name!r}] must be numeric, got dtype {a.dtype}") target = np.dtype(dtype) if a.size and np.issubdtype(a.dtype, np.inexact) and not np.isfinite(a).all(): raise InputError(f"inputs[{name!r}] holds non-finite values") if a.size and np.issubdtype(target, np.integer) and a.dtype != np.bool_: if np.issubdtype(a.dtype, np.inexact) and not np.array_equal(a, np.round(a)): raise InputError(f"inputs[{name!r}] must hold integers ({target})") info = np.iinfo(target) if a.min() < info.min or a.max() > info.max: raise InputError(f"inputs[{name!r}] has values outside the {target} range") out[name] = np.ascontiguousarray(a.astype(target, copy=False)) return out def decode_named_arrays(spec: Mapping[str, Any], schema: Optional[Mapping[str, tuple]] = None, *, max_bytes: Optional[int] = None) -> dict: """``{"format": "npz", "data": }`` or ``{"format": "json", "arrays": {name: nested list}}`` -> ``{name: ndarray}``. With ``schema`` the arrays are checked and cast by :func:`check_named_arrays`.""" if not isinstance(spec, Mapping): raise InputError("inputs must be an object {format, data | arrays}") fmt = str(spec.get("format") or "npz").lower() if fmt == "npz": raw = b64decode(spec.get("data"), field="inputs.data", max_bytes=max_bytes) try: z = np.load(io.BytesIO(raw), allow_pickle=False) arrays = {k: z[k] for k in z.files} except (ValueError, OSError) as e: raise InputError(f"inputs npz: {e}") from None elif fmt == "json": src = spec.get("arrays") if not isinstance(src, Mapping): raise InputError("inputs.arrays must be an object {name: nested list}") arrays = {} for k, v in src.items(): try: arrays[k] = np.asarray(v) except (TypeError, ValueError) as e: raise InputError(f"inputs.arrays[{k!r}]: {e}") from None else: raise InputError(f"inputs.format {fmt!r} must be 'npz' or 'json'") return arrays if schema is None else check_named_arrays(arrays, schema) def load_named_arrays(source: Any, schema: Optional[Mapping[str, tuple]] = None) -> dict: """Python-API named inputs (planner): a ``{name: array}`` mapping, the JSON envelope of :func:`decode_named_arrays`, an ``.npz`` path or its bytes -> ``{name: ndarray}`` (checked against ``schema`` when given, so ``model(inputs=...)`` and ``POST /predict`` accept and refuse the same inputs).""" if isinstance(source, Mapping) and "format" in source and ("data" in source or "arrays" in source): return decode_named_arrays(source, schema) if isinstance(source, (str, Path, bytes, bytearray, memoryview)): try: payload = Path(source).read_bytes() if isinstance(source, (str, Path)) else bytes(source) with np.load(io.BytesIO(payload), allow_pickle=False) as z: arrays = {k: z[k] for k in z.files} except (ValueError, OSError) as e: raise InputError(f"inputs npz: {e}") from None elif isinstance(source, Mapping): arrays = {str(k): (v.detach().cpu().numpy() if hasattr(v, "detach") else v) for k, v in source.items()} else: raise InputError(f"inputs must be a mapping, an .npz path or bytes, not {type(source).__name__}") return arrays if schema is None else check_named_arrays(arrays, schema) # ------------------------------------------------------------------------------------------- outputs def encode_array(arr: np.ndarray, *, fmt: str = "npz", key: str = "array") -> dict: """A dense output for the JSON envelope: ``{"format", "key", "dtype", "shape", "data"}``. ``npz`` (default): base64 of ``np.savez`` -> ``np.load(io.BytesIO(base64.b64decode(d["data"])))[d["key"]]``; ``npz_compressed``: the same with ``savez_compressed`` (smaller, slower); ``raw``: base64 of the little-endian bytes (``np.frombuffer(..., dtype).reshape(shape)``); ``list``: nested JSON lists (small arrays only).""" arr = np.asarray(arr) head = {"format": fmt, "key": key, "dtype": str(arr.dtype), "shape": list(arr.shape)} if fmt in ("npz", "npz_compressed"): buf = io.BytesIO() (np.savez_compressed if fmt == "npz_compressed" else np.savez)(buf, **{key: arr}) return {**head, "data": base64.b64encode(buf.getvalue()).decode("ascii")} if fmt == "raw": a = arr.astype(arr.dtype.newbyteorder("<"), copy=False) return {**head, "data": base64.b64encode(np.ascontiguousarray(a).tobytes()).decode("ascii")} if fmt == "list": return {**head, "data": arr.tolist()} raise ValueError(f"unknown array encoding {fmt!r}") def encode_png(image: np.ndarray, *, key: str = "image") -> dict: """A uint8 mask / label map (H, W) or RGB image (H, W, 3) as base64 PNG (lossless): ``{"format": "png", "key", "dtype", "shape", "data"}``.""" from PIL import Image a = np.asarray(image) if a.dtype != np.uint8: if a.size and (a.min() < 0 or a.max() > 255): raise ValueError("encode_png needs values in 0..255 (use encode_array for wider label ranges)") a = a.astype(np.uint8) if a.ndim not in (2, 3): raise ValueError(f"encode_png expects (H, W) or (H, W, 3), got {a.shape}") buf = io.BytesIO() Image.fromarray(a).save(buf, format="PNG") return {"format": "png", "key": key, "dtype": "uint8", "shape": list(a.shape), "data": base64.b64encode(buf.getvalue()).decode("ascii")} def to_jsonable(obj: Any) -> Any: """numpy scalars / arrays -> plain Python, recursively. float32 / float16 scalars are rounded to 6 significant digits (their precision); float64 scalars and array elements are kept exactly (timestamps, poses).""" if isinstance(obj, Mapping): return {str(k): to_jsonable(v) for k, v in obj.items()} if isinstance(obj, (list, tuple)): return [to_jsonable(v) for v in obj] if isinstance(obj, np.ndarray): return to_jsonable(obj.tolist()) if isinstance(obj, (np.float32, np.float16)): return float(f"{float(obj):.6g}") if isinstance(obj, np.floating): return float(obj) if isinstance(obj, np.integer): return int(obj) if isinstance(obj, np.bool_): return bool(obj) if isinstance(obj, Path): return str(obj) return obj