# SPDX-License-Identifier: Apache-2.0 """C14 image pre-processing, METEOR preset: a bit-exact host emulation of OpenCV's ``cv::resize(..., INTER_AREA)`` on uint8 images (down-scaling), driven by cached per-size tables. numpy only; no ttnn, no torch, no OpenCV. METEOR's runtimes feed every camera as RGB uint8 768x432 and resize larger frames with ``cv2.resize(img, (768, 432), interpolation=cv2.INTER_AREA)`` (tier4/METEOR ``hf/onnx_smoke_test.py:42-44``; the training extraction ``bevlane/extract_gt.py:114-130`` resized the raw, unrectified JPEGs the same way; research/meteor/SPEC.md section 3), anisotropically when the aspect ratio differs (2880x1860 -> 768x432). :func:`meteor_resize` reproduces it; the intrinsics follow with :func:`scale_intrinsics` (row 0 x 768 / w0, row 1 x 432 / h0, ``extract_gt.py:128-130``). What OpenCV computes (``modules/imgproc/src/resize.cpp``, identical in 4.8.1 and 4.11.0; the IPP path declines INTER_AREA because it is not bit-exact, ``ipp_resize``: ``ippSuper`` needs ``useIPP_NotExact``): - ``inv_scale = (double)dst / src`` per axis, ``scale = 1.0 / inv_scale`` (double). - **Integer factors** (``|scale - (int)scale| < DBL_EPSILON`` on both axes; ``resizeAreaFast_``): every output pixel is the integer sum ``s`` of its ``fx * fy`` source pixels, then ``saturate_cast(s * (1.f / area))``: one float32 product, rounded half to even (``cvRound``). Exception, the 2x2 ``fast_mode`` of ``ResizeAreaFastVec`` for 1-, 3- and 4-channel images: ``(s + 2) >> 2`` (half up). Measured on both OpenCV builds (2-channel images take the float path; 3x3 and 5x5 factors the float path). - **Other factors** (``resizeArea_``): ``computeResizeAreaTab`` lists, per output column (and row), the covered source pixels with float32 weights computed in double (``(sx1 - fsx1) / cellWidth``, ``1.0 / cellWidth``, ``min(min(fsx2 - sx2, 1), cellWidth) / cellWidth``; partial pixels below 1e-3 skipped). Each source row is first reduced horizontally, ``buf[dx] += S[sx] * alpha`` in table order (float32 multiply, then float32 add: no FMA, the generic code is compiled for the SSE baseline), then the rows are combined, ``sum[dx] += beta * buf[dx]`` in table order, and ``saturate_cast(sum)`` rounds half to even and clamps to [0, 255]. The emulation replays exactly these float32 operations in the same order, vectorised over the image (zero-weight padding of the per-pixel tap lists adds exact zeros). Verified against cv2 4.8.1 (tt-env) and 4.11.0 (research venv) on random and real images of many sizes (``common/tests/host/test_image_area_host.py``): 0 mismatches. Up-scaling (any axis with ``src < dst``) runs a different OpenCV path (linear-like fixed-point coefficients) and is refused: METEOR's cameras are all larger than 768x432. """ from __future__ import annotations import sys from dataclasses import dataclass from typing import Any, Optional, Tuple import numpy as np from .image import LUTCache __all__ = ["AreaResizeLUT", "opencv_area_lut", "opencv_resize_area_u8", "meteor_resize", "scale_intrinsics", "METEOR_INPUT_HW"] F32 = np.float32 METEOR_INPUT_HW = (432, 768) # every METEOR camera slot (meteor_v157.param.yaml:11-12) _DBL_EPSILON = sys.float_info.epsilon def _area_axis(n_src: int, n_dst: int, scale: float) -> Tuple[np.ndarray, np.ndarray]: """``computeResizeAreaTab`` of one axis as padded per-output tap lists: (src index [n_dst, T], weight [n_dst, T]) in table order; unused taps have weight 0 and index 0.""" taps = [] for dx in range(n_dst): fsx1 = dx * scale fsx2 = fsx1 + scale cell = min(scale, n_src - fsx1) sx1 = int(np.ceil(fsx1)) sx2 = int(np.floor(fsx2)) sx2 = min(sx2, n_src - 1) sx1 = min(sx1, sx2) row = [] if sx1 - fsx1 > 1e-3: row.append((sx1 - 1, F32((sx1 - fsx1) / cell))) for sx in range(sx1, sx2): row.append((sx, F32(1.0 / cell))) if fsx2 - sx2 > 1e-3: row.append((sx2, F32(min(min(fsx2 - sx2, 1.0), cell) / cell))) taps.append(row) width = max(len(r) for r in taps) idx = np.zeros((n_dst, width), np.int64) w = np.zeros((n_dst, width), F32) for i, row in enumerate(taps): for k, (s, a) in enumerate(row): idx[i, k] = s w[i, k] = a return idx, w @dataclass(frozen=True) class AreaResizeLUT: """The cached tables of one (source size, destination size) pair; :meth:`apply` resizes one uint8 image.""" src_hw: Tuple[int, int] dst_hw: Tuple[int, int] fast: Optional[Tuple[int, int]] # integer factors (fy, fx) of the resizeAreaFast_ path, else None rows: Optional[np.ndarray] # [dst_h, Ty] source rows (general path) row_w: Optional[np.ndarray] # [dst_h, Ty] float32 beta cols: Optional[np.ndarray] # [dst_w, Tx] source columns col_w: Optional[np.ndarray] # [dst_w, Tx] float32 alpha def apply(self, image: np.ndarray, *, out: Optional[np.ndarray] = None) -> np.ndarray: img = np.asarray(image) if img.dtype != np.uint8 or img.ndim not in (2, 3) or tuple(img.shape[:2]) != self.src_hw: raise ValueError(f"expected a uint8 {self.src_hw} image (HxW or HxWxC), got {img.dtype} {img.shape}") squeeze = img.ndim == 2 x = img[:, :, None] if squeeze else img dh, dw = self.dst_hw if self.fast is not None: fy, fx = self.fast s = x[:dh * fy, :dw * fx].astype(np.int64).reshape(dh, fy, dw, fx, x.shape[2]).sum(axis=(1, 3)) if (fy, fx) == (2, 2) and x.shape[2] in (1, 3, 4): res = (s + 2) >> 2 # ResizeAreaFastVec fast_mode: half up else: res = np.rint(s.astype(F32) * F32(1.0 / (fx * fy))) # float32 product, half to even else: # horizontal pass per source row: buf[dx] = sum_k S[cols[dx, k]] * col_w[dx, k] (float32, table order) buf = np.zeros((x.shape[0], dw, x.shape[2]), F32) src = x.astype(F32) for k in range(self.cols.shape[1]): buf += src[:, self.cols[:, k], :] * self.col_w[None, :, k, None] # vertical pass: sum[dy] = sum_j beta[dy, j] * buf[rows[dy, j]] (float32, table order) acc = np.zeros((dh, dw, x.shape[2]), F32) for j in range(self.rows.shape[1]): acc += self.row_w[:, j, None, None] * buf[self.rows[:, j]] res = np.rint(acc) # cvRound: half to even res = np.clip(res, 0, 255).astype(np.uint8) if squeeze: res = res[:, :, 0] if out is not None: out[...] = res return out return res _AREA_LUTS = LUTCache(maxsize=16) # 8 cameras of up to two sizes each def opencv_area_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> AreaResizeLUT: """Tables of ``cv::resize(INTER_AREA)`` from ``src_hw`` down to ``dst_hw`` (cached per size pair).""" src_hw = (int(src_hw[0]), int(src_hw[1])) dst_hw = (int(dst_hw[0]), int(dst_hw[1])) return _AREA_LUTS.get((src_hw, dst_hw), lambda: _build_area_lut(src_hw, dst_hw)) def _build_area_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> AreaResizeLUT: (sh, sw), (dh, dw) = src_hw, dst_hw if min(sh, sw, dh, dw) < 1: raise ValueError(f"bad sizes {src_hw} -> {dst_hw}") if dh > sh or dw > sw: raise ValueError(f"INTER_AREA up-scaling {src_hw} -> {dst_hw} is not emulated (OpenCV uses another path)") scale_y = 1.0 / (float(dh) / sh) scale_x = 1.0 / (float(dw) / sw) iy, ix = int(round(scale_y)), int(round(scale_x)) if abs(scale_x - ix) < _DBL_EPSILON and abs(scale_y - iy) < _DBL_EPSILON: return AreaResizeLUT(src_hw, dst_hw, (iy, ix), None, None, None, None) rows, row_w = _area_axis(sh, dh, scale_y) cols, col_w = _area_axis(sw, dw, scale_x) return AreaResizeLUT(src_hw, dst_hw, None, rows, row_w, cols, col_w) def opencv_resize_area_u8(image: np.ndarray, dst_hw: Tuple[int, int], *, out: Optional[np.ndarray] = None) -> np.ndarray: """``cv2.resize(image, (dst_w, dst_h), interpolation=cv2.INTER_AREA)`` for uint8 HxW / HxWxC down-scaling; the same size returns a copy (OpenCV's ``copyTo`` shortcut).""" img = np.asarray(image) if tuple(img.shape[:2]) == tuple(dst_hw): res = np.array(img, dtype=np.uint8, copy=True) if out is not None: out[...] = res return out return res return opencv_area_lut(img.shape[:2], dst_hw).apply(img, out=out) def meteor_resize(image: np.ndarray, *, out: Optional[np.ndarray] = None) -> np.ndarray: """One camera frame (uint8 HxWx3, any channel order: the resize is per channel) -> 432x768 as METEOR's runtimes (``INTER_AREA`` when the size differs, a copy otherwise).""" return opencv_resize_area_u8(image, METEOR_INPUT_HW, out=out) def scale_intrinsics(K: Any, src_hw: Tuple[int, int], dst_hw: Tuple[int, int] = METEOR_INPUT_HW) -> np.ndarray: """K of a ``src_hw`` image -> K of the resized ``dst_hw`` image: row 0 x dst_w / src_w, row 1 x dst_h / src_h (``extract_gt.py:128-130``; skew is scaled with row 0 and ignored by the network). float64.""" k = np.array(K, dtype=np.float64).reshape(3, 3).copy() k[0] *= dst_hw[1] / float(src_hw[1]) k[1] *= dst_hw[0] / float(src_hw[0]) return k