changh95's picture
tt-model push diffusion-planner-p150 (container)
4d9b003 verified
Raw History Blame Contribute Delete
9.59 kB
# SPDX-License-Identifier: Apache-2.0
"""C14 image pre-processing, METEOR preset: a bit-exact host emulation of OpenCV's ``cv::resize(..., INTER_AREA)`` on
uint8 images (down-scaling), driven by cached per-size tables. numpy only; no ttnn, no torch, no OpenCV.
METEOR's runtimes feed every camera as RGB uint8 768x432 and resize larger frames with ``cv2.resize(img, (768, 432),
interpolation=cv2.INTER_AREA)`` (tier4/METEOR ``hf/onnx_smoke_test.py:42-44``; the training extraction
``bevlane/extract_gt.py:114-130`` resized the raw, unrectified JPEGs the same way; research/meteor/SPEC.md section 3),
anisotropically when the aspect ratio differs (2880x1860 -> 768x432). :func:`meteor_resize` reproduces it; the
intrinsics follow with :func:`scale_intrinsics` (row 0 x 768 / w0, row 1 x 432 / h0, ``extract_gt.py:128-130``).
What OpenCV computes (``modules/imgproc/src/resize.cpp``, identical in 4.8.1 and 4.11.0; the IPP path declines
INTER_AREA because it is not bit-exact, ``ipp_resize``: ``ippSuper`` needs ``useIPP_NotExact``):
- ``inv_scale = (double)dst / src`` per axis, ``scale = 1.0 / inv_scale`` (double).
- **Integer factors** (``|scale - (int)scale| < DBL_EPSILON`` on both axes; ``resizeAreaFast_``): every output pixel is
the integer sum ``s`` of its ``fx * fy`` source pixels, then ``saturate_cast<uchar>(s * (1.f / area))``: one float32
product, rounded half to even (``cvRound``). Exception, the 2x2 ``fast_mode`` of ``ResizeAreaFastVec`` for 1-, 3-
and 4-channel images: ``(s + 2) >> 2`` (half up). Measured on both OpenCV builds (2-channel images take the float
path; 3x3 and 5x5 factors the float path).
- **Other factors** (``resizeArea_<uchar, float>``): ``computeResizeAreaTab`` lists, per output column (and row), the
covered source pixels with float32 weights computed in double (``(sx1 - fsx1) / cellWidth``, ``1.0 / cellWidth``,
``min(min(fsx2 - sx2, 1), cellWidth) / cellWidth``; partial pixels below 1e-3 skipped). Each source row is first
reduced horizontally, ``buf[dx] += S[sx] * alpha`` in table order (float32 multiply, then float32 add: no FMA, the
generic code is compiled for the SSE baseline), then the rows are combined, ``sum[dx] += beta * buf[dx]`` in table
order, and ``saturate_cast<uchar>(sum)`` rounds half to even and clamps to [0, 255].
The emulation replays exactly these float32 operations in the same order, vectorised over the image (zero-weight
padding of the per-pixel tap lists adds exact zeros). Verified against cv2 4.8.1 (tt-env) and 4.11.0 (research venv)
on random and real images of many sizes (``common/tests/host/test_image_area_host.py``): 0 mismatches.
Up-scaling (any axis with ``src < dst``) runs a different OpenCV path (linear-like fixed-point coefficients) and is
refused: METEOR's cameras are all larger than 768x432.
"""
from __future__ import annotations
import sys
from dataclasses import dataclass
from typing import Any, Optional, Tuple
import numpy as np
from .image import LUTCache
__all__ = ["AreaResizeLUT", "opencv_area_lut", "opencv_resize_area_u8", "meteor_resize", "scale_intrinsics",
"METEOR_INPUT_HW"]
F32 = np.float32
METEOR_INPUT_HW = (432, 768) # every METEOR camera slot (meteor_v157.param.yaml:11-12)
_DBL_EPSILON = sys.float_info.epsilon
def _area_axis(n_src: int, n_dst: int, scale: float) -> Tuple[np.ndarray, np.ndarray]:
"""``computeResizeAreaTab`` of one axis as padded per-output tap lists: (src index [n_dst, T], weight [n_dst, T])
in table order; unused taps have weight 0 and index 0."""
taps = []
for dx in range(n_dst):
fsx1 = dx * scale
fsx2 = fsx1 + scale
cell = min(scale, n_src - fsx1)
sx1 = int(np.ceil(fsx1))
sx2 = int(np.floor(fsx2))
sx2 = min(sx2, n_src - 1)
sx1 = min(sx1, sx2)
row = []
if sx1 - fsx1 > 1e-3:
row.append((sx1 - 1, F32((sx1 - fsx1) / cell)))
for sx in range(sx1, sx2):
row.append((sx, F32(1.0 / cell)))
if fsx2 - sx2 > 1e-3:
row.append((sx2, F32(min(min(fsx2 - sx2, 1.0), cell) / cell)))
taps.append(row)
width = max(len(r) for r in taps)
idx = np.zeros((n_dst, width), np.int64)
w = np.zeros((n_dst, width), F32)
for i, row in enumerate(taps):
for k, (s, a) in enumerate(row):
idx[i, k] = s
w[i, k] = a
return idx, w
@dataclass(frozen=True)
class AreaResizeLUT:
"""The cached tables of one (source size, destination size) pair; :meth:`apply` resizes one uint8 image."""
src_hw: Tuple[int, int]
dst_hw: Tuple[int, int]
fast: Optional[Tuple[int, int]] # integer factors (fy, fx) of the resizeAreaFast_ path, else None
rows: Optional[np.ndarray] # [dst_h, Ty] source rows (general path)
row_w: Optional[np.ndarray] # [dst_h, Ty] float32 beta
cols: Optional[np.ndarray] # [dst_w, Tx] source columns
col_w: Optional[np.ndarray] # [dst_w, Tx] float32 alpha
def apply(self, image: np.ndarray, *, out: Optional[np.ndarray] = None) -> np.ndarray:
img = np.asarray(image)
if img.dtype != np.uint8 or img.ndim not in (2, 3) or tuple(img.shape[:2]) != self.src_hw:
raise ValueError(f"expected a uint8 {self.src_hw} image (HxW or HxWxC), got {img.dtype} {img.shape}")
squeeze = img.ndim == 2
x = img[:, :, None] if squeeze else img
dh, dw = self.dst_hw
if self.fast is not None:
fy, fx = self.fast
s = x[:dh * fy, :dw * fx].astype(np.int64).reshape(dh, fy, dw, fx, x.shape[2]).sum(axis=(1, 3))
if (fy, fx) == (2, 2) and x.shape[2] in (1, 3, 4):
res = (s + 2) >> 2 # ResizeAreaFastVec fast_mode: half up
else:
res = np.rint(s.astype(F32) * F32(1.0 / (fx * fy))) # float32 product, half to even
else:
# horizontal pass per source row: buf[dx] = sum_k S[cols[dx, k]] * col_w[dx, k] (float32, table order)
buf = np.zeros((x.shape[0], dw, x.shape[2]), F32)
src = x.astype(F32)
for k in range(self.cols.shape[1]):
buf += src[:, self.cols[:, k], :] * self.col_w[None, :, k, None]
# vertical pass: sum[dy] = sum_j beta[dy, j] * buf[rows[dy, j]] (float32, table order)
acc = np.zeros((dh, dw, x.shape[2]), F32)
for j in range(self.rows.shape[1]):
acc += self.row_w[:, j, None, None] * buf[self.rows[:, j]]
res = np.rint(acc) # cvRound: half to even
res = np.clip(res, 0, 255).astype(np.uint8)
if squeeze:
res = res[:, :, 0]
if out is not None:
out[...] = res
return out
return res
_AREA_LUTS = LUTCache(maxsize=16) # 8 cameras of up to two sizes each
def opencv_area_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> AreaResizeLUT:
"""Tables of ``cv::resize(INTER_AREA)`` from ``src_hw`` down to ``dst_hw`` (cached per size pair)."""
src_hw = (int(src_hw[0]), int(src_hw[1]))
dst_hw = (int(dst_hw[0]), int(dst_hw[1]))
return _AREA_LUTS.get((src_hw, dst_hw), lambda: _build_area_lut(src_hw, dst_hw))
def _build_area_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> AreaResizeLUT:
(sh, sw), (dh, dw) = src_hw, dst_hw
if min(sh, sw, dh, dw) < 1:
raise ValueError(f"bad sizes {src_hw} -> {dst_hw}")
if dh > sh or dw > sw:
raise ValueError(f"INTER_AREA up-scaling {src_hw} -> {dst_hw} is not emulated (OpenCV uses another path)")
scale_y = 1.0 / (float(dh) / sh)
scale_x = 1.0 / (float(dw) / sw)
iy, ix = int(round(scale_y)), int(round(scale_x))
if abs(scale_x - ix) < _DBL_EPSILON and abs(scale_y - iy) < _DBL_EPSILON:
return AreaResizeLUT(src_hw, dst_hw, (iy, ix), None, None, None, None)
rows, row_w = _area_axis(sh, dh, scale_y)
cols, col_w = _area_axis(sw, dw, scale_x)
return AreaResizeLUT(src_hw, dst_hw, None, rows, row_w, cols, col_w)
def opencv_resize_area_u8(image: np.ndarray, dst_hw: Tuple[int, int], *,
out: Optional[np.ndarray] = None) -> np.ndarray:
"""``cv2.resize(image, (dst_w, dst_h), interpolation=cv2.INTER_AREA)`` for uint8 HxW / HxWxC down-scaling;
the same size returns a copy (OpenCV's ``copyTo`` shortcut)."""
img = np.asarray(image)
if tuple(img.shape[:2]) == tuple(dst_hw):
res = np.array(img, dtype=np.uint8, copy=True)
if out is not None:
out[...] = res
return out
return res
return opencv_area_lut(img.shape[:2], dst_hw).apply(img, out=out)
def meteor_resize(image: np.ndarray, *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""One camera frame (uint8 HxWx3, any channel order: the resize is per channel) -> 432x768 as METEOR's runtimes
(``INTER_AREA`` when the size differs, a copy otherwise)."""
return opencv_resize_area_u8(image, METEOR_INPUT_HW, out=out)
def scale_intrinsics(K: Any, src_hw: Tuple[int, int], dst_hw: Tuple[int, int] = METEOR_INPUT_HW) -> np.ndarray:
"""K of a ``src_hw`` image -> K of the resized ``dst_hw`` image: row 0 x dst_w / src_w, row 1 x dst_h / src_h
(``extract_gt.py:128-130``; skew is scaled with row 0 and ignored by the network). float64."""
k = np.array(K, dtype=np.float64).reshape(3, 3).copy()
k[0] *= dst_hw[1] / float(src_hw[1])
k[1] *= dst_hw[0] / float(src_hw[0])
return k