changh95's picture
tt-model push diffusion-planner-p150 (container)
4d9b003 verified
Raw History Blame Contribute Delete
15.8 kB
# SPDX-License-Identifier: Apache-2.0
"""C14 image pre-processing, SceneSeg preset: bit-exact host emulations of OpenCV's ``cv::resize`` on uint8 images
with ``INTER_LINEAR`` (the default interpolation) and ``INTER_NEAREST``, driven by cached per-size tables, and the
pre-processing of Autoware VisionPilot's SceneSeg node built on them. numpy only; no ttnn, no torch, no OpenCV.
The SceneSeg node (VisionPilot ``middleware_recipes/common/backends/onnx_runtime_backend.cpp:41-60`` and the identical
``tensorrt_backend.cpp:160-177`` at autowarefoundation/vision_pilot ``04fa3e80``; research/sceneseg/SPEC.md section 3)
takes the ``bgr8`` image of ``cv_bridge``, then::
cv::resize(input_image, resized_image, cv::Size(640, 320)); // INTER_LINEAR, aspect ratio not kept
resized_image.convertTo(float_image, CV_32FC3, 1.0 / 255.0);
cv::subtract(float_image, cv::Scalar(0.406, 0.456, 0.485), float_image);
cv::divide(float_image, cv::Scalar(0.225, 0.224, 0.229), float_image);
cv::split(float_image, channels); // planes B, G, R -> NCHW
and publishes the argmax mask resized back to the source size with ``cv::resize(..., INTER_NEAREST)``
(``run_model_node.cpp:117-188``). :func:`sceneseg_resize` / :func:`sceneseg_normalize` / :func:`sceneseg_preprocess`
reproduce the first part, :func:`opencv_resize_nearest` the mask resize.
What OpenCV computes (``modules/imgproc/src/resize.cpp`` and ``core/src/arithm.cpp``; identical in 4.8.1 and 4.11.0
on this x86-64 host, measured; the IPP path declines both resizes: uint8 ``ippLinear`` and ``ippNearest`` need
``useIPP_NotExact``):
- ``cv::resize`` with ``dsize == ssize`` copies. ``inv_scale = (double)dst / src`` and ``scale = 1.0 / inv_scale``
per axis.
- **INTER_LINEAR, exactly 2x down on both axes**: OpenCV switches to ``INTER_AREA`` (``resizeAreaFast_``), which
:func:`.image_area.opencv_resize_area_u8` emulates (2x2 ``fast_mode``: ``(s + 2) >> 2``).
- **INTER_LINEAR otherwise** (``resizeGeneric_`` with ``HResizeLinear<uchar, int, short, 2048>`` and the uchar
specialisation of ``VResizeLinear``): per output column ``fx = (float)((dx + 0.5) * scale - 0.5)``,
``sx = floor(fx)``, ``fx -= sx`` (float); at the borders ``sx < 0 -> sx = 0, fx = 0`` and ``sx >= W - 1 ->
sx = W - 1, fx = 0``; integer weights ``a0 = cvRound((1.f - fx) * 2048)``, ``a1 = cvRound(fx * 2048)`` (round half
to even). Rows take the same coefficients **without** the border fix (the row indices are clipped to [0, H - 1]
instead, ``clip(sy + k, 0, H)``). The horizontal pass is exact integer arithmetic,
``h = S[sx] * a0 + S[sx + 1] * a1`` (int32), and the vertical pass is the fixed-point
``(((b0 * (h0 >> 4)) >> 16) + ((b1 * (h1 >> 4)) >> 16) + 2) >> 2`` that the SIMD and the scalar code share.
- **INTER_NEAREST** (``resizeNN``): ``sx = min(floor(dx * (1.0 / inv_scale_x)), W - 1)`` in double, rows alike.
- **Normalisation**: ``convertTo(CV_32F, 1/255)`` is one float32 product ``x * (float)(1/255)``; ``cv::subtract``
with a ``Scalar`` runs in float32 (``x - (float)mean_c``); ``cv::divide`` with a ``Scalar`` runs in double and
rounds once to float32 (``(float)((double)x / std_c)``), measured with OpenCV's own scalar arithmetic. A float32
division instead differs by one ulp in about 27 % of the values (the research reference's arithmetic).
Channel orders (SPEC section 3, PLAN.md D10): ``"bgr"`` is the deployed path: B, G, R planes normalised with the
BGR-ordered ImageNet constants, fed to a network trained on R, G, B. ``"rgb"`` is the training-consistent option: the
same arithmetic on R, G, B planes with the RGB constants (an exact channel permutation of the image first, as a
``cv_bridge`` ``rgb8`` conversion would give). Every physical colour gets the same mean and std in both orders, so
the two inputs are channel permutations of each other.
Verified bit-exact against cv2 4.8.1 (tt-env) and 4.11.0 (research venv) on random and real images of many sizes
(down- and up-scaling, odd sizes, 1-4 channels), against a scalar transliteration of ``resize.cpp``, and against the
research golden inputs of the SceneSeg port (``common/tests/host/test_image_sceneseg_host.py``).
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Any, Optional, Tuple
import numpy as np
from .image import LUTCache
from .image_area import opencv_resize_area_u8
__all__ = [
"LinearResizeU8LUT",
"opencv_linear_u8_lut",
"opencv_resize_linear_u8",
"NearestResizeLUT",
"opencv_nearest_lut",
"opencv_resize_nearest",
"sceneseg_resize",
"sceneseg_normalize",
"sceneseg_preprocess",
"SCENESEG_INPUT_HW",
"SCENESEG_MEAN_RGB",
"SCENESEG_STD_RGB",
"SCENESEG_CHANNEL_ORDERS",
]
F32 = np.float32
INTER_RESIZE_COEF_SCALE = 2048 # 1 << INTER_RESIZE_COEF_BITS (11), resize.cpp
SCENESEG_INPUT_HW = (320, 640) # (H, W) of the ONNX input; fixed by the SceneContext reshape (SPEC 4.2)
SCENESEG_MEAN_RGB = (0.485, 0.456, 0.406) # ImageNet, R G B (the C++ code lists them reversed for B G R planes)
SCENESEG_STD_RGB = (0.229, 0.224, 0.225)
SCENESEG_CHANNEL_ORDERS = ("bgr", "rgb")
def _check_hw(hw: Tuple[int, int], what: str) -> Tuple[int, int]:
h, w = (int(v) for v in hw)
if h < 1 or w < 1:
raise ValueError(f"{what} must be positive, got {hw}")
return h, w
def _as_image(image: Any, dtype: Optional[type] = np.uint8) -> np.ndarray:
a = np.asarray(image)
if a.ndim not in (2, 3) or a.shape[0] < 1 or a.shape[1] < 1:
raise ValueError(f"expected an HxW or HxWxC image, got shape {a.shape}")
if dtype is not None and a.dtype != dtype:
raise ValueError(f"expected {np.dtype(dtype).name} data, got {a.dtype}")
return a
# ----------------------------------------------------------------------------------------------- INTER_LINEAR
def _linear_axis(n_src: int, n_dst: int, border_fix: bool) -> Tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
"""One axis of ``resize()``'s INTER_LINEAR tables for uint8: source indices (clipped) and integer weights."""
scale = 1.0 / (n_dst / n_src) # scale = 1. / inv_scale, inv = (double)dst / src
f = ((np.arange(n_dst, dtype=np.float64) + 0.5) * scale - 0.5).astype(F32)
s = np.floor(f).astype(np.int64) # cvFloor(fx)
f = (f - s.astype(F32)).astype(F32) # fx -= sx (float; exact)
if border_fix: # x axis only (the y loop has no such fix)
lo = s < 0
f[lo], s[lo] = F32(0), 0
hi = s >= n_src - 1
f[hi], s[hi] = F32(0), n_src - 1
one = F32(INTER_RESIZE_COEF_SCALE)
w0 = np.rint((F32(1) - f).astype(F32) * one).astype(np.int32) # saturate_cast<short>(cbuf[k] * 2048): cvRound
w1 = np.rint(f * one).astype(np.int32)
return np.clip(s, 0, n_src - 1), np.clip(s + 1, 0, n_src - 1), w0, w1
@dataclass(frozen=True)
class LinearResizeU8LUT:
"""The tables of one ``cv::resize(uint8, INTER_LINEAR)`` geometry. ``mode``: ``"copy"`` (same size),
``"area2x"`` (exact 2x down on both axes: OpenCV's INTER_AREA fast path) or ``"linear"``."""
src_hw: Tuple[int, int]
dst_hw: Tuple[int, int]
mode: str
cols0: Optional[np.ndarray] = None # (w,) source column of the first tap (clipped)
cols1: Optional[np.ndarray] = None # (w,) second tap
a0: Optional[np.ndarray] = None # (w,) int32 weights, a0 + a1 == 2048 up to OpenCV's rounding
a1: Optional[np.ndarray] = None
rows: Optional[np.ndarray] = None # (R,) the source rows the vertical pass reads (sorted, unique)
r0: Optional[np.ndarray] = None # (h,) index into ``rows`` of the first row tap
r1: Optional[np.ndarray] = None # (h,) second row tap
b0: Optional[np.ndarray] = None # (h,) int32 row weights
b1: Optional[np.ndarray] = None
def apply(self, image: Any, *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""Resize one HxW or HxWxC uint8 image of size ``src_hw`` -> ``dst_hw`` (same channel count)."""
img = _as_image(image)
if img.shape[:2] != self.src_hw:
raise ValueError(f"table built for {self.src_hw}, image is {img.shape[:2]}")
if self.mode == "copy":
res = img.copy()
elif self.mode == "area2x":
res = opencv_resize_area_u8(img, self.dst_hw)
else:
src = img if img.ndim == 3 else img[:, :, None]
sub = src[self.rows] # only the rows the vertical pass reads
h = sub[:, self.cols0, :].astype(np.int32) # horizontal pass, exact int32 (<= 255 * 2048)
h *= self.a0[None, :, None]
h1 = sub[:, self.cols1, :].astype(np.int32)
h1 *= self.a1[None, :, None]
h += h1
h >>= 4 # S[x] >> 4 (h >= 0)
v = h[self.r0] # vertical pass, the uchar VResizeLinear
v *= self.b0[:, None, None]
v >>= 16
v1 = h[self.r1]
v1 *= self.b1[:, None, None]
v1 >>= 16
v += v1
v += 2
v >>= 2
res = v.astype(np.uint8) # in [0, 255] (weights sum to 2048)
if img.ndim == 2:
res = res[:, :, 0]
if out is not None:
out[...] = res
return out
return res
def _build_linear_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> LinearResizeU8LUT:
(H, W), (h, w) = src_hw, dst_hw
if (h, w) == (H, W):
return LinearResizeU8LUT(src_hw, dst_hw, "copy")
if 2 * h == H and 2 * w == W: # is_area_fast && iscale_x == iscale_y == 2
return LinearResizeU8LUT(src_hw, dst_hw, "area2x")
c0, c1, a0, a1 = _linear_axis(W, w, True)
y0, y1, b0, b1 = _linear_axis(H, h, False)
rows = np.unique(np.concatenate([y0, y1]))
r0, r1 = np.searchsorted(rows, y0), np.searchsorted(rows, y1)
return LinearResizeU8LUT(src_hw, dst_hw, "linear", c0, c1, a0, a1, rows, r0, r1, b0, b1)
_LINEAR_U8_LUTS = LUTCache(maxsize=8)
def opencv_linear_u8_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> LinearResizeU8LUT:
"""The cached tables of ``cv::resize(uint8 src_hw -> dst_hw, INTER_LINEAR)`` (any scale, up or down)."""
key = (_check_hw(src_hw, "src_hw"), _check_hw(dst_hw, "dst_hw"))
return _LINEAR_U8_LUTS.get(key, lambda: _build_linear_lut(*key))
def opencv_resize_linear_u8(image: Any, dst_hw: Tuple[int, int], *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""``cv::resize(image, dst, Size(w, h))`` (INTER_LINEAR) of an HxW or HxWxC uint8 image, bit-exact."""
img = _as_image(image)
return opencv_linear_u8_lut(img.shape[:2], dst_hw).apply(img, out=out)
# ---------------------------------------------------------------------------------------------- INTER_NEAREST
@dataclass(frozen=True)
class NearestResizeLUT:
"""``resizeNN`` index tables: ``rows`` (h,), ``cols`` (w,); a copy when the sizes are equal."""
src_hw: Tuple[int, int]
dst_hw: Tuple[int, int]
rows: np.ndarray
cols: np.ndarray
def apply(self, image: Any, *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""Resize one HxW or HxWxC image of any dtype (a gather: values are copied, never mixed)."""
img = _as_image(image, dtype=None)
if img.shape[:2] != self.src_hw:
raise ValueError(f"table built for {self.src_hw}, image is {img.shape[:2]}")
res = img.copy() if self.src_hw == self.dst_hw else img[self.rows][:, self.cols]
if out is not None:
out[...] = res
return out
return res
def _nearest_axis(n_src: int, n_dst: int) -> np.ndarray:
inv = 1.0 / (n_dst / n_src) # ifx = 1. / fx, fx = inv_scale_x
return np.minimum(np.floor(np.arange(n_dst, dtype=np.float64) * inv).astype(np.int64), n_src - 1)
_NEAREST_LUTS = LUTCache(maxsize=8)
def opencv_nearest_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> NearestResizeLUT:
"""The cached index tables of ``cv::resize(src_hw -> dst_hw, INTER_NEAREST)``."""
key = (_check_hw(src_hw, "src_hw"), _check_hw(dst_hw, "dst_hw"))
return _NEAREST_LUTS.get(key, lambda: NearestResizeLUT(key[0], key[1], _nearest_axis(key[0][0], key[1][0]),
_nearest_axis(key[0][1], key[1][1])))
def opencv_resize_nearest(image: Any, dst_hw: Tuple[int, int], *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""``cv::resize(image, dst, Size(w, h), 0, 0, INTER_NEAREST)`` of an HxW or HxWxC image, bit-exact."""
img = _as_image(image, dtype=None)
return opencv_nearest_lut(img.shape[:2], dst_hw).apply(img, out=out)
# ------------------------------------------------------------------------------------------- SceneSeg preset
def _order(channel_order: str) -> str:
if channel_order not in SCENESEG_CHANNEL_ORDERS:
raise ValueError(f"channel_order must be one of {SCENESEG_CHANNEL_ORDERS}, got {channel_order!r}")
return channel_order
def sceneseg_resize(image_bgr: Any, *, dst_hw: Tuple[int, int] = SCENESEG_INPUT_HW,
out: Optional[np.ndarray] = None) -> np.ndarray:
"""The node's squash, ``cv::resize(input_image, resized, Size(640, 320))`` (INTER_LINEAR, aspect ratio not
kept) of one HxWx3 uint8 image (channel order kept: give it B, G, R as ``cv_bridge`` ``bgr8`` does) ->
``(320, 640, 3)`` uint8. This is the device input of the TT port (614 KB)."""
img = _as_image(image_bgr)
if img.ndim != 3 or img.shape[2] != 3:
raise ValueError(f"expected an HxWx3 BGR image, got shape {img.shape}")
return opencv_resize_linear_u8(img, dst_hw, out=out)
def sceneseg_normalize(resized_bgr: Any, *, channel_order: str = "bgr") -> np.ndarray:
"""The node's normalisation of the resized B, G, R uint8 image -> the ONNX input ``float32 [1, 3, H, W]``.
``"bgr"`` (deployed): planes B, G, R; ``convertTo(CV_32F, 1/255)`` (float32 product), ``cv::subtract`` of
(0.406, 0.456, 0.485) in float32, ``cv::divide`` by (0.225, 0.224, 0.229) in double rounded to float32.
``"rgb"`` (training order): the same arithmetic on planes R, G, B with (0.485, 0.456, 0.406) /
(0.229, 0.224, 0.225)."""
img = _as_image(resized_bgr)
if img.ndim != 3 or img.shape[2] != 3:
raise ValueError(f"expected an HxWx3 BGR image, got shape {img.shape}")
if _order(channel_order) == "rgb":
img = img[:, :, ::-1]
mean, std = SCENESEG_MEAN_RGB, SCENESEG_STD_RGB
else:
mean, std = SCENESEG_MEAN_RGB[::-1], SCENESEG_STD_RGB[::-1]
planes = np.ascontiguousarray(img.transpose(2, 0, 1)).astype(F32) # cv::split -> NCHW
x = planes * F32(1.0 / 255.0) # convertTo: float32 product
x = x - np.asarray(mean, F32)[:, None, None] # cv::subtract (float32)
x = (x.astype(np.float64) / np.asarray(std, np.float64)[:, None, None]).astype(F32) # cv::divide (double)
return x[None]
def sceneseg_preprocess(image_bgr: Any, *, channel_order: str = "bgr") -> np.ndarray:
"""One HxWx3 B, G, R uint8 camera image -> the SceneSeg ONNX input ``float32 [1, 3, 320, 640]``, bit-exact with
the deployed C++ (``"bgr"``) or its training-order variant (``"rgb"``): :func:`sceneseg_normalize` of
:func:`sceneseg_resize`."""
return sceneseg_normalize(sceneseg_resize(image_bgr), channel_order=channel_order)