File size: 15,812 Bytes
4d9b003 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 | # SPDX-License-Identifier: Apache-2.0
"""C14 image pre-processing, SceneSeg preset: bit-exact host emulations of OpenCV's ``cv::resize`` on uint8 images
with ``INTER_LINEAR`` (the default interpolation) and ``INTER_NEAREST``, driven by cached per-size tables, and the
pre-processing of Autoware VisionPilot's SceneSeg node built on them. numpy only; no ttnn, no torch, no OpenCV.
The SceneSeg node (VisionPilot ``middleware_recipes/common/backends/onnx_runtime_backend.cpp:41-60`` and the identical
``tensorrt_backend.cpp:160-177`` at autowarefoundation/vision_pilot ``04fa3e80``; research/sceneseg/SPEC.md section 3)
takes the ``bgr8`` image of ``cv_bridge``, then::
cv::resize(input_image, resized_image, cv::Size(640, 320)); // INTER_LINEAR, aspect ratio not kept
resized_image.convertTo(float_image, CV_32FC3, 1.0 / 255.0);
cv::subtract(float_image, cv::Scalar(0.406, 0.456, 0.485), float_image);
cv::divide(float_image, cv::Scalar(0.225, 0.224, 0.229), float_image);
cv::split(float_image, channels); // planes B, G, R -> NCHW
and publishes the argmax mask resized back to the source size with ``cv::resize(..., INTER_NEAREST)``
(``run_model_node.cpp:117-188``). :func:`sceneseg_resize` / :func:`sceneseg_normalize` / :func:`sceneseg_preprocess`
reproduce the first part, :func:`opencv_resize_nearest` the mask resize.
What OpenCV computes (``modules/imgproc/src/resize.cpp`` and ``core/src/arithm.cpp``; identical in 4.8.1 and 4.11.0
on this x86-64 host, measured; the IPP path declines both resizes: uint8 ``ippLinear`` and ``ippNearest`` need
``useIPP_NotExact``):
- ``cv::resize`` with ``dsize == ssize`` copies. ``inv_scale = (double)dst / src`` and ``scale = 1.0 / inv_scale``
per axis.
- **INTER_LINEAR, exactly 2x down on both axes**: OpenCV switches to ``INTER_AREA`` (``resizeAreaFast_``), which
:func:`.image_area.opencv_resize_area_u8` emulates (2x2 ``fast_mode``: ``(s + 2) >> 2``).
- **INTER_LINEAR otherwise** (``resizeGeneric_`` with ``HResizeLinear<uchar, int, short, 2048>`` and the uchar
specialisation of ``VResizeLinear``): per output column ``fx = (float)((dx + 0.5) * scale - 0.5)``,
``sx = floor(fx)``, ``fx -= sx`` (float); at the borders ``sx < 0 -> sx = 0, fx = 0`` and ``sx >= W - 1 ->
sx = W - 1, fx = 0``; integer weights ``a0 = cvRound((1.f - fx) * 2048)``, ``a1 = cvRound(fx * 2048)`` (round half
to even). Rows take the same coefficients **without** the border fix (the row indices are clipped to [0, H - 1]
instead, ``clip(sy + k, 0, H)``). The horizontal pass is exact integer arithmetic,
``h = S[sx] * a0 + S[sx + 1] * a1`` (int32), and the vertical pass is the fixed-point
``(((b0 * (h0 >> 4)) >> 16) + ((b1 * (h1 >> 4)) >> 16) + 2) >> 2`` that the SIMD and the scalar code share.
- **INTER_NEAREST** (``resizeNN``): ``sx = min(floor(dx * (1.0 / inv_scale_x)), W - 1)`` in double, rows alike.
- **Normalisation**: ``convertTo(CV_32F, 1/255)`` is one float32 product ``x * (float)(1/255)``; ``cv::subtract``
with a ``Scalar`` runs in float32 (``x - (float)mean_c``); ``cv::divide`` with a ``Scalar`` runs in double and
rounds once to float32 (``(float)((double)x / std_c)``), measured with OpenCV's own scalar arithmetic. A float32
division instead differs by one ulp in about 27 % of the values (the research reference's arithmetic).
Channel orders (SPEC section 3, PLAN.md D10): ``"bgr"`` is the deployed path: B, G, R planes normalised with the
BGR-ordered ImageNet constants, fed to a network trained on R, G, B. ``"rgb"`` is the training-consistent option: the
same arithmetic on R, G, B planes with the RGB constants (an exact channel permutation of the image first, as a
``cv_bridge`` ``rgb8`` conversion would give). Every physical colour gets the same mean and std in both orders, so
the two inputs are channel permutations of each other.
Verified bit-exact against cv2 4.8.1 (tt-env) and 4.11.0 (research venv) on random and real images of many sizes
(down- and up-scaling, odd sizes, 1-4 channels), against a scalar transliteration of ``resize.cpp``, and against the
research golden inputs of the SceneSeg port (``common/tests/host/test_image_sceneseg_host.py``).
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Any, Optional, Tuple
import numpy as np
from .image import LUTCache
from .image_area import opencv_resize_area_u8
__all__ = [
"LinearResizeU8LUT",
"opencv_linear_u8_lut",
"opencv_resize_linear_u8",
"NearestResizeLUT",
"opencv_nearest_lut",
"opencv_resize_nearest",
"sceneseg_resize",
"sceneseg_normalize",
"sceneseg_preprocess",
"SCENESEG_INPUT_HW",
"SCENESEG_MEAN_RGB",
"SCENESEG_STD_RGB",
"SCENESEG_CHANNEL_ORDERS",
]
F32 = np.float32
INTER_RESIZE_COEF_SCALE = 2048 # 1 << INTER_RESIZE_COEF_BITS (11), resize.cpp
SCENESEG_INPUT_HW = (320, 640) # (H, W) of the ONNX input; fixed by the SceneContext reshape (SPEC 4.2)
SCENESEG_MEAN_RGB = (0.485, 0.456, 0.406) # ImageNet, R G B (the C++ code lists them reversed for B G R planes)
SCENESEG_STD_RGB = (0.229, 0.224, 0.225)
SCENESEG_CHANNEL_ORDERS = ("bgr", "rgb")
def _check_hw(hw: Tuple[int, int], what: str) -> Tuple[int, int]:
h, w = (int(v) for v in hw)
if h < 1 or w < 1:
raise ValueError(f"{what} must be positive, got {hw}")
return h, w
def _as_image(image: Any, dtype: Optional[type] = np.uint8) -> np.ndarray:
a = np.asarray(image)
if a.ndim not in (2, 3) or a.shape[0] < 1 or a.shape[1] < 1:
raise ValueError(f"expected an HxW or HxWxC image, got shape {a.shape}")
if dtype is not None and a.dtype != dtype:
raise ValueError(f"expected {np.dtype(dtype).name} data, got {a.dtype}")
return a
# ----------------------------------------------------------------------------------------------- INTER_LINEAR
def _linear_axis(n_src: int, n_dst: int, border_fix: bool) -> Tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
"""One axis of ``resize()``'s INTER_LINEAR tables for uint8: source indices (clipped) and integer weights."""
scale = 1.0 / (n_dst / n_src) # scale = 1. / inv_scale, inv = (double)dst / src
f = ((np.arange(n_dst, dtype=np.float64) + 0.5) * scale - 0.5).astype(F32)
s = np.floor(f).astype(np.int64) # cvFloor(fx)
f = (f - s.astype(F32)).astype(F32) # fx -= sx (float; exact)
if border_fix: # x axis only (the y loop has no such fix)
lo = s < 0
f[lo], s[lo] = F32(0), 0
hi = s >= n_src - 1
f[hi], s[hi] = F32(0), n_src - 1
one = F32(INTER_RESIZE_COEF_SCALE)
w0 = np.rint((F32(1) - f).astype(F32) * one).astype(np.int32) # saturate_cast<short>(cbuf[k] * 2048): cvRound
w1 = np.rint(f * one).astype(np.int32)
return np.clip(s, 0, n_src - 1), np.clip(s + 1, 0, n_src - 1), w0, w1
@dataclass(frozen=True)
class LinearResizeU8LUT:
"""The tables of one ``cv::resize(uint8, INTER_LINEAR)`` geometry. ``mode``: ``"copy"`` (same size),
``"area2x"`` (exact 2x down on both axes: OpenCV's INTER_AREA fast path) or ``"linear"``."""
src_hw: Tuple[int, int]
dst_hw: Tuple[int, int]
mode: str
cols0: Optional[np.ndarray] = None # (w,) source column of the first tap (clipped)
cols1: Optional[np.ndarray] = None # (w,) second tap
a0: Optional[np.ndarray] = None # (w,) int32 weights, a0 + a1 == 2048 up to OpenCV's rounding
a1: Optional[np.ndarray] = None
rows: Optional[np.ndarray] = None # (R,) the source rows the vertical pass reads (sorted, unique)
r0: Optional[np.ndarray] = None # (h,) index into ``rows`` of the first row tap
r1: Optional[np.ndarray] = None # (h,) second row tap
b0: Optional[np.ndarray] = None # (h,) int32 row weights
b1: Optional[np.ndarray] = None
def apply(self, image: Any, *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""Resize one HxW or HxWxC uint8 image of size ``src_hw`` -> ``dst_hw`` (same channel count)."""
img = _as_image(image)
if img.shape[:2] != self.src_hw:
raise ValueError(f"table built for {self.src_hw}, image is {img.shape[:2]}")
if self.mode == "copy":
res = img.copy()
elif self.mode == "area2x":
res = opencv_resize_area_u8(img, self.dst_hw)
else:
src = img if img.ndim == 3 else img[:, :, None]
sub = src[self.rows] # only the rows the vertical pass reads
h = sub[:, self.cols0, :].astype(np.int32) # horizontal pass, exact int32 (<= 255 * 2048)
h *= self.a0[None, :, None]
h1 = sub[:, self.cols1, :].astype(np.int32)
h1 *= self.a1[None, :, None]
h += h1
h >>= 4 # S[x] >> 4 (h >= 0)
v = h[self.r0] # vertical pass, the uchar VResizeLinear
v *= self.b0[:, None, None]
v >>= 16
v1 = h[self.r1]
v1 *= self.b1[:, None, None]
v1 >>= 16
v += v1
v += 2
v >>= 2
res = v.astype(np.uint8) # in [0, 255] (weights sum to 2048)
if img.ndim == 2:
res = res[:, :, 0]
if out is not None:
out[...] = res
return out
return res
def _build_linear_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> LinearResizeU8LUT:
(H, W), (h, w) = src_hw, dst_hw
if (h, w) == (H, W):
return LinearResizeU8LUT(src_hw, dst_hw, "copy")
if 2 * h == H and 2 * w == W: # is_area_fast && iscale_x == iscale_y == 2
return LinearResizeU8LUT(src_hw, dst_hw, "area2x")
c0, c1, a0, a1 = _linear_axis(W, w, True)
y0, y1, b0, b1 = _linear_axis(H, h, False)
rows = np.unique(np.concatenate([y0, y1]))
r0, r1 = np.searchsorted(rows, y0), np.searchsorted(rows, y1)
return LinearResizeU8LUT(src_hw, dst_hw, "linear", c0, c1, a0, a1, rows, r0, r1, b0, b1)
_LINEAR_U8_LUTS = LUTCache(maxsize=8)
def opencv_linear_u8_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> LinearResizeU8LUT:
"""The cached tables of ``cv::resize(uint8 src_hw -> dst_hw, INTER_LINEAR)`` (any scale, up or down)."""
key = (_check_hw(src_hw, "src_hw"), _check_hw(dst_hw, "dst_hw"))
return _LINEAR_U8_LUTS.get(key, lambda: _build_linear_lut(*key))
def opencv_resize_linear_u8(image: Any, dst_hw: Tuple[int, int], *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""``cv::resize(image, dst, Size(w, h))`` (INTER_LINEAR) of an HxW or HxWxC uint8 image, bit-exact."""
img = _as_image(image)
return opencv_linear_u8_lut(img.shape[:2], dst_hw).apply(img, out=out)
# ---------------------------------------------------------------------------------------------- INTER_NEAREST
@dataclass(frozen=True)
class NearestResizeLUT:
"""``resizeNN`` index tables: ``rows`` (h,), ``cols`` (w,); a copy when the sizes are equal."""
src_hw: Tuple[int, int]
dst_hw: Tuple[int, int]
rows: np.ndarray
cols: np.ndarray
def apply(self, image: Any, *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""Resize one HxW or HxWxC image of any dtype (a gather: values are copied, never mixed)."""
img = _as_image(image, dtype=None)
if img.shape[:2] != self.src_hw:
raise ValueError(f"table built for {self.src_hw}, image is {img.shape[:2]}")
res = img.copy() if self.src_hw == self.dst_hw else img[self.rows][:, self.cols]
if out is not None:
out[...] = res
return out
return res
def _nearest_axis(n_src: int, n_dst: int) -> np.ndarray:
inv = 1.0 / (n_dst / n_src) # ifx = 1. / fx, fx = inv_scale_x
return np.minimum(np.floor(np.arange(n_dst, dtype=np.float64) * inv).astype(np.int64), n_src - 1)
_NEAREST_LUTS = LUTCache(maxsize=8)
def opencv_nearest_lut(src_hw: Tuple[int, int], dst_hw: Tuple[int, int]) -> NearestResizeLUT:
"""The cached index tables of ``cv::resize(src_hw -> dst_hw, INTER_NEAREST)``."""
key = (_check_hw(src_hw, "src_hw"), _check_hw(dst_hw, "dst_hw"))
return _NEAREST_LUTS.get(key, lambda: NearestResizeLUT(key[0], key[1], _nearest_axis(key[0][0], key[1][0]),
_nearest_axis(key[0][1], key[1][1])))
def opencv_resize_nearest(image: Any, dst_hw: Tuple[int, int], *, out: Optional[np.ndarray] = None) -> np.ndarray:
"""``cv::resize(image, dst, Size(w, h), 0, 0, INTER_NEAREST)`` of an HxW or HxWxC image, bit-exact."""
img = _as_image(image, dtype=None)
return opencv_nearest_lut(img.shape[:2], dst_hw).apply(img, out=out)
# ------------------------------------------------------------------------------------------- SceneSeg preset
def _order(channel_order: str) -> str:
if channel_order not in SCENESEG_CHANNEL_ORDERS:
raise ValueError(f"channel_order must be one of {SCENESEG_CHANNEL_ORDERS}, got {channel_order!r}")
return channel_order
def sceneseg_resize(image_bgr: Any, *, dst_hw: Tuple[int, int] = SCENESEG_INPUT_HW,
out: Optional[np.ndarray] = None) -> np.ndarray:
"""The node's squash, ``cv::resize(input_image, resized, Size(640, 320))`` (INTER_LINEAR, aspect ratio not
kept) of one HxWx3 uint8 image (channel order kept: give it B, G, R as ``cv_bridge`` ``bgr8`` does) ->
``(320, 640, 3)`` uint8. This is the device input of the TT port (614 KB)."""
img = _as_image(image_bgr)
if img.ndim != 3 or img.shape[2] != 3:
raise ValueError(f"expected an HxWx3 BGR image, got shape {img.shape}")
return opencv_resize_linear_u8(img, dst_hw, out=out)
def sceneseg_normalize(resized_bgr: Any, *, channel_order: str = "bgr") -> np.ndarray:
"""The node's normalisation of the resized B, G, R uint8 image -> the ONNX input ``float32 [1, 3, H, W]``.
``"bgr"`` (deployed): planes B, G, R; ``convertTo(CV_32F, 1/255)`` (float32 product), ``cv::subtract`` of
(0.406, 0.456, 0.485) in float32, ``cv::divide`` by (0.225, 0.224, 0.229) in double rounded to float32.
``"rgb"`` (training order): the same arithmetic on planes R, G, B with (0.485, 0.456, 0.406) /
(0.229, 0.224, 0.225)."""
img = _as_image(resized_bgr)
if img.ndim != 3 or img.shape[2] != 3:
raise ValueError(f"expected an HxWx3 BGR image, got shape {img.shape}")
if _order(channel_order) == "rgb":
img = img[:, :, ::-1]
mean, std = SCENESEG_MEAN_RGB, SCENESEG_STD_RGB
else:
mean, std = SCENESEG_MEAN_RGB[::-1], SCENESEG_STD_RGB[::-1]
planes = np.ascontiguousarray(img.transpose(2, 0, 1)).astype(F32) # cv::split -> NCHW
x = planes * F32(1.0 / 255.0) # convertTo: float32 product
x = x - np.asarray(mean, F32)[:, None, None] # cv::subtract (float32)
x = (x.astype(np.float64) / np.asarray(std, np.float64)[:, None, None]).astype(F32) # cv::divide (double)
return x[None]
def sceneseg_preprocess(image_bgr: Any, *, channel_order: str = "bgr") -> np.ndarray:
"""One HxWx3 B, G, R uint8 camera image -> the SceneSeg ONNX input ``float32 [1, 3, 320, 640]``, bit-exact with
the deployed C++ (``"bgr"``) or its training-order variant (``"rgb"``): :func:`sceneseg_normalize` of
:func:`sceneseg_resize`."""
return sceneseg_normalize(sceneseg_resize(image_bgr), channel_order=channel_order)
|