File size: 6,425 Bytes
4d9b003 be62f78 4d9b003 be62f78 4d9b003 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 | # SPDX-License-Identifier: Apache-2.0
"""Input normalization of the Autoware node (``preprocessing_utils.cpp:34-84``) and the speed-limit masks.
The rule cannot be folded into the first-layer weights (SPEC 9.1, PLAN 2.12): each row of the last dimension is
normalised as ``(v - mean) / std`` in float32 unless every value of the row satisfies ``|v| < FLT_EPSILON``, in which
case the row is left untouched (padding stays zero while ``x`` has mean 10 m). ``ego_shape``,
``sampled_trajectories``, ``turn_indicators`` and ``delay`` are never normalised.
"""
from __future__ import annotations
from typing import Dict, Mapping, Tuple
import numpy as np
from ..reference import config as C
__all__ = ["FLT_EPSILON", "normalize_inputs", "normalize_array", "speed_masks"]
FLT_EPSILON = np.float32(np.finfo(np.float32).eps) # std::numeric_limits<float>::epsilon()
def _host_fast() -> bool:
"""``DIFFUSION_PLANNER_HOST_FAST`` (OPT round 5 item 3), read once: the vectorised host functions (bit-exact)
instead of the first port's ``*_ref`` versions."""
from ..tt.config import KNOBS
return bool(KNOBS.read().HOST_FAST)
HOST_FAST = _host_fast()
def normalize_array_ref(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
"""``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).
Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
branch); a zero standard deviation raises like the C++."""
data = np.array(value, dtype=np.float32, copy=True)
mean = np.asarray(mean, np.float32).reshape(-1)
std = np.asarray(std, np.float32).reshape(-1)
if mean.size != std.size:
raise ValueError("Mean and std must be same size")
cols = std.size
if cols == 0 or data.size % cols:
raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
if np.any(np.abs(std) < FLT_EPSILON):
raise ValueError("Standard deviation is zero, cannot normalize data")
rows = data.reshape(-1, cols)
zero_row = np.all(np.abs(rows) < FLT_EPSILON, axis=1)
m = np.broadcast_to(mean if mean.size > 1 else mean[:1], (cols,))
s = np.broadcast_to(std if std.size > 1 else std[:1], (cols,))
normed = ((rows - m) / s).astype(np.float32) # float32 subtract then divide, element by element
rows[~zero_row] = normed[~zero_row]
return rows.reshape(data.shape)
_NORMALIZERS: Dict[Tuple[int, int], tuple] = {}
def _normalizer(mean: np.ndarray, std: np.ndarray):
"""``(cols, m, s)`` of a normalizer, validated once per (mean, std) object pair (the weights' constants)."""
key = (id(mean), id(std))
hit = _NORMALIZERS.get(key)
if hit is not None and hit[0] is mean and hit[1] is std:
return hit[2]
m32 = np.asarray(mean, np.float32).reshape(-1)
s32 = np.asarray(std, np.float32).reshape(-1)
if m32.size != s32.size:
raise ValueError("Mean and std must be same size")
cols = s32.size
if cols and np.any(np.abs(s32) < FLT_EPSILON):
raise ValueError("Standard deviation is zero, cannot normalize data")
m = np.ascontiguousarray(np.broadcast_to(m32 if m32.size > 1 else m32[:1], (cols,)))
s = np.ascontiguousarray(np.broadcast_to(s32 if s32.size > 1 else s32[:1], (cols,)))
out = (cols, m, s, np.ones(cols, np.uint8) if cols < 256 else None)
if len(_NORMALIZERS) > 256:
_NORMALIZERS.clear()
_NORMALIZERS[key] = (mean, std, out)
return out
def normalize_array(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
"""``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).
Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
branch); a zero standard deviation raises like the C++. Only the rows that are not all-small are computed
(``(v - mean) / std`` in float32, element by element); the others are copied unchanged."""
if not HOST_FAST:
return normalize_array_ref(value, mean, std)
data = np.asarray(value, dtype=np.float32) # not modified: the result is a new array
cols, m, s, ones = _normalizer(mean, std)
if cols == 0 or data.size % cols:
raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
rows = data.reshape(-1, cols)
big = ~(np.abs(rows) < FLT_EPSILON) # NaN counts as not small, as in the C++ test
if ones is not None: # per-row count of the non-small values: a uint8 dot product
keep = (big.view(np.uint8) @ ones) != 0 # (faster than a short-axis any())
else:
keep = big.any(axis=1)
idx = np.flatnonzero(keep)
if 2 * idx.size > keep.size: # mostly valid rows: compute all, keep the small rows
return np.where(keep[:, None], (rows - m) / s, rows).reshape(data.shape)
out = rows.copy()
out[idx] = (rows[idx] - m) / s
return out.reshape(data.shape)
def normalize_inputs(raw: Mapping[str, np.ndarray],
observation: Mapping[str, Tuple[np.ndarray, np.ndarray]]) -> Dict[str, np.ndarray]:
"""``normalize_input_data(input_data_map, normalization_map)``: every key except the four skipped ones must have
a normalizer (``Missing key ... from normalization map`` otherwise)."""
out: Dict[str, np.ndarray] = {}
for key, value in raw.items():
if key in C.SKIP_NORMALIZATION:
out[key] = np.array(value, dtype=np.float32, copy=True)
continue
if key not in observation:
raise KeyError(f"Missing key {key} from normalization map")
mean, std = observation[key]
out[key] = normalize_array(value, mean, std)
return out
def speed_masks(norm_inputs: Mapping[str, np.ndarray]) -> Dict[str, np.ndarray]:
"""``lanes_has_speed_limit`` / ``route_lanes_has_speed_limit`` of the TensorRT path: normalised speed limit
``> FLT_EPSILON`` (``inference/utils.hpp:112-123``; the ORT backend uses ``> 0``,
``onnxruntime_inference.cpp:41-48``; Autoware's default backend is TensorRT)."""
return {"lanes_has_speed_limit": np.asarray(norm_inputs["lanes_speed_limit"]) > FLT_EPSILON,
"route_lanes_has_speed_limit": np.asarray(norm_inputs["route_lanes_speed_limit"]) > FLT_EPSILON}
|