# SPDX-License-Identifier: Apache-2.0 """Input normalization of the Autoware node (``preprocessing_utils.cpp:34-84``) and the speed-limit masks. The rule cannot be folded into the first-layer weights (SPEC 9.1, PLAN 2.12): each row of the last dimension is normalised as ``(v - mean) / std`` in float32 unless every value of the row satisfies ``|v| < FLT_EPSILON``, in which case the row is left untouched (padding stays zero while ``x`` has mean 10 m). ``ego_shape``, ``sampled_trajectories``, ``turn_indicators`` and ``delay`` are never normalised. """ from __future__ import annotations from typing import Dict, Mapping, Tuple import numpy as np from ..reference import config as C __all__ = ["FLT_EPSILON", "normalize_inputs", "normalize_array", "speed_masks"] FLT_EPSILON = np.float32(np.finfo(np.float32).eps) # std::numeric_limits::epsilon() def _host_fast() -> bool: """``DIFFUSION_PLANNER_HOST_FAST`` (OPT round 5 item 3), read once: the vectorised host functions (bit-exact) instead of the first port's ``*_ref`` versions.""" from ..tt.config import KNOBS return bool(KNOBS.read().HOST_FAST) HOST_FAST = _host_fast() def normalize_array_ref(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray: """``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array). Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1`` branch); a zero standard deviation raises like the C++.""" data = np.array(value, dtype=np.float32, copy=True) mean = np.asarray(mean, np.float32).reshape(-1) std = np.asarray(std, np.float32).reshape(-1) if mean.size != std.size: raise ValueError("Mean and std must be same size") cols = std.size if cols == 0 or data.size % cols: raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}") if np.any(np.abs(std) < FLT_EPSILON): raise ValueError("Standard deviation is zero, cannot normalize data") rows = data.reshape(-1, cols) zero_row = np.all(np.abs(rows) < FLT_EPSILON, axis=1) m = np.broadcast_to(mean if mean.size > 1 else mean[:1], (cols,)) s = np.broadcast_to(std if std.size > 1 else std[:1], (cols,)) normed = ((rows - m) / s).astype(np.float32) # float32 subtract then divide, element by element rows[~zero_row] = normed[~zero_row] return rows.reshape(data.shape) _NORMALIZERS: Dict[Tuple[int, int], tuple] = {} def _normalizer(mean: np.ndarray, std: np.ndarray): """``(cols, m, s)`` of a normalizer, validated once per (mean, std) object pair (the weights' constants).""" key = (id(mean), id(std)) hit = _NORMALIZERS.get(key) if hit is not None and hit[0] is mean and hit[1] is std: return hit[2] m32 = np.asarray(mean, np.float32).reshape(-1) s32 = np.asarray(std, np.float32).reshape(-1) if m32.size != s32.size: raise ValueError("Mean and std must be same size") cols = s32.size if cols and np.any(np.abs(s32) < FLT_EPSILON): raise ValueError("Standard deviation is zero, cannot normalize data") m = np.ascontiguousarray(np.broadcast_to(m32 if m32.size > 1 else m32[:1], (cols,))) s = np.ascontiguousarray(np.broadcast_to(s32 if s32.size > 1 else s32[:1], (cols,))) out = (cols, m, s, np.ones(cols, np.uint8) if cols < 256 else None) if len(_NORMALIZERS) > 256: _NORMALIZERS.clear() _NORMALIZERS[key] = (mean, std, out) return out def normalize_array(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray: """``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array). Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1`` branch); a zero standard deviation raises like the C++. Only the rows that are not all-small are computed (``(v - mean) / std`` in float32, element by element); the others are copied unchanged.""" if not HOST_FAST: return normalize_array_ref(value, mean, std) data = np.asarray(value, dtype=np.float32) # not modified: the result is a new array cols, m, s, ones = _normalizer(mean, std) if cols == 0 or data.size % cols: raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}") rows = data.reshape(-1, cols) big = ~(np.abs(rows) < FLT_EPSILON) # NaN counts as not small, as in the C++ test if ones is not None: # per-row count of the non-small values: a uint8 dot product keep = (big.view(np.uint8) @ ones) != 0 # (faster than a short-axis any()) else: keep = big.any(axis=1) idx = np.flatnonzero(keep) if 2 * idx.size > keep.size: # mostly valid rows: compute all, keep the small rows return np.where(keep[:, None], (rows - m) / s, rows).reshape(data.shape) out = rows.copy() out[idx] = (rows[idx] - m) / s return out.reshape(data.shape) def normalize_inputs(raw: Mapping[str, np.ndarray], observation: Mapping[str, Tuple[np.ndarray, np.ndarray]]) -> Dict[str, np.ndarray]: """``normalize_input_data(input_data_map, normalization_map)``: every key except the four skipped ones must have a normalizer (``Missing key ... from normalization map`` otherwise).""" out: Dict[str, np.ndarray] = {} for key, value in raw.items(): if key in C.SKIP_NORMALIZATION: out[key] = np.array(value, dtype=np.float32, copy=True) continue if key not in observation: raise KeyError(f"Missing key {key} from normalization map") mean, std = observation[key] out[key] = normalize_array(value, mean, std) return out def speed_masks(norm_inputs: Mapping[str, np.ndarray]) -> Dict[str, np.ndarray]: """``lanes_has_speed_limit`` / ``route_lanes_has_speed_limit`` of the TensorRT path: normalised speed limit ``> FLT_EPSILON`` (``inference/utils.hpp:112-123``; the ORT backend uses ``> 0``, ``onnxruntime_inference.cpp:41-48``; Autoware's default backend is TensorRT).""" return {"lanes_has_speed_limit": np.asarray(norm_inputs["lanes_speed_limit"]) > FLT_EPSILON, "route_lanes_has_speed_limit": np.asarray(norm_inputs["route_lanes_speed_limit"]) > FLT_EPSILON}