changh95's picture
tt-model push diffusion-planner-p150 (container)
be62f78 verified
Raw History Blame Contribute Delete
6.43 kB
# SPDX-License-Identifier: Apache-2.0
"""Input normalization of the Autoware node (``preprocessing_utils.cpp:34-84``) and the speed-limit masks.
The rule cannot be folded into the first-layer weights (SPEC 9.1, PLAN 2.12): each row of the last dimension is
normalised as ``(v - mean) / std`` in float32 unless every value of the row satisfies ``|v| < FLT_EPSILON``, in which
case the row is left untouched (padding stays zero while ``x`` has mean 10 m). ``ego_shape``,
``sampled_trajectories``, ``turn_indicators`` and ``delay`` are never normalised.
"""
from __future__ import annotations
from typing import Dict, Mapping, Tuple
import numpy as np
from ..reference import config as C
__all__ = ["FLT_EPSILON", "normalize_inputs", "normalize_array", "speed_masks"]
FLT_EPSILON = np.float32(np.finfo(np.float32).eps) # std::numeric_limits<float>::epsilon()
def _host_fast() -> bool:
"""``DIFFUSION_PLANNER_HOST_FAST`` (OPT round 5 item 3), read once: the vectorised host functions (bit-exact)
instead of the first port's ``*_ref`` versions."""
from ..tt.config import KNOBS
return bool(KNOBS.read().HOST_FAST)
HOST_FAST = _host_fast()
def normalize_array_ref(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
"""``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).
Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
branch); a zero standard deviation raises like the C++."""
data = np.array(value, dtype=np.float32, copy=True)
mean = np.asarray(mean, np.float32).reshape(-1)
std = np.asarray(std, np.float32).reshape(-1)
if mean.size != std.size:
raise ValueError("Mean and std must be same size")
cols = std.size
if cols == 0 or data.size % cols:
raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
if np.any(np.abs(std) < FLT_EPSILON):
raise ValueError("Standard deviation is zero, cannot normalize data")
rows = data.reshape(-1, cols)
zero_row = np.all(np.abs(rows) < FLT_EPSILON, axis=1)
m = np.broadcast_to(mean if mean.size > 1 else mean[:1], (cols,))
s = np.broadcast_to(std if std.size > 1 else std[:1], (cols,))
normed = ((rows - m) / s).astype(np.float32) # float32 subtract then divide, element by element
rows[~zero_row] = normed[~zero_row]
return rows.reshape(data.shape)
_NORMALIZERS: Dict[Tuple[int, int], tuple] = {}
def _normalizer(mean: np.ndarray, std: np.ndarray):
"""``(cols, m, s)`` of a normalizer, validated once per (mean, std) object pair (the weights' constants)."""
key = (id(mean), id(std))
hit = _NORMALIZERS.get(key)
if hit is not None and hit[0] is mean and hit[1] is std:
return hit[2]
m32 = np.asarray(mean, np.float32).reshape(-1)
s32 = np.asarray(std, np.float32).reshape(-1)
if m32.size != s32.size:
raise ValueError("Mean and std must be same size")
cols = s32.size
if cols and np.any(np.abs(s32) < FLT_EPSILON):
raise ValueError("Standard deviation is zero, cannot normalize data")
m = np.ascontiguousarray(np.broadcast_to(m32 if m32.size > 1 else m32[:1], (cols,)))
s = np.ascontiguousarray(np.broadcast_to(s32 if s32.size > 1 else s32[:1], (cols,)))
out = (cols, m, s, np.ones(cols, np.uint8) if cols < 256 else None)
if len(_NORMALIZERS) > 256:
_NORMALIZERS.clear()
_NORMALIZERS[key] = (mean, std, out)
return out
def normalize_array(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
"""``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).
Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
branch); a zero standard deviation raises like the C++. Only the rows that are not all-small are computed
(``(v - mean) / std`` in float32, element by element); the others are copied unchanged."""
if not HOST_FAST:
return normalize_array_ref(value, mean, std)
data = np.asarray(value, dtype=np.float32) # not modified: the result is a new array
cols, m, s, ones = _normalizer(mean, std)
if cols == 0 or data.size % cols:
raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
rows = data.reshape(-1, cols)
big = ~(np.abs(rows) < FLT_EPSILON) # NaN counts as not small, as in the C++ test
if ones is not None: # per-row count of the non-small values: a uint8 dot product
keep = (big.view(np.uint8) @ ones) != 0 # (faster than a short-axis any())
else:
keep = big.any(axis=1)
idx = np.flatnonzero(keep)
if 2 * idx.size > keep.size: # mostly valid rows: compute all, keep the small rows
return np.where(keep[:, None], (rows - m) / s, rows).reshape(data.shape)
out = rows.copy()
out[idx] = (rows[idx] - m) / s
return out.reshape(data.shape)
def normalize_inputs(raw: Mapping[str, np.ndarray],
observation: Mapping[str, Tuple[np.ndarray, np.ndarray]]) -> Dict[str, np.ndarray]:
"""``normalize_input_data(input_data_map, normalization_map)``: every key except the four skipped ones must have
a normalizer (``Missing key ... from normalization map`` otherwise)."""
out: Dict[str, np.ndarray] = {}
for key, value in raw.items():
if key in C.SKIP_NORMALIZATION:
out[key] = np.array(value, dtype=np.float32, copy=True)
continue
if key not in observation:
raise KeyError(f"Missing key {key} from normalization map")
mean, std = observation[key]
out[key] = normalize_array(value, mean, std)
return out
def speed_masks(norm_inputs: Mapping[str, np.ndarray]) -> Dict[str, np.ndarray]:
"""``lanes_has_speed_limit`` / ``route_lanes_has_speed_limit`` of the TensorRT path: normalised speed limit
``> FLT_EPSILON`` (``inference/utils.hpp:112-123``; the ORT backend uses ``> 0``,
``onnxruntime_inference.cpp:41-48``; Autoware's default backend is TensorRT)."""
return {"lanes_has_speed_limit": np.asarray(norm_inputs["lanes_speed_limit"]) > FLT_EPSILON,
"route_lanes_has_speed_limit": np.asarray(norm_inputs["route_lanes_speed_limit"]) > FLT_EPSILON}