File size: 6,425 Bytes
4d9b003
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
be62f78
 
 
 
 
 
 
 
 
 
 
 
4d9b003
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
be62f78
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4d9b003
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
# SPDX-License-Identifier: Apache-2.0
"""Input normalization of the Autoware node (``preprocessing_utils.cpp:34-84``) and the speed-limit masks.

The rule cannot be folded into the first-layer weights (SPEC 9.1, PLAN 2.12): each row of the last dimension is
normalised as ``(v - mean) / std`` in float32 unless every value of the row satisfies ``|v| < FLT_EPSILON``, in which
case the row is left untouched (padding stays zero while ``x`` has mean 10 m). ``ego_shape``,
``sampled_trajectories``, ``turn_indicators`` and ``delay`` are never normalised.
"""
from __future__ import annotations

from typing import Dict, Mapping, Tuple

import numpy as np

from ..reference import config as C

__all__ = ["FLT_EPSILON", "normalize_inputs", "normalize_array", "speed_masks"]

FLT_EPSILON = np.float32(np.finfo(np.float32).eps)  # std::numeric_limits<float>::epsilon()


def _host_fast() -> bool:
    """``DIFFUSION_PLANNER_HOST_FAST`` (OPT round 5 item 3), read once: the vectorised host functions (bit-exact)
    instead of the first port's ``*_ref`` versions."""
    from ..tt.config import KNOBS

    return bool(KNOBS.read().HOST_FAST)


HOST_FAST = _host_fast()


def normalize_array_ref(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
    """``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).

    Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
    branch); a zero standard deviation raises like the C++."""
    data = np.array(value, dtype=np.float32, copy=True)
    mean = np.asarray(mean, np.float32).reshape(-1)
    std = np.asarray(std, np.float32).reshape(-1)
    if mean.size != std.size:
        raise ValueError("Mean and std must be same size")
    cols = std.size
    if cols == 0 or data.size % cols:
        raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
    if np.any(np.abs(std) < FLT_EPSILON):
        raise ValueError("Standard deviation is zero, cannot normalize data")
    rows = data.reshape(-1, cols)
    zero_row = np.all(np.abs(rows) < FLT_EPSILON, axis=1)
    m = np.broadcast_to(mean if mean.size > 1 else mean[:1], (cols,))
    s = np.broadcast_to(std if std.size > 1 else std[:1], (cols,))
    normed = ((rows - m) / s).astype(np.float32)  # float32 subtract then divide, element by element
    rows[~zero_row] = normed[~zero_row]
    return rows.reshape(data.shape)


_NORMALIZERS: Dict[Tuple[int, int], tuple] = {}


def _normalizer(mean: np.ndarray, std: np.ndarray):
    """``(cols, m, s)`` of a normalizer, validated once per (mean, std) object pair (the weights' constants)."""
    key = (id(mean), id(std))
    hit = _NORMALIZERS.get(key)
    if hit is not None and hit[0] is mean and hit[1] is std:
        return hit[2]
    m32 = np.asarray(mean, np.float32).reshape(-1)
    s32 = np.asarray(std, np.float32).reshape(-1)
    if m32.size != s32.size:
        raise ValueError("Mean and std must be same size")
    cols = s32.size
    if cols and np.any(np.abs(s32) < FLT_EPSILON):
        raise ValueError("Standard deviation is zero, cannot normalize data")
    m = np.ascontiguousarray(np.broadcast_to(m32 if m32.size > 1 else m32[:1], (cols,)))
    s = np.ascontiguousarray(np.broadcast_to(s32 if s32.size > 1 else s32[:1], (cols,)))
    out = (cols, m, s, np.ones(cols, np.uint8) if cols < 256 else None)
    if len(_NORMALIZERS) > 256:
        _NORMALIZERS.clear()
    _NORMALIZERS[key] = (mean, std, out)
    return out


def normalize_array(value: np.ndarray, mean: np.ndarray, std: np.ndarray) -> np.ndarray:
    """``normalize_vector`` of ``preprocessing_utils.cpp:36-63`` on one tensor (a new float32 array).

    Rows have ``std.size`` columns; a single-value mean / std broadcasts over the columns (the C++ ``mean.size() == 1``
    branch); a zero standard deviation raises like the C++. Only the rows that are not all-small are computed
    (``(v - mean) / std`` in float32, element by element); the others are copied unchanged."""
    if not HOST_FAST:
        return normalize_array_ref(value, mean, std)
    data = np.asarray(value, dtype=np.float32)   # not modified: the result is a new array
    cols, m, s, ones = _normalizer(mean, std)
    if cols == 0 or data.size % cols:
        raise ValueError(f"data size {data.size} is not divisible by the normalizer size {cols}")
    rows = data.reshape(-1, cols)
    big = ~(np.abs(rows) < FLT_EPSILON)             # NaN counts as not small, as in the C++ test
    if ones is not None:                            # per-row count of the non-small values: a uint8 dot product
        keep = (big.view(np.uint8) @ ones) != 0     # (faster than a short-axis any())
    else:
        keep = big.any(axis=1)
    idx = np.flatnonzero(keep)
    if 2 * idx.size > keep.size:                    # mostly valid rows: compute all, keep the small rows
        return np.where(keep[:, None], (rows - m) / s, rows).reshape(data.shape)
    out = rows.copy()
    out[idx] = (rows[idx] - m) / s
    return out.reshape(data.shape)


def normalize_inputs(raw: Mapping[str, np.ndarray],
                     observation: Mapping[str, Tuple[np.ndarray, np.ndarray]]) -> Dict[str, np.ndarray]:
    """``normalize_input_data(input_data_map, normalization_map)``: every key except the four skipped ones must have
    a normalizer (``Missing key ... from normalization map`` otherwise)."""
    out: Dict[str, np.ndarray] = {}
    for key, value in raw.items():
        if key in C.SKIP_NORMALIZATION:
            out[key] = np.array(value, dtype=np.float32, copy=True)
            continue
        if key not in observation:
            raise KeyError(f"Missing key {key} from normalization map")
        mean, std = observation[key]
        out[key] = normalize_array(value, mean, std)
    return out


def speed_masks(norm_inputs: Mapping[str, np.ndarray]) -> Dict[str, np.ndarray]:
    """``lanes_has_speed_limit`` / ``route_lanes_has_speed_limit`` of the TensorRT path: normalised speed limit
    ``> FLT_EPSILON`` (``inference/utils.hpp:112-123``; the ORT backend uses ``> 0``,
    ``onnxruntime_inference.cpp:41-48``; Autoware's default backend is TensorRT)."""
    return {"lanes_has_speed_limit": np.asarray(norm_inputs["lanes_speed_limit"]) > FLT_EPSILON,
            "route_lanes_has_speed_limit": np.asarray(norm_inputs["route_lanes_speed_limit"]) > FLT_EPSILON}