File size: 10,048 Bytes
4d9b003
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
# SPDX-License-Identifier: Apache-2.0
"""Constants of the Autoware Diffusion Planner v5.0 contract, each with its source.

Shared by the host pre/post-processing (``tt_diffusion_planner.host``), the CPU reference (``reference``) and the
ttnn graph (``tt``). Every value that changes a device shape is a COMPILE parameter (PLAN.md section 0.4): a change
means a new trace and a new image.

Source abbreviations: ``PKG`` = autoware_universe @ 9ceaccf ``planning/autoware_diffusion_planner``; ``HFD`` = the
weights ``AutowareFoundation/diffusion_planner@v5.0`` (423efde6); ``SPEC`` = ``research/diffusion-planner/SPEC.md`` of
the porting workspace; ``T4M`` = tier4/Diffusion-Planner @ 40114a8 ``diffusion_planner/diffusion_planner`` (read
for semantics only, never imported).
"""
from __future__ import annotations

from typing import Dict, Tuple

import numpy as np

# ---- dimensions (PKG/include/autoware/diffusion_planner/dimensions.hpp:26-105) --------------------------------------
NUM_SEGMENTS_IN_LANE = 140
NUM_SEGMENTS_IN_ROUTE = 25
NUM_POLYGONS = 10
NUM_LINE_STRINGS = 60
NUM_STATIC_OBJECTS = 5
MAX_NUM_NEIGHBORS = 320
MAX_NUM_AGENTS = MAX_NUM_NEIGHBORS + 1  # ego + neighbours
HIDDEN_DIM = 256
POINTS_PER_SEGMENT = 20
POINTS_PER_POLYGON = 40
POINTS_PER_LINE_STRING = 20
LINE_TYPE_NUM = 10
POLYGON_TYPE_NUM = 1  # intersection_area
LINE_STRING_TYPE_NUM = 2  # stop_line, road_border
SEGMENT_POINT_DIM = 13 + 2 * LINE_TYPE_NUM  # 33
INPUT_T = 30  # history steps before the current one (31 samples with the current)
OUTPUT_T = 80  # future steps (8 s at 0.1 s)
POSE_DIM = 4  # x, y, cos(yaw), sin(yaw)
AGENT_STATE_DIM = 11  # x, y, cos, sin, vx, vy, width, length, is_vehicle, is_pedestrian, is_bicycle
EGO_CURRENT_STATE_DIM = 10
STATIC_OBJECT_DIM = 10
EGO_SHAPE_DIM = 3  # wheel_base, length, width
TURN_INDICATOR_OUTPUT_DIM = 5
# logit order (dimensions.hpp:74-79); the published command is the index for 0..3 (TurnIndicatorsCommand)
TURN_INDICATOR_LABELS = ("NONE", "DISABLE", "ENABLE_LEFT", "ENABLE_RIGHT", "KEEP")
TURN_INDICATOR_OUTPUT_KEEP = 4
TURN_INDICATOR_COMMAND_NAMES = {0: "NO_COMMAND", 1: "DISABLE", 2: "ENABLE_LEFT", 3: "ENABLE_RIGHT"}
TURN_INDICATORS_REPORT_DISABLE = 1  # TurnIndicatorsReport::DISABLE, the prev_report default (core.cpp:635-637)

# ---- the encoder's 564 scene tokens, in concatenation order (T4M/model/module/encoder.py:251-265; SPEC 2.5) -------
TOKEN_LAYOUT: Tuple[Tuple[str, int], ...] = (
    ("ego", 1),
    ("neighbor", MAX_NUM_NEIGHBORS),
    ("static", NUM_STATIC_OBJECTS),
    ("lane", NUM_SEGMENTS_IN_LANE),
    ("route", NUM_SEGMENTS_IN_ROUTE),
    ("polygon", NUM_POLYGONS),
    ("line_string", NUM_LINE_STRINGS),
    ("goal", 1),
    ("ego_shape", 1),
    ("turn", 1),
)
ENCODING_TOKEN_NUM = sum(n for _, n in TOKEN_LAYOUT)  # 564 (dimensions.hpp:33-35)


def token_slices() -> Dict[str, slice]:
    """``{category: slice of the 564 tokens}``."""
    out, start = {}, 0
    for name, n in TOKEN_LAYOUT:
        out[name] = slice(start, start + n)
        start += n
    return out


TOKEN_SLICES = token_slices()

# class ids of the 14-dim positional feature (T4M encoder.py:10-21). The turn-indicator token reuses the ego-shape
# id 8 (FloatsEncoder hard-codes CLASS_TYPE_EGO_SHAPE, encoder.py:808; confirmed by the ONNX ConstantOfShape values).
POS_CLASS = {"ego": 0, "neighbor": 1, "static": 2, "lane": 3, "route": 4, "polygon": 5, "line_string": 6,
             "goal": 7, "ego_shape": 8, "turn": 8}
POS_CLASS_NUM = 10
POS_FEATURE_DIM = 4 + POS_CLASS_NUM  # 14

# ---- in-graph pre-processing of the encoder (SPEC 3.8; T4M encoder.py) ------------------------------------------
EGO_HISTORY_KEEP = slice(0, 6)  # the 6 OLDEST ego samples are kept, rows 6..30 zeroed (encoder.py:170-175)
NEIGHBOR_HISTORY_KEEP = slice(INPUT_T + 1 - 6, INPUT_T + 1)  # the 6 newest neighbour samples (rows 25..30)
TURN_INDICATOR_HISTORY = INPUT_T  # turn_indicators[:, :-1]: the 30 values before the current report (encoder.py:209)
LANE_POS_INDEX = POINTS_PER_SEGMENT // 2  # 10: point used for the lane / route position feature (encoder.py:592)
POLYGON_POS_INDEX = POINTS_PER_POLYGON // 2  # 20 (LineEncoder, encoder.py:685)
LINE_STRING_POS_INDEX = POINTS_PER_LINE_STRING // 2  # 10
LANE_FEATURE_DIM = 8  # x, y, dx, dy, left - centre (x, y), right - centre (x, y)
LANE_ATTRIBUTE_DIM = SEGMENT_POINT_DIM - LANE_FEATURE_DIM  # 25: traffic light (5) + line types (2 x 10) of point 0
NEIGHBOR_FEATURE_DIM = 9  # x, y, cos, sin, 0, 0 (velocities zeroed), width, length, valid-step flag
POLYGON_FEATURE_DIM = 2 + POLYGON_TYPE_NUM + 2  # x, y, type, dx, dy
LINE_STRING_FEATURE_DIM = 2 + LINE_STRING_TYPE_NUM + 2  # x, y, stop_line, road_border, dx, dy

# ---- network constants (ONNX, SPEC 4.1-4.4) ------------------------------------------------------------------------
LN_EPS = 1e-5  # every LayerNormalization of the three graphs (109 nodes; ttnn's default is 1e-12: pass it explicitly)
NUM_HEADS = 8
HEAD_DIM = HIDDEN_DIM // NUM_HEADS  # 32
ATTN_SCALE = np.float32(0.17677669)  # ONNX constant Mul_3 / Mul_2 = 1/sqrt(32) (applied to Q before Q.K^T)
MIXER_TOKENS = 64  # token_pre_project out, tokens_mlp width
MIXER_CHANNELS = 128  # channel_pre_project out, channels_mlp width
MIXER_DEPTH = 6  # encoder_mixer_depth (HFD param.json)
FUSION_DEPTH = 6  # encoder_fusion_depth
FUSION_MLP_DIM = 4 * HIDDEN_DIM
DIT_DEPTH = 3  # decoder_depth
DIT_MLP_DIM = 4 * HIDDEN_DIM
DIT_INPUT_DIM = (OUTPUT_T + 1) * POSE_DIM  # 324 = preproj input / final projection output
DIT_TIME_DIM = OUTPUT_T + 1  # 81 = t_embedder input (one diffusion time per trajectory point)
TURN_HEAD_STEPS = tuple(range(1, OUTPUT_T, 10))  # final_x0[0, 1::10, :2] -> 8 points, 16 values (SPEC 4.4)

# ---- DPM-Solver++(2M) (PKG/src/inference/solver/dpm_solver.cpp:29-33; PKG/config/diffusion_planner.param.yaml) -----
# yaml l.17 model.multi_step_model.dpm_solver_steps: NFE = steps + 1 (COMPILE: the loop length of the trace)
DPM_SOLVER_STEPS = 10
DPM_SOLVER_ORDER = 2
NOISE_SCHEDULE_T = np.float32(1.0)
NOISE_SCHEDULE_TOTAL_N = np.float32(1000.0)
NOISE_SCHEDULE_BETA0 = np.float32(0.1)
NOISE_SCHEDULE_BETA1 = np.float32(20.0)
# SPEC 5.1 (recomputed from dpm_solver.cpp:80-94): the 11 solver timesteps for steps = 10, float32
DPM_TIMESTEPS_STEPS10 = (1.0, 0.89912426, 0.78557116, 0.65344393, 0.49344844, 0.30464348, 0.14064588, 0.05360911,
                         0.01809745, 0.00499264, 0.00100062)

# ---- node parameters this port reproduces (PKG/config/diffusion_planner.param.yaml, effective YAML defaults) -------
VELOCITY_SMOOTHING_WINDOW = 8  # yaml l.32
STOPPING_THRESHOLD = 0.3  # yaml l.33, m/s
TURN_INDICATOR_KEEP_OFFSET = -1.25  # yaml l.34
TURN_INDICATOR_HOLD_DURATION_S = 1.0  # yaml l.35 (stateful; see host.postprocess.TurnIndicatorManager)
TRAJECTORY_DT = 0.1  # postprocessing_utils.cpp:370 (constexpr double dt)
DELAY_STEP_MAX = OUTPUT_T // 2  # core.cpp:437 clamps delay_step to [0, 40]

# ---- raw input tensors: the InputDataMap of DiffusionPlannerCore::create_input_data (core.cpp:414-595), batch 1 ----
# All float32 and in the ego (base_link) frame, BEFORE normalization. ``sampled_trajectories`` is already in the
# normalised state space (x_T: zeros with the default temperature 0), ``delay`` is only read by the single-step graph.
# The speed-limit masks (lanes_has_speed_limit, route_lanes_has_speed_limit) are not inputs: the inference backend
# derives them from the normalised speed limits (PKG/include/.../inference/utils.hpp:112-123).
INPUT_SHAPES: Dict[str, Tuple[int, ...]] = {
    "sampled_trajectories": (1, MAX_NUM_AGENTS, OUTPUT_T + 1, POSE_DIM),
    "ego_agent_past": (1, INPUT_T + 1, POSE_DIM),
    "ego_current_state": (1, EGO_CURRENT_STATE_DIM),
    "neighbor_agents_past": (1, MAX_NUM_NEIGHBORS, INPUT_T + 1, AGENT_STATE_DIM),
    "static_objects": (1, NUM_STATIC_OBJECTS, STATIC_OBJECT_DIM),
    "lanes": (1, NUM_SEGMENTS_IN_LANE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
    "lanes_speed_limit": (1, NUM_SEGMENTS_IN_LANE, 1),
    "route_lanes": (1, NUM_SEGMENTS_IN_ROUTE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
    "route_lanes_speed_limit": (1, NUM_SEGMENTS_IN_ROUTE, 1),
    "polygons": (1, NUM_POLYGONS, POINTS_PER_POLYGON, 2 + POLYGON_TYPE_NUM),
    "line_strings": (1, NUM_LINE_STRINGS, POINTS_PER_LINE_STRING, 2 + LINE_STRING_TYPE_NUM),
    "goal_pose": (1, POSE_DIM),
    "ego_shape": (1, EGO_SHAPE_DIM),
    "turn_indicators": (1, INPUT_T + 1),
    "delay": (1, 1),
}
INPUT_NAMES = tuple(INPUT_SHAPES)
# name -> (shape, dtype): the ModelBase.INPUT_SCHEMA of the API and the server (ttaw API.md section 10)
INPUT_SCHEMA = {name: (shape, np.float32) for name, shape in INPUT_SHAPES.items()}
# preprocessing_utils.cpp:34-84 normalises every tensor except these (core.cpp / node.cpp:633-634)
SKIP_NORMALIZATION = ("ego_shape", "sampled_trajectories", "turn_indicators", "delay")
# the inputs of the encoder graph, in its input order (HFD diffusion_planner_encoder.onnx)
ENCODER_INPUTS = ("ego_agent_past", "neighbor_agents_past", "static_objects", "lanes", "lanes_speed_limit",
                  "lanes_has_speed_limit", "route_lanes", "route_lanes_speed_limit", "route_lanes_has_speed_limit",
                  "polygons", "line_strings", "goal_pose", "ego_shape", "turn_indicators")

# ---- weight files (HFD; BUNDLE_CONVENTIONS.md section 12) -----------------------------------------------------------
ENCODER_ONNX = "diffusion_planner_encoder.onnx"
DECODER_ONNX = "diffusion_planner_decoder.onnx"
TURN_INDICATOR_ONNX = "diffusion_planner_turn_indicator.onnx"
PARAM_JSON = "diffusion_planner.param.json"
WEIGHT_MAJOR_VERSION = 5  # PKG/include/autoware/diffusion_planner/constants.hpp:22 (arg_reader.hpp:54-78)
FILE_SHA256 = {  # SPEC 1.4
    ENCODER_ONNX: "2856886a3ed63b963cb18b457876d0bbd8916d345549ee5f199ceb49e3fcca49",
    DECODER_ONNX: "eb30c0c0c8e80b8460d293d600c7b8c9e21f615ff6ce8de5ed01670c3d1057ca",
    TURN_INDICATOR_ONNX: "07acfb58a5de1f25587fa86deb39a6de5605f8d93518e97f960ee27145f4a732",
    PARAM_JSON: "ee3145b68fd1e1e44e532933dfe66cfee4384fbd637382c87ab5190c66a8e268",
}