File size: 10,048 Bytes
4d9b003 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 | # SPDX-License-Identifier: Apache-2.0
"""Constants of the Autoware Diffusion Planner v5.0 contract, each with its source.
Shared by the host pre/post-processing (``tt_diffusion_planner.host``), the CPU reference (``reference``) and the
ttnn graph (``tt``). Every value that changes a device shape is a COMPILE parameter (PLAN.md section 0.4): a change
means a new trace and a new image.
Source abbreviations: ``PKG`` = autoware_universe @ 9ceaccf ``planning/autoware_diffusion_planner``; ``HFD`` = the
weights ``AutowareFoundation/diffusion_planner@v5.0`` (423efde6); ``SPEC`` = ``research/diffusion-planner/SPEC.md`` of
the porting workspace; ``T4M`` = tier4/Diffusion-Planner @ 40114a8 ``diffusion_planner/diffusion_planner`` (read
for semantics only, never imported).
"""
from __future__ import annotations
from typing import Dict, Tuple
import numpy as np
# ---- dimensions (PKG/include/autoware/diffusion_planner/dimensions.hpp:26-105) --------------------------------------
NUM_SEGMENTS_IN_LANE = 140
NUM_SEGMENTS_IN_ROUTE = 25
NUM_POLYGONS = 10
NUM_LINE_STRINGS = 60
NUM_STATIC_OBJECTS = 5
MAX_NUM_NEIGHBORS = 320
MAX_NUM_AGENTS = MAX_NUM_NEIGHBORS + 1 # ego + neighbours
HIDDEN_DIM = 256
POINTS_PER_SEGMENT = 20
POINTS_PER_POLYGON = 40
POINTS_PER_LINE_STRING = 20
LINE_TYPE_NUM = 10
POLYGON_TYPE_NUM = 1 # intersection_area
LINE_STRING_TYPE_NUM = 2 # stop_line, road_border
SEGMENT_POINT_DIM = 13 + 2 * LINE_TYPE_NUM # 33
INPUT_T = 30 # history steps before the current one (31 samples with the current)
OUTPUT_T = 80 # future steps (8 s at 0.1 s)
POSE_DIM = 4 # x, y, cos(yaw), sin(yaw)
AGENT_STATE_DIM = 11 # x, y, cos, sin, vx, vy, width, length, is_vehicle, is_pedestrian, is_bicycle
EGO_CURRENT_STATE_DIM = 10
STATIC_OBJECT_DIM = 10
EGO_SHAPE_DIM = 3 # wheel_base, length, width
TURN_INDICATOR_OUTPUT_DIM = 5
# logit order (dimensions.hpp:74-79); the published command is the index for 0..3 (TurnIndicatorsCommand)
TURN_INDICATOR_LABELS = ("NONE", "DISABLE", "ENABLE_LEFT", "ENABLE_RIGHT", "KEEP")
TURN_INDICATOR_OUTPUT_KEEP = 4
TURN_INDICATOR_COMMAND_NAMES = {0: "NO_COMMAND", 1: "DISABLE", 2: "ENABLE_LEFT", 3: "ENABLE_RIGHT"}
TURN_INDICATORS_REPORT_DISABLE = 1 # TurnIndicatorsReport::DISABLE, the prev_report default (core.cpp:635-637)
# ---- the encoder's 564 scene tokens, in concatenation order (T4M/model/module/encoder.py:251-265; SPEC 2.5) -------
TOKEN_LAYOUT: Tuple[Tuple[str, int], ...] = (
("ego", 1),
("neighbor", MAX_NUM_NEIGHBORS),
("static", NUM_STATIC_OBJECTS),
("lane", NUM_SEGMENTS_IN_LANE),
("route", NUM_SEGMENTS_IN_ROUTE),
("polygon", NUM_POLYGONS),
("line_string", NUM_LINE_STRINGS),
("goal", 1),
("ego_shape", 1),
("turn", 1),
)
ENCODING_TOKEN_NUM = sum(n for _, n in TOKEN_LAYOUT) # 564 (dimensions.hpp:33-35)
def token_slices() -> Dict[str, slice]:
"""``{category: slice of the 564 tokens}``."""
out, start = {}, 0
for name, n in TOKEN_LAYOUT:
out[name] = slice(start, start + n)
start += n
return out
TOKEN_SLICES = token_slices()
# class ids of the 14-dim positional feature (T4M encoder.py:10-21). The turn-indicator token reuses the ego-shape
# id 8 (FloatsEncoder hard-codes CLASS_TYPE_EGO_SHAPE, encoder.py:808; confirmed by the ONNX ConstantOfShape values).
POS_CLASS = {"ego": 0, "neighbor": 1, "static": 2, "lane": 3, "route": 4, "polygon": 5, "line_string": 6,
"goal": 7, "ego_shape": 8, "turn": 8}
POS_CLASS_NUM = 10
POS_FEATURE_DIM = 4 + POS_CLASS_NUM # 14
# ---- in-graph pre-processing of the encoder (SPEC 3.8; T4M encoder.py) ------------------------------------------
EGO_HISTORY_KEEP = slice(0, 6) # the 6 OLDEST ego samples are kept, rows 6..30 zeroed (encoder.py:170-175)
NEIGHBOR_HISTORY_KEEP = slice(INPUT_T + 1 - 6, INPUT_T + 1) # the 6 newest neighbour samples (rows 25..30)
TURN_INDICATOR_HISTORY = INPUT_T # turn_indicators[:, :-1]: the 30 values before the current report (encoder.py:209)
LANE_POS_INDEX = POINTS_PER_SEGMENT // 2 # 10: point used for the lane / route position feature (encoder.py:592)
POLYGON_POS_INDEX = POINTS_PER_POLYGON // 2 # 20 (LineEncoder, encoder.py:685)
LINE_STRING_POS_INDEX = POINTS_PER_LINE_STRING // 2 # 10
LANE_FEATURE_DIM = 8 # x, y, dx, dy, left - centre (x, y), right - centre (x, y)
LANE_ATTRIBUTE_DIM = SEGMENT_POINT_DIM - LANE_FEATURE_DIM # 25: traffic light (5) + line types (2 x 10) of point 0
NEIGHBOR_FEATURE_DIM = 9 # x, y, cos, sin, 0, 0 (velocities zeroed), width, length, valid-step flag
POLYGON_FEATURE_DIM = 2 + POLYGON_TYPE_NUM + 2 # x, y, type, dx, dy
LINE_STRING_FEATURE_DIM = 2 + LINE_STRING_TYPE_NUM + 2 # x, y, stop_line, road_border, dx, dy
# ---- network constants (ONNX, SPEC 4.1-4.4) ------------------------------------------------------------------------
LN_EPS = 1e-5 # every LayerNormalization of the three graphs (109 nodes; ttnn's default is 1e-12: pass it explicitly)
NUM_HEADS = 8
HEAD_DIM = HIDDEN_DIM // NUM_HEADS # 32
ATTN_SCALE = np.float32(0.17677669) # ONNX constant Mul_3 / Mul_2 = 1/sqrt(32) (applied to Q before Q.K^T)
MIXER_TOKENS = 64 # token_pre_project out, tokens_mlp width
MIXER_CHANNELS = 128 # channel_pre_project out, channels_mlp width
MIXER_DEPTH = 6 # encoder_mixer_depth (HFD param.json)
FUSION_DEPTH = 6 # encoder_fusion_depth
FUSION_MLP_DIM = 4 * HIDDEN_DIM
DIT_DEPTH = 3 # decoder_depth
DIT_MLP_DIM = 4 * HIDDEN_DIM
DIT_INPUT_DIM = (OUTPUT_T + 1) * POSE_DIM # 324 = preproj input / final projection output
DIT_TIME_DIM = OUTPUT_T + 1 # 81 = t_embedder input (one diffusion time per trajectory point)
TURN_HEAD_STEPS = tuple(range(1, OUTPUT_T, 10)) # final_x0[0, 1::10, :2] -> 8 points, 16 values (SPEC 4.4)
# ---- DPM-Solver++(2M) (PKG/src/inference/solver/dpm_solver.cpp:29-33; PKG/config/diffusion_planner.param.yaml) -----
# yaml l.17 model.multi_step_model.dpm_solver_steps: NFE = steps + 1 (COMPILE: the loop length of the trace)
DPM_SOLVER_STEPS = 10
DPM_SOLVER_ORDER = 2
NOISE_SCHEDULE_T = np.float32(1.0)
NOISE_SCHEDULE_TOTAL_N = np.float32(1000.0)
NOISE_SCHEDULE_BETA0 = np.float32(0.1)
NOISE_SCHEDULE_BETA1 = np.float32(20.0)
# SPEC 5.1 (recomputed from dpm_solver.cpp:80-94): the 11 solver timesteps for steps = 10, float32
DPM_TIMESTEPS_STEPS10 = (1.0, 0.89912426, 0.78557116, 0.65344393, 0.49344844, 0.30464348, 0.14064588, 0.05360911,
0.01809745, 0.00499264, 0.00100062)
# ---- node parameters this port reproduces (PKG/config/diffusion_planner.param.yaml, effective YAML defaults) -------
VELOCITY_SMOOTHING_WINDOW = 8 # yaml l.32
STOPPING_THRESHOLD = 0.3 # yaml l.33, m/s
TURN_INDICATOR_KEEP_OFFSET = -1.25 # yaml l.34
TURN_INDICATOR_HOLD_DURATION_S = 1.0 # yaml l.35 (stateful; see host.postprocess.TurnIndicatorManager)
TRAJECTORY_DT = 0.1 # postprocessing_utils.cpp:370 (constexpr double dt)
DELAY_STEP_MAX = OUTPUT_T // 2 # core.cpp:437 clamps delay_step to [0, 40]
# ---- raw input tensors: the InputDataMap of DiffusionPlannerCore::create_input_data (core.cpp:414-595), batch 1 ----
# All float32 and in the ego (base_link) frame, BEFORE normalization. ``sampled_trajectories`` is already in the
# normalised state space (x_T: zeros with the default temperature 0), ``delay`` is only read by the single-step graph.
# The speed-limit masks (lanes_has_speed_limit, route_lanes_has_speed_limit) are not inputs: the inference backend
# derives them from the normalised speed limits (PKG/include/.../inference/utils.hpp:112-123).
INPUT_SHAPES: Dict[str, Tuple[int, ...]] = {
"sampled_trajectories": (1, MAX_NUM_AGENTS, OUTPUT_T + 1, POSE_DIM),
"ego_agent_past": (1, INPUT_T + 1, POSE_DIM),
"ego_current_state": (1, EGO_CURRENT_STATE_DIM),
"neighbor_agents_past": (1, MAX_NUM_NEIGHBORS, INPUT_T + 1, AGENT_STATE_DIM),
"static_objects": (1, NUM_STATIC_OBJECTS, STATIC_OBJECT_DIM),
"lanes": (1, NUM_SEGMENTS_IN_LANE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
"lanes_speed_limit": (1, NUM_SEGMENTS_IN_LANE, 1),
"route_lanes": (1, NUM_SEGMENTS_IN_ROUTE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
"route_lanes_speed_limit": (1, NUM_SEGMENTS_IN_ROUTE, 1),
"polygons": (1, NUM_POLYGONS, POINTS_PER_POLYGON, 2 + POLYGON_TYPE_NUM),
"line_strings": (1, NUM_LINE_STRINGS, POINTS_PER_LINE_STRING, 2 + LINE_STRING_TYPE_NUM),
"goal_pose": (1, POSE_DIM),
"ego_shape": (1, EGO_SHAPE_DIM),
"turn_indicators": (1, INPUT_T + 1),
"delay": (1, 1),
}
INPUT_NAMES = tuple(INPUT_SHAPES)
# name -> (shape, dtype): the ModelBase.INPUT_SCHEMA of the API and the server (ttaw API.md section 10)
INPUT_SCHEMA = {name: (shape, np.float32) for name, shape in INPUT_SHAPES.items()}
# preprocessing_utils.cpp:34-84 normalises every tensor except these (core.cpp / node.cpp:633-634)
SKIP_NORMALIZATION = ("ego_shape", "sampled_trajectories", "turn_indicators", "delay")
# the inputs of the encoder graph, in its input order (HFD diffusion_planner_encoder.onnx)
ENCODER_INPUTS = ("ego_agent_past", "neighbor_agents_past", "static_objects", "lanes", "lanes_speed_limit",
"lanes_has_speed_limit", "route_lanes", "route_lanes_speed_limit", "route_lanes_has_speed_limit",
"polygons", "line_strings", "goal_pose", "ego_shape", "turn_indicators")
# ---- weight files (HFD; BUNDLE_CONVENTIONS.md section 12) -----------------------------------------------------------
ENCODER_ONNX = "diffusion_planner_encoder.onnx"
DECODER_ONNX = "diffusion_planner_decoder.onnx"
TURN_INDICATOR_ONNX = "diffusion_planner_turn_indicator.onnx"
PARAM_JSON = "diffusion_planner.param.json"
WEIGHT_MAJOR_VERSION = 5 # PKG/include/autoware/diffusion_planner/constants.hpp:22 (arg_reader.hpp:54-78)
FILE_SHA256 = { # SPEC 1.4
ENCODER_ONNX: "2856886a3ed63b963cb18b457876d0bbd8916d345549ee5f199ceb49e3fcca49",
DECODER_ONNX: "eb30c0c0c8e80b8460d293d600c7b8c9e21f615ff6ce8de5ed01670c3d1057ca",
TURN_INDICATOR_ONNX: "07acfb58a5de1f25587fa86deb39a6de5605f8d93518e97f960ee27145f4a732",
PARAM_JSON: "ee3145b68fd1e1e44e532933dfe66cfee4384fbd637382c87ab5190c66a8e268",
}
|