changh95's picture
tt-model push diffusion-planner-p150 (container)
4d9b003 verified
Raw History Blame Contribute Delete
10 kB
# SPDX-License-Identifier: Apache-2.0
"""Constants of the Autoware Diffusion Planner v5.0 contract, each with its source.
Shared by the host pre/post-processing (``tt_diffusion_planner.host``), the CPU reference (``reference``) and the
ttnn graph (``tt``). Every value that changes a device shape is a COMPILE parameter (PLAN.md section 0.4): a change
means a new trace and a new image.
Source abbreviations: ``PKG`` = autoware_universe @ 9ceaccf ``planning/autoware_diffusion_planner``; ``HFD`` = the
weights ``AutowareFoundation/diffusion_planner@v5.0`` (423efde6); ``SPEC`` = ``research/diffusion-planner/SPEC.md`` of
the porting workspace; ``T4M`` = tier4/Diffusion-Planner @ 40114a8 ``diffusion_planner/diffusion_planner`` (read
for semantics only, never imported).
"""
from __future__ import annotations
from typing import Dict, Tuple
import numpy as np
# ---- dimensions (PKG/include/autoware/diffusion_planner/dimensions.hpp:26-105) --------------------------------------
NUM_SEGMENTS_IN_LANE = 140
NUM_SEGMENTS_IN_ROUTE = 25
NUM_POLYGONS = 10
NUM_LINE_STRINGS = 60
NUM_STATIC_OBJECTS = 5
MAX_NUM_NEIGHBORS = 320
MAX_NUM_AGENTS = MAX_NUM_NEIGHBORS + 1 # ego + neighbours
HIDDEN_DIM = 256
POINTS_PER_SEGMENT = 20
POINTS_PER_POLYGON = 40
POINTS_PER_LINE_STRING = 20
LINE_TYPE_NUM = 10
POLYGON_TYPE_NUM = 1 # intersection_area
LINE_STRING_TYPE_NUM = 2 # stop_line, road_border
SEGMENT_POINT_DIM = 13 + 2 * LINE_TYPE_NUM # 33
INPUT_T = 30 # history steps before the current one (31 samples with the current)
OUTPUT_T = 80 # future steps (8 s at 0.1 s)
POSE_DIM = 4 # x, y, cos(yaw), sin(yaw)
AGENT_STATE_DIM = 11 # x, y, cos, sin, vx, vy, width, length, is_vehicle, is_pedestrian, is_bicycle
EGO_CURRENT_STATE_DIM = 10
STATIC_OBJECT_DIM = 10
EGO_SHAPE_DIM = 3 # wheel_base, length, width
TURN_INDICATOR_OUTPUT_DIM = 5
# logit order (dimensions.hpp:74-79); the published command is the index for 0..3 (TurnIndicatorsCommand)
TURN_INDICATOR_LABELS = ("NONE", "DISABLE", "ENABLE_LEFT", "ENABLE_RIGHT", "KEEP")
TURN_INDICATOR_OUTPUT_KEEP = 4
TURN_INDICATOR_COMMAND_NAMES = {0: "NO_COMMAND", 1: "DISABLE", 2: "ENABLE_LEFT", 3: "ENABLE_RIGHT"}
TURN_INDICATORS_REPORT_DISABLE = 1 # TurnIndicatorsReport::DISABLE, the prev_report default (core.cpp:635-637)
# ---- the encoder's 564 scene tokens, in concatenation order (T4M/model/module/encoder.py:251-265; SPEC 2.5) -------
TOKEN_LAYOUT: Tuple[Tuple[str, int], ...] = (
("ego", 1),
("neighbor", MAX_NUM_NEIGHBORS),
("static", NUM_STATIC_OBJECTS),
("lane", NUM_SEGMENTS_IN_LANE),
("route", NUM_SEGMENTS_IN_ROUTE),
("polygon", NUM_POLYGONS),
("line_string", NUM_LINE_STRINGS),
("goal", 1),
("ego_shape", 1),
("turn", 1),
)
ENCODING_TOKEN_NUM = sum(n for _, n in TOKEN_LAYOUT) # 564 (dimensions.hpp:33-35)
def token_slices() -> Dict[str, slice]:
"""``{category: slice of the 564 tokens}``."""
out, start = {}, 0
for name, n in TOKEN_LAYOUT:
out[name] = slice(start, start + n)
start += n
return out
TOKEN_SLICES = token_slices()
# class ids of the 14-dim positional feature (T4M encoder.py:10-21). The turn-indicator token reuses the ego-shape
# id 8 (FloatsEncoder hard-codes CLASS_TYPE_EGO_SHAPE, encoder.py:808; confirmed by the ONNX ConstantOfShape values).
POS_CLASS = {"ego": 0, "neighbor": 1, "static": 2, "lane": 3, "route": 4, "polygon": 5, "line_string": 6,
"goal": 7, "ego_shape": 8, "turn": 8}
POS_CLASS_NUM = 10
POS_FEATURE_DIM = 4 + POS_CLASS_NUM # 14
# ---- in-graph pre-processing of the encoder (SPEC 3.8; T4M encoder.py) ------------------------------------------
EGO_HISTORY_KEEP = slice(0, 6) # the 6 OLDEST ego samples are kept, rows 6..30 zeroed (encoder.py:170-175)
NEIGHBOR_HISTORY_KEEP = slice(INPUT_T + 1 - 6, INPUT_T + 1) # the 6 newest neighbour samples (rows 25..30)
TURN_INDICATOR_HISTORY = INPUT_T # turn_indicators[:, :-1]: the 30 values before the current report (encoder.py:209)
LANE_POS_INDEX = POINTS_PER_SEGMENT // 2 # 10: point used for the lane / route position feature (encoder.py:592)
POLYGON_POS_INDEX = POINTS_PER_POLYGON // 2 # 20 (LineEncoder, encoder.py:685)
LINE_STRING_POS_INDEX = POINTS_PER_LINE_STRING // 2 # 10
LANE_FEATURE_DIM = 8 # x, y, dx, dy, left - centre (x, y), right - centre (x, y)
LANE_ATTRIBUTE_DIM = SEGMENT_POINT_DIM - LANE_FEATURE_DIM # 25: traffic light (5) + line types (2 x 10) of point 0
NEIGHBOR_FEATURE_DIM = 9 # x, y, cos, sin, 0, 0 (velocities zeroed), width, length, valid-step flag
POLYGON_FEATURE_DIM = 2 + POLYGON_TYPE_NUM + 2 # x, y, type, dx, dy
LINE_STRING_FEATURE_DIM = 2 + LINE_STRING_TYPE_NUM + 2 # x, y, stop_line, road_border, dx, dy
# ---- network constants (ONNX, SPEC 4.1-4.4) ------------------------------------------------------------------------
LN_EPS = 1e-5 # every LayerNormalization of the three graphs (109 nodes; ttnn's default is 1e-12: pass it explicitly)
NUM_HEADS = 8
HEAD_DIM = HIDDEN_DIM // NUM_HEADS # 32
ATTN_SCALE = np.float32(0.17677669) # ONNX constant Mul_3 / Mul_2 = 1/sqrt(32) (applied to Q before Q.K^T)
MIXER_TOKENS = 64 # token_pre_project out, tokens_mlp width
MIXER_CHANNELS = 128 # channel_pre_project out, channels_mlp width
MIXER_DEPTH = 6 # encoder_mixer_depth (HFD param.json)
FUSION_DEPTH = 6 # encoder_fusion_depth
FUSION_MLP_DIM = 4 * HIDDEN_DIM
DIT_DEPTH = 3 # decoder_depth
DIT_MLP_DIM = 4 * HIDDEN_DIM
DIT_INPUT_DIM = (OUTPUT_T + 1) * POSE_DIM # 324 = preproj input / final projection output
DIT_TIME_DIM = OUTPUT_T + 1 # 81 = t_embedder input (one diffusion time per trajectory point)
TURN_HEAD_STEPS = tuple(range(1, OUTPUT_T, 10)) # final_x0[0, 1::10, :2] -> 8 points, 16 values (SPEC 4.4)
# ---- DPM-Solver++(2M) (PKG/src/inference/solver/dpm_solver.cpp:29-33; PKG/config/diffusion_planner.param.yaml) -----
# yaml l.17 model.multi_step_model.dpm_solver_steps: NFE = steps + 1 (COMPILE: the loop length of the trace)
DPM_SOLVER_STEPS = 10
DPM_SOLVER_ORDER = 2
NOISE_SCHEDULE_T = np.float32(1.0)
NOISE_SCHEDULE_TOTAL_N = np.float32(1000.0)
NOISE_SCHEDULE_BETA0 = np.float32(0.1)
NOISE_SCHEDULE_BETA1 = np.float32(20.0)
# SPEC 5.1 (recomputed from dpm_solver.cpp:80-94): the 11 solver timesteps for steps = 10, float32
DPM_TIMESTEPS_STEPS10 = (1.0, 0.89912426, 0.78557116, 0.65344393, 0.49344844, 0.30464348, 0.14064588, 0.05360911,
0.01809745, 0.00499264, 0.00100062)
# ---- node parameters this port reproduces (PKG/config/diffusion_planner.param.yaml, effective YAML defaults) -------
VELOCITY_SMOOTHING_WINDOW = 8 # yaml l.32
STOPPING_THRESHOLD = 0.3 # yaml l.33, m/s
TURN_INDICATOR_KEEP_OFFSET = -1.25 # yaml l.34
TURN_INDICATOR_HOLD_DURATION_S = 1.0 # yaml l.35 (stateful; see host.postprocess.TurnIndicatorManager)
TRAJECTORY_DT = 0.1 # postprocessing_utils.cpp:370 (constexpr double dt)
DELAY_STEP_MAX = OUTPUT_T // 2 # core.cpp:437 clamps delay_step to [0, 40]
# ---- raw input tensors: the InputDataMap of DiffusionPlannerCore::create_input_data (core.cpp:414-595), batch 1 ----
# All float32 and in the ego (base_link) frame, BEFORE normalization. ``sampled_trajectories`` is already in the
# normalised state space (x_T: zeros with the default temperature 0), ``delay`` is only read by the single-step graph.
# The speed-limit masks (lanes_has_speed_limit, route_lanes_has_speed_limit) are not inputs: the inference backend
# derives them from the normalised speed limits (PKG/include/.../inference/utils.hpp:112-123).
INPUT_SHAPES: Dict[str, Tuple[int, ...]] = {
"sampled_trajectories": (1, MAX_NUM_AGENTS, OUTPUT_T + 1, POSE_DIM),
"ego_agent_past": (1, INPUT_T + 1, POSE_DIM),
"ego_current_state": (1, EGO_CURRENT_STATE_DIM),
"neighbor_agents_past": (1, MAX_NUM_NEIGHBORS, INPUT_T + 1, AGENT_STATE_DIM),
"static_objects": (1, NUM_STATIC_OBJECTS, STATIC_OBJECT_DIM),
"lanes": (1, NUM_SEGMENTS_IN_LANE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
"lanes_speed_limit": (1, NUM_SEGMENTS_IN_LANE, 1),
"route_lanes": (1, NUM_SEGMENTS_IN_ROUTE, POINTS_PER_SEGMENT, SEGMENT_POINT_DIM),
"route_lanes_speed_limit": (1, NUM_SEGMENTS_IN_ROUTE, 1),
"polygons": (1, NUM_POLYGONS, POINTS_PER_POLYGON, 2 + POLYGON_TYPE_NUM),
"line_strings": (1, NUM_LINE_STRINGS, POINTS_PER_LINE_STRING, 2 + LINE_STRING_TYPE_NUM),
"goal_pose": (1, POSE_DIM),
"ego_shape": (1, EGO_SHAPE_DIM),
"turn_indicators": (1, INPUT_T + 1),
"delay": (1, 1),
}
INPUT_NAMES = tuple(INPUT_SHAPES)
# name -> (shape, dtype): the ModelBase.INPUT_SCHEMA of the API and the server (ttaw API.md section 10)
INPUT_SCHEMA = {name: (shape, np.float32) for name, shape in INPUT_SHAPES.items()}
# preprocessing_utils.cpp:34-84 normalises every tensor except these (core.cpp / node.cpp:633-634)
SKIP_NORMALIZATION = ("ego_shape", "sampled_trajectories", "turn_indicators", "delay")
# the inputs of the encoder graph, in its input order (HFD diffusion_planner_encoder.onnx)
ENCODER_INPUTS = ("ego_agent_past", "neighbor_agents_past", "static_objects", "lanes", "lanes_speed_limit",
"lanes_has_speed_limit", "route_lanes", "route_lanes_speed_limit", "route_lanes_has_speed_limit",
"polygons", "line_strings", "goal_pose", "ego_shape", "turn_indicators")
# ---- weight files (HFD; BUNDLE_CONVENTIONS.md section 12) -----------------------------------------------------------
ENCODER_ONNX = "diffusion_planner_encoder.onnx"
DECODER_ONNX = "diffusion_planner_decoder.onnx"
TURN_INDICATOR_ONNX = "diffusion_planner_turn_indicator.onnx"
PARAM_JSON = "diffusion_planner.param.json"
WEIGHT_MAJOR_VERSION = 5 # PKG/include/autoware/diffusion_planner/constants.hpp:22 (arg_reader.hpp:54-78)
FILE_SHA256 = { # SPEC 1.4
ENCODER_ONNX: "2856886a3ed63b963cb18b457876d0bbd8916d345549ee5f199ceb49e3fcca49",
DECODER_ONNX: "eb30c0c0c8e80b8460d293d600c7b8c9e21f615ff6ce8de5ed01670c3d1057ca",
TURN_INDICATOR_ONNX: "07acfb58a5de1f25587fa86deb39a6de5605f8d93518e97f960ee27145f4a732",
PARAM_JSON: "ee3145b68fd1e1e44e532933dfe66cfee4384fbd637382c87ab5190c66a8e268",
}