diffusion-planner-p150 / tt_kernel_manifest.json
changh95's picture
tt-model push diffusion-planner-p150 (container)
be62f78 verified
Raw History Blame Contribute Delete
13.1 kB
{
"schema_version": "5.1",
"name": "diffusion-planner-p150",
"tt_metal_version": "0.65.2.dev11169+g44d66500520",
"arch": "blackhole",
"device_count": 1,
"producer": {
"tt_kernel_version": "0.1.0",
"created_at": "2026-10-11T04:02:31.337918+00:00",
"hostname": "hchang-bh"
},
"weights": {
"repo_id": "AutowareFoundation/diffusion_planner",
"revision": "423efde67f5414734da43a7ad856c17ceb8b51aa",
"allow_patterns": [
"diffusion_planner_encoder.onnx",
"diffusion_planner_decoder.onnx",
"diffusion_planner_turn_indicator.onnx",
"diffusion_planner.param.json"
],
"ignore_patterns": null,
"repo_type": "model"
},
"mesh": null,
"entrypoint": null,
"resources": null,
"capabilities": null,
"env": {},
"bundled": null,
"deps": null,
"container": {
"image": {
"registry": "hf",
"repository": "diffusion-planner-p150",
"tag": "tt-model/diffusion-planner-p150:dde78ac2f0be",
"digest": "sha256:dde78ac2f0be5b7e637ddceba1a7c30fd832c2a50dd3e728acbf187f86354cf4"
},
"kind": "tt-dit-server",
"runtime": {
"app": "tt_diffusion_planner.server.app:app",
"mesh_shape_env": "TT_MESH_SHAPE",
"packages": [
"numpy>=1.24.4,<2",
"huggingface_hub",
"safetensors",
"pyyaml",
"onnx>=1.17,<2"
],
"lock": "requirements.lock"
},
"serve": {
"hardware": "p150",
"mesh_device": "P150",
"port": 20000,
"max_model_len": null,
"max_num_seqs": null,
"block_size": null,
"server_timeout": null,
"capabilities": null,
"additional_config": {},
"args": [],
"env": {
"TT_WEIGHTS_REVISION": "423efde67f5414734da43a7ad856c17ceb8b51aa",
"TT_METAL_VISIBLE_DEVICES": "0",
"HF_HUB_DISABLE_IMPLICIT_TOKEN": "1",
"DIFFUSION_PLANNER_DISPATCH": "eth",
"DIFFUSION_PLANNER_NUM_CQS": "1",
"DIFFUSION_PLANNER_VARIANT": "default",
"DIFFUSION_PLANNER_LN_FP32": "enc.mixer.*,dec.*",
"DIFFUSION_PLANNER_HIDDEN_FP32": "",
"DIFFUSION_PLANNER_SPLIT_MATMUL": "enc.island.*,enc.pre.*,dec.*",
"DIFFUSION_PLANNER_ATTN_FP32_ACC": "",
"DIFFUSION_PLANNER_ATTN_MATMUL": "enc.fusion.attn,dec.*",
"DIFFUSION_PLANNER_ENC_CH2D": "1",
"DIFFUSION_PLANNER_ATTN_FAST": "1",
"DIFFUSION_PLANNER_DEC_MMCFG": "1",
"DIFFUSION_PLANNER_LN_KERNEL": "2",
"DIFFUSION_PLANNER_LN_RESID": "1",
"DIFFUSION_PLANNER_SPLIT_KCAT": "2",
"DIFFUSION_PLANNER_ATTN_SMASK": "1",
"DIFFUSION_PLANNER_ATTN_SMSM": "1",
"DIFFUSION_PLANNER_ATTN_FUSED": "1",
"DIFFUSION_PLANNER_KCAT_EMIT": "1",
"DIFFUSION_PLANNER_LN_TR": "1",
"DIFFUSION_PLANNER_KCAT_L1": "1",
"DIFFUSION_PLANNER_KCAT_ACT_ONCE": "0",
"DIFFUSION_PLANNER_ATTN_L1": "2",
"DIFFUSION_PLANNER_DEC_L1": "2",
"DIFFUSION_PLANNER_ENC_L1": "0",
"DIFFUSION_PLANNER_FUS_L1": "0",
"DIFFUSION_PLANNER_ENC_KCAT": "1",
"DIFFUSION_PLANNER_LIN_ACT": "0",
"DIFFUSION_PLANNER_KCAT_ACT": "1",
"DIFFUSION_PLANNER_LN_SPLIT": "1",
"DIFFUSION_PLANNER_LN_SFPU_BCAST": "1",
"DIFFUSION_PLANNER_COMPACT": "2",
"DIFFUSION_PLANNER_AGENT_BUCKETS": "32,64,96,128,192",
"DIFFUSION_PLANNER_HOST_FAST": "1",
"DIFFUSION_PLANNER_INPUT_TRIM": "1",
"DIFFUSION_PLANNER_PRECISION": ""
}
},
"serve_profiles": [
{
"hardware": null,
"mesh_device": null,
"port": null,
"max_model_len": null,
"max_num_seqs": null,
"block_size": null,
"server_timeout": null,
"capabilities": null,
"additional_config": {},
"args": [],
"env": {},
"name": "default",
"description": null
}
],
"default_profile": null,
"code_dir": "code",
"verify": [
"assert 'single_chip_arch_1cq_no_dispatch_s' in open('/opt/tt-metal/tt_metal/impl/dispatch/topology.cpp').read()",
"from tt_diffusion_planner.ttaw.device import eth_dispatch_patch_present as p; assert p(), 'ttaw does not see the ETH patch'",
"assert 'dual_kernel_min_dram_dest_page_bytes' in open('/opt/tt-metal/ttnn/cpp/ttnn/operations/data_movement/reshape_view/device/reshape_rm_program_factory.cpp').read()",
"import tt_diffusion_planner.server.app as a; assert a.app",
"from tt_diffusion_planner.server.app import parse_mesh_shape as p; assert p('1x1') == p('(1, 1)') == p('1,1') == (1, 1)",
"import numpy; assert int(numpy.__version__.split('.')[0]) < 2, numpy.__version__",
"import onnx, yaml, huggingface_hub, safetensors, PIL",
"from tt_diffusion_planner.api import DiffusionPlanner; assert DiffusionPlanner.DEFAULT_REVISION == '423efde67f5414734da43a7ad856c17ceb8b51aa'",
"import ttnn; assert hasattr(ttnn, 'DispatchCoreConfig') and hasattr(ttnn, 'begin_trace_capture') and hasattr(ttnn, 'execute_trace') and hasattr(ttnn, 'release_trace')",
"from pathlib import Path; d = Path('/opt/tt-metal/tt_diffusion_planner/samples'); assert all((d / f).is_file() for f in ('kashiwanoha_dense.npz', 'kashiwanoha_dense.reference.json'))",
"from pathlib import Path; d = Path('/opt/tt-metal/tt_diffusion_planner/tt/kernels'); assert all((d / f).is_file() for f in ('ln32_reader.cpp', 'ln32_compute.cpp', 'ln32_writer.cpp', 'kcat_reader.cpp', 'kcat_compute.cpp', 'kcat_writer.cpp', 'smask_reader.cpp', 'smask_compute.cpp', 'smask_writer.cpp', 'ln32s_reader.cpp', 'ln32s_compute.cpp', 'ln32s_writer.cpp', 'ln32_sfpu.h', 'smsm_reader.cpp', 'smsm_compute.cpp', 'smsm_writer.cpp', 'fattn_reader.cpp', 'fattn_compute.cpp', 'fattn_writer.cpp'))"
],
"built": {
"image": "tt-model/diffusion-planner-p150:dde78ac2f0be",
"repo": "changh95/diffusion-planner-p150",
"tt_model_version": "0.1.0",
"created_at": "2026-10-11T03:53:34+00:00",
"tt_metal": {
"sha": "44d66500520fda9f2c7060c0f6b41ec48f7ab37e",
"describe": "v0.80.0-dev20261006-78-g44d6650052-dirty",
"dirty": true,
"scm_version": "0.65.2.dev11169+g44d66500520",
"mode": "local",
"remote": "https://github.com/tenstorrent/tt-metal.git",
"branch": "main",
"pushed": true
},
"code_sha256": "5da22b97bbf89b03f83133089a6a3d01a8862a7a1601437774063d8118d9cae3",
"image_digest": "sha256:dde78ac2f0be5b7e637ddceba1a7c30fd832c2a50dd3e728acbf187f86354cf4"
},
"card": {
"description": "Diffusion Planner v5.0 (Autoware diffusion_planner): the network Autoware deploys (autoware_diffusion_planner, weights AutowareFoundation/diffusion_planner v5.0) running on one Tenstorrent Blackhole p150 via tt-nn, the whole plan (scene encoder, 11 DiT evaluations of the DPM-Solver++(2M) loop, turn-indicator head) as one metal trace per agent bucket: the Autoware planner tensors (ego and neighbour histories, lanes, route, polygons, line strings, goal, ego shape, turn-indicator history) in, an 8 s ego trajectory, predicted paths of the valid neighbours and a turn-indicator command out.\n",
"quickstart": "The first start compiles the kernels and captures the 6 traces (minutes on a cold cache, seconds later); the\nserver is ready when its log shows `Application startup complete`. Build a request from a file with\n`python3 code/tt_diffusion_planner/server/client.py --inputs <scene.npz> --out req.json`, then\n`curl -s localhost:20000/predict -H 'Content-Type: application/json' -d @req.json`.\n",
"architecture": "Six MLP-Mixer entity encoders + small MLP encoders, a 6-block transformer fusion encoder (564 scene tokens), a 3-block DiT decoder with adaLN (self-attention over 321 agents, cross-attention to the scene) evaluated 11 times by a DPM-Solver++(2M) loop, and a linear turn-indicator head (three ONNX graphs, 14.55 M parameters)",
"status": "Optimized release (5 optimization rounds: 102.0 -> 20.0 ms per plan on the device for the shipped sample, 5.1x). Community port, validated by agreement with the fp32 CPU reference on 99 scenes (2 shipped samples, 5 research scenes, 92 nuScenes v1.0-mini planning instants) with the same frozen gates as the first release.\n",
"intended_use": "Running and benchmarking the network of Autoware's diffusion planner node on one Blackhole p150, fed with the planner tensors the node builds from ROS messages and the Lanelet2 map (that conversion stays with the client).\n",
"out_of_scope_use": "Safety-critical driving decisions (this is not a certified Autoware component); multi-chip meshes; batch > 1; inputs outside the Autoware sensor configuration the weights were trained for. Closed-loop vehicle control; the node's guidance services (start / stop / centerline guidance).\n",
"usage": "`POST /predict` (JSON): `inputs` (the 15 raw tensors of the Autoware node's `create_input_data()` in the ego frame, before normalization: a base64 `.npz` or `{\"format\": \"json\", \"arrays\": {...}}`); optional `params` (`velocity_smoothing_window` 8, `stopping_threshold` 0.3, `turn_indicator_keep_offset` -1.25, `return_denoising_steps` false), `output_format`. `GET /health`, `GET /info`, `GET /v1/models` (stub).\nFull request / response contract: `SERVING.md` section 3.\n",
"performance": "| metric | value |\n|---|---|\n| back-to-back trace replays (device time per plan), shipped sample (88 neighbours, bucket r96) | 20.00 ms (50.0 plans/s; first release 102.04 ms) |\n| back-to-back trace replays by agent bucket: r32 / r64 / r96 / r128 / r192 / full | 17.44 / 19.25 / 20.00 / 21.11 / 22.30 / 26.65 ms |\n| one blocking plan (device trace: encoder + 11 DiT evaluations + solver + turn head), shipped sample | 20.05 ms (first release 102.13 ms) |\n| Python `model()` call, shipped sample | 26.7 ms p50 (first release 117.9 ms) |\n| served `/predict` `timing_ms.total` (uvicorn on the host) | 37.2 ms median (first release 123.1 ms) |\n| agreement with the fp32 CPU reference, 99 scenes | ego max 0.346 m / mean 0.154 m worst case (gates 1.0 / 0.3 m), turn command 99 / 99 |\n| configuration | ETH dispatch, 1 CQ, 12x10 grid, one p150; the pinned numerics |\n",
"limitations": "- Optimized release: fused custom kernels (fp32 LayerNorm, K-concatenated split matmuls, fp32 attention), L1-resident decoder intermediates and exact agent-bucket compaction (one trace per bucket); no megakernel ships yet. Two gated precision changes moved the 99-scene worst case from 0.313 / 0.143 m to 0.346 / 0.154 m (OPT_REPORT.md).\n- Stateless API: the node's turn-indicator hold window, the RTC prefix / temperature of the initial solver state and the agent / ego histories are the client's; converting ROS messages and the Lanelet2 map into the 15 input tensors is the client's; guidance services off.\n- Batch 1, one plan per request; the v5.0 capacities (320 neighbours, 140 lanes, 25 route lanes, 10 polygons, 60 line strings) and 10 DPM-Solver steps are compiled into the traces; the device time depends on the neighbour count (agent bucket, 17.4-26.7 ms).\n- Accuracy is agreement with the fp32 CPU reference (99 scenes); no dataset-level planning metric: the paper's nuPlan closed-loop benchmark was not run, and the weights were trained on non-public TIER IV data.\n- ETH dispatch, single chip: no multi-chip mesh; WORKER dispatch (11x10) is an A/B opt-in.\n- Not a certified Autoware component; not for safety-critical driving decisions.\n",
"risks": "Device numerics (split hi / lo matmuls, fp32 LayerNorm and attention, bf16 weights elsewhere) move the plan slightly from the fp32 CPU reference: up to 0.31 m (max) / 0.14 m (mean over the 8 s) on the worst of 99 scenes, a few cm on most; the sensitivity is chaotic per scene, so any numerics change must be re-checked on all scenes. The driving quality is the weights' (trained by TIER IV on non-public data): on nuScenes, a domain they never saw, the plans are plausible but conservative. A client that skips the node's state (turn-indicator hold window, agent and ego histories, RTC prefix) gets different behaviour than the Autoware node.\n",
"licensing": "- Weights: [AutowareFoundation/diffusion_planner](https://huggingface.co/AutowareFoundation/diffusion_planner) at tag `v5.0` (commit `423efde67f5414734da43a7ad856c17ceb8b51aa`), Apache-2.0 per its model card; not redistributed here. The upstream card states that TIER IV trained the models on TIER IV synthetic and real driving data; the dataset composition is not publicly documented.\n- Pre-/post-processing ported from autoware_universe `planning/autoware_diffusion_planner` @ `9ceaccf026c31ffc5319bc9eeb4bd7bede0af3fd` (Apache-2.0).\n- Port and serving code (`code/`): Apache-2.0.\n",
"related": "Other Autoware models on Blackhole: the \"Autoware\" collection of https://huggingface.co/changh95\n",
"license": {
"id": "apache-2.0",
"name": null,
"link": null
},
"pipeline_tag": "robotics",
"base_model": [
"AutowareFoundation/diffusion_planner"
]
}
}
}