// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. // SPDX-License-Identifier: Apache-2.0 // // Descriptor head (1x1 conv + L2 norm + untilize, models/tt/desc_head.py), PROC 0: publish the local // input shard (CB_IN) and the core's L1-resident copy of the 1x1 weights + bias (CB_W, both bound to // sharded tensors), and build the reduce scaler tile (1.0 in row 0 of each face). #include #include "api/dataflow/dataflow_api.h" void kernel_main() { constexpr uint32_t cb_in = get_compile_time_arg_val(0); constexpr uint32_t cb_one = get_compile_time_arg_val(1); constexpr uint32_t NT = get_compile_time_arg_val(2); constexpr uint32_t cb_w = get_compile_time_arg_val(3); constexpr uint32_t NW = get_compile_time_arg_val(4); cb_reserve_back(cb_in, NT); cb_push_back(cb_in, NT); cb_reserve_back(cb_w, NW); cb_push_back(cb_w, NW); cb_reserve_back(cb_one, 1); volatile tt_l1_ptr uint32_t* p = reinterpret_cast(get_write_ptr(cb_one)); for (uint32_t i = 0; i < 512; ++i) { p[i] = 0; } for (uint32_t f = 0; f < 4; ++f) { for (uint32_t j = 0; j < 8; ++j) { p[f * 128 + j] = 0x3F803F80u; } } (void)p[511]; cb_push_back(cb_one, 1); }