// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. // SPDX-License-Identifier: Apache-2.0 // // Descriptor norm, PROC 1: write each untilized [32, 256] block as 32 rows (512 B pages) of the // row-major output (interleaved). RT args: out_addr, first row of this core. #include #include "api/dataflow/dataflow_api.h" void kernel_main() { const uint32_t out_addr = get_arg_val(0); const uint32_t row0 = get_arg_val(1); constexpr uint32_t cb_out = get_compile_time_arg_val(0); constexpr uint32_t TR = get_compile_time_arg_val(1); constexpr uint32_t WT = 8, ROWB = 512; constexpr auto o_args = TensorAccessorArgs<2>(); const auto oacc = TensorAccessor(o_args, out_addr, ROWB); #ifdef DH_SIGNAL_CB // merged head + device softmax cores (SP_HEAD_SM=1): once the compute kernel has pushed score tile row r // into the logits shard, increment semaphore 0 of the softmax core that owns it (RT arg 2 + r: noc x << 16 | y) for (uint32_t r = 0; r < TR; ++r) { cb_wait_front(DH_SIGNAL_CB, 3 * (r + 1)); const uint32_t xy = get_arg_val(2 + r); noc_semaphore_inc(get_noc_addr(xy >> 16, xy & 0xFFFF, get_semaphore(0)), 1); } noc_async_atomic_barrier(); #endif for (uint32_t r = 0; r < TR; ++r) { cb_wait_front(cb_out, WT); const uint32_t l1 = get_read_ptr(cb_out); for (uint32_t i = 0; i < 32; ++i) { noc_async_write(l1 + i * ROWB, oacc.get_noc_addr(row0 + r * 32 + i), ROWB); } noc_async_write_barrier(); cb_pop_front(cb_out, WT); } }