superpoint-p150 / code /kernels /sp_desc /dn_writer.cpp
changh95's picture
Optimized build (2026-10-03): trace 3.73 -> 0.50 ms
c699c4c verified
Raw History Blame Contribute Delete
1.61 kB
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// Descriptor norm, PROC 1: write each untilized [32, 256] block as 32 rows (512 B pages) of the
// row-major output (interleaved). RT args: out_addr, first row of this core.
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
const uint32_t out_addr = get_arg_val<uint32_t>(0);
const uint32_t row0 = get_arg_val<uint32_t>(1);
constexpr uint32_t cb_out = get_compile_time_arg_val(0);
constexpr uint32_t TR = get_compile_time_arg_val(1);
constexpr uint32_t WT = 8, ROWB = 512;
constexpr auto o_args = TensorAccessorArgs<2>();
const auto oacc = TensorAccessor(o_args, out_addr, ROWB);
#ifdef DH_SIGNAL_CB
// merged head + device softmax cores (SP_HEAD_SM=1): once the compute kernel has pushed score tile row r
// into the logits shard, increment semaphore 0 of the softmax core that owns it (RT arg 2 + r: noc x << 16 | y)
for (uint32_t r = 0; r < TR; ++r) {
cb_wait_front(DH_SIGNAL_CB, 3 * (r + 1));
const uint32_t xy = get_arg_val<uint32_t>(2 + r);
noc_semaphore_inc(get_noc_addr(xy >> 16, xy & 0xFFFF, get_semaphore(0)), 1);
}
noc_async_atomic_barrier();
#endif
for (uint32_t r = 0; r < TR; ++r) {
cb_wait_front(cb_out, WT);
const uint32_t l1 = get_read_ptr(cb_out);
for (uint32_t i = 0; i < 32; ++i) {
noc_async_write(l1 + i * ROWB, oacc.get_noc_addr(row0 + r * 32 + i), ROWB);
}
noc_async_write_barrier();
cb_pop_front(cb_out, WT);
}
}