File size: 1,608 Bytes
c699c4c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 | // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// Descriptor norm, PROC 1: write each untilized [32, 256] block as 32 rows (512 B pages) of the
// row-major output (interleaved). RT args: out_addr, first row of this core.
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
const uint32_t out_addr = get_arg_val<uint32_t>(0);
const uint32_t row0 = get_arg_val<uint32_t>(1);
constexpr uint32_t cb_out = get_compile_time_arg_val(0);
constexpr uint32_t TR = get_compile_time_arg_val(1);
constexpr uint32_t WT = 8, ROWB = 512;
constexpr auto o_args = TensorAccessorArgs<2>();
const auto oacc = TensorAccessor(o_args, out_addr, ROWB);
#ifdef DH_SIGNAL_CB
// merged head + device softmax cores (SP_HEAD_SM=1): once the compute kernel has pushed score tile row r
// into the logits shard, increment semaphore 0 of the softmax core that owns it (RT arg 2 + r: noc x << 16 | y)
for (uint32_t r = 0; r < TR; ++r) {
cb_wait_front(DH_SIGNAL_CB, 3 * (r + 1));
const uint32_t xy = get_arg_val<uint32_t>(2 + r);
noc_semaphore_inc(get_noc_addr(xy >> 16, xy & 0xFFFF, get_semaphore(0)), 1);
}
noc_async_atomic_barrier();
#endif
for (uint32_t r = 0; r < TR; ++r) {
cb_wait_front(cb_out, WT);
const uint32_t l1 = get_read_ptr(cb_out);
for (uint32_t i = 0; i < 32; ++i) {
noc_async_write(l1 + i * ROWB, oacc.get_noc_addr(row0 + r * 32 + i), ROWB);
}
noc_async_write_barrier();
cb_pop_front(cb_out, WT);
}
}
|