File size: 1,608 Bytes
c699c4c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// Descriptor norm, PROC 1: write each untilized [32, 256] block as 32 rows (512 B pages) of the
// row-major output (interleaved). RT args: out_addr, first row of this core.
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"

void kernel_main() {
    const uint32_t out_addr = get_arg_val<uint32_t>(0);
    const uint32_t row0 = get_arg_val<uint32_t>(1);
    constexpr uint32_t cb_out = get_compile_time_arg_val(0);
    constexpr uint32_t TR = get_compile_time_arg_val(1);
    constexpr uint32_t WT = 8, ROWB = 512;
    constexpr auto o_args = TensorAccessorArgs<2>();
    const auto oacc = TensorAccessor(o_args, out_addr, ROWB);
#ifdef DH_SIGNAL_CB
    // merged head + device softmax cores (SP_HEAD_SM=1): once the compute kernel has pushed score tile row r
    // into the logits shard, increment semaphore 0 of the softmax core that owns it (RT arg 2 + r: noc x << 16 | y)
    for (uint32_t r = 0; r < TR; ++r) {
        cb_wait_front(DH_SIGNAL_CB, 3 * (r + 1));
        const uint32_t xy = get_arg_val<uint32_t>(2 + r);
        noc_semaphore_inc(get_noc_addr(xy >> 16, xy & 0xFFFF, get_semaphore(0)), 1);
    }
    noc_async_atomic_barrier();
#endif
    for (uint32_t r = 0; r < TR; ++r) {
        cb_wait_front(cb_out, WT);
        const uint32_t l1 = get_read_ptr(cb_out);
        for (uint32_t i = 0; i < 32; ++i) {
            noc_async_write(l1 + i * ROWB, oacc.get_noc_addr(row0 + r * 32 + i), ROWB);
        }
        noc_async_write_barrier();
        cb_pop_front(cb_out, WT);
    }
}