File size: 1,661 Bytes
13b4736
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
// SPDX-License-Identifier: Apache-2.0
// Pixel-shuffle reader for a column block of a wider TILE matrix (moge-2 opt kernel, round 9): work unit u =
// (tile row t = u / 2, dy = u % 2) reads the WT tiles [COL0 + dy * WT, +WT) of tile row t of a matrix with
// ROW_TILES tiles per tile row into c_0 (the stream's [HW, (dy, dx, o)] block of the combined level-1 fold
// matmul output). Runtime args: src_addr, num_units, start_unit. Compile args: WT, ROW_TILES, COL0,
// TensorAccessorArgs.
#include "api/dataflow/dataflow_api.h"
#include "api/dataflow/noc.h"
#include "api/dataflow/dataflow_buffer.h"
#include "api/tensor/noc_traits.h"

void kernel_main() {
    const uint32_t src_addr = get_arg_val<uint32_t>(0);
    const uint32_t nunits = get_arg_val<uint32_t>(1);
    const uint32_t u0 = get_arg_val<uint32_t>(2);
    constexpr uint32_t WT = get_compile_time_arg_val(0);
    constexpr uint32_t ROW_TILES = get_compile_time_arg_val(1);
    constexpr uint32_t COL0 = get_compile_time_arg_val(2);
    constexpr auto src_args = TensorAccessorArgs<3>();
    constexpr uint32_t cb_id = 0;
    const uint32_t page_bytes = get_local_cb_interface(cb_id).fifo_page_size;
    const auto s = TensorAccessor(src_args, src_addr);
    DataflowBuffer dfb(cb_id);
    for (uint32_t u = u0; u < u0 + nunits; ++u) {
        const uint32_t base = (u >> 1) * ROW_TILES + COL0 + (u & 1) * WT;
        dfb.reserve_back(WT);
        const uint32_t l = dfb.get_write_ptr();
        for (uint32_t j = 0; j < WT; ++j) {
            noc_async_read(s.get_noc_addr(base + j), l + j * page_bytes, page_bytes);
        }
        noc_async_read_barrier();
        dfb.push_back(WT);
    }
}