// SPDX-License-Identifier: Apache-2.0 // Pixel-shuffle reader for a column block of a wider TILE matrix (moge-2 opt kernel, round 9): work unit u = // (tile row t = u / 2, dy = u % 2) reads the WT tiles [COL0 + dy * WT, +WT) of tile row t of a matrix with // ROW_TILES tiles per tile row into c_0 (the stream's [HW, (dy, dx, o)] block of the combined level-1 fold // matmul output). Runtime args: src_addr, num_units, start_unit. Compile args: WT, ROW_TILES, COL0, // TensorAccessorArgs. #include "api/dataflow/dataflow_api.h" #include "api/dataflow/noc.h" #include "api/dataflow/dataflow_buffer.h" #include "api/tensor/noc_traits.h" void kernel_main() { const uint32_t src_addr = get_arg_val(0); const uint32_t nunits = get_arg_val(1); const uint32_t u0 = get_arg_val(2); constexpr uint32_t WT = get_compile_time_arg_val(0); constexpr uint32_t ROW_TILES = get_compile_time_arg_val(1); constexpr uint32_t COL0 = get_compile_time_arg_val(2); constexpr auto src_args = TensorAccessorArgs<3>(); constexpr uint32_t cb_id = 0; const uint32_t page_bytes = get_local_cb_interface(cb_id).fifo_page_size; const auto s = TensorAccessor(src_args, src_addr); DataflowBuffer dfb(cb_id); for (uint32_t u = u0; u < u0 + nunits; ++u) { const uint32_t base = (u >> 1) * ROW_TILES + COL0 + (u & 1) * WT; dfb.reserve_back(WT); const uint32_t l = dfb.get_write_ptr(); for (uint32_t j = 0; j < WT; ++j) { noc_async_read(s.get_noc_addr(base + j), l + j * page_bytes, page_bytes); } noc_async_read_barrier(); dfb.push_back(WT); } }