moge-2-p150 / code /tt_moge /kernels /pixshuf_reader.cpp
changh95's picture
Optimized build (2026-10-03): model call 19.0 ms, trace 17.7 ms
13b4736 verified
Raw History Blame Contribute Delete
1.66 kB
// SPDX-License-Identifier: Apache-2.0
// Pixel-shuffle reader for a column block of a wider TILE matrix (moge-2 opt kernel, round 9): work unit u =
// (tile row t = u / 2, dy = u % 2) reads the WT tiles [COL0 + dy * WT, +WT) of tile row t of a matrix with
// ROW_TILES tiles per tile row into c_0 (the stream's [HW, (dy, dx, o)] block of the combined level-1 fold
// matmul output). Runtime args: src_addr, num_units, start_unit. Compile args: WT, ROW_TILES, COL0,
// TensorAccessorArgs.
#include "api/dataflow/dataflow_api.h"
#include "api/dataflow/noc.h"
#include "api/dataflow/dataflow_buffer.h"
#include "api/tensor/noc_traits.h"
void kernel_main() {
const uint32_t src_addr = get_arg_val<uint32_t>(0);
const uint32_t nunits = get_arg_val<uint32_t>(1);
const uint32_t u0 = get_arg_val<uint32_t>(2);
constexpr uint32_t WT = get_compile_time_arg_val(0);
constexpr uint32_t ROW_TILES = get_compile_time_arg_val(1);
constexpr uint32_t COL0 = get_compile_time_arg_val(2);
constexpr auto src_args = TensorAccessorArgs<3>();
constexpr uint32_t cb_id = 0;
const uint32_t page_bytes = get_local_cb_interface(cb_id).fifo_page_size;
const auto s = TensorAccessor(src_args, src_addr);
DataflowBuffer dfb(cb_id);
for (uint32_t u = u0; u < u0 + nunits; ++u) {
const uint32_t base = (u >> 1) * ROW_TILES + COL0 + (u & 1) * WT;
dfb.reserve_back(WT);
const uint32_t l = dfb.get_write_ptr();
for (uint32_t j = 0; j < WT; ++j) {
noc_async_read(s.get_noc_addr(base + j), l + j * page_bytes, page_bytes);
}
noc_async_read_barrier();
dfb.push_back(WT);
}
}