moge-2-p150 / code /tt_moge /kernels /reader_batched.cpp
changh95's picture
Optimized build (2026-10-03): model call 19.0 ms, trace 17.7 ms
13b4736 verified
Raw History Blame Contribute Delete
1.32 kB
// SPDX-License-Identifier: Apache-2.0
// Interleaved page reader into c_0, BLOCK pages per NoC barrier (moge-2 opt kernel).
// Runtime args: src_addr, num_pages, start_id. Compile args: BLOCK, TensorAccessorArgs.
#include "api/dataflow/dataflow_api.h"
#include "api/dataflow/noc.h"
#include "api/dataflow/dataflow_buffer.h"
#include "api/tensor/noc_traits.h"
void kernel_main() {
const uint32_t src_addr = get_arg_val<uint32_t>(0);
const uint32_t num_pages = get_arg_val<uint32_t>(1);
const uint32_t start_id = get_arg_val<uint32_t>(2);
constexpr uint32_t BLOCK = get_compile_time_arg_val(0);
constexpr auto src_args = TensorAccessorArgs<1>();
constexpr uint32_t cb_id = 0;
const uint32_t page_bytes = get_local_cb_interface(cb_id).fifo_page_size;
const auto s = TensorAccessor(src_args, src_addr);
Noc noc;
DataflowBuffer dfb(cb_id);
uint32_t i = start_id;
const uint32_t end_id = start_id + num_pages;
while (i < end_id) {
const uint32_t n = (end_id - i) < BLOCK ? (end_id - i) : BLOCK;
dfb.reserve_back(n);
for (uint32_t j = 0; j < n; ++j) {
noc.async_read(s, dfb, page_bytes, {.page_id = i + j}, {.offset_bytes = j * page_bytes});
}
noc.async_read_barrier();
dfb.push_back(n);
i += n;
}
}