// SPDX-License-Identifier: Apache-2.0 // Split-matmul operand build (tt/kcat_kernel.py), reader (RISCV_0): this core's contiguous range of x tiles. // CT args: [0] per-core RT-arg count (P3), then the TensorAccessorArgs of x. // Common RT args: [x_addr]. Per-core RT args: [t0, n] (flattened x tile indices). #include #include "api/dataflow/dataflow_api.h" constexpr auto x_args = TensorAccessorArgs<1>(); constexpr uint32_t cb_x = 0; constexpr uint32_t TB = 4096; void kernel_main() { const uint32_t x_addr = get_common_arg_val(0); const uint32_t t0 = get_arg_val(0); const uint32_t n = get_arg_val(1); const auto x = TensorAccessor(x_args, x_addr, TB); for (uint32_t t = t0; t < t0 + n; ++t) { cb_reserve_back(cb_x, 1); noc_async_read(x.get_noc_addr(t), get_write_ptr(cb_x), TB); noc_async_read_barrier(); cb_push_back(cb_x, 1); } }