// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. // SPDX-License-Identifier: Apache-2.0 // // SuperPoint NMS fold on 5 RISCs per core (SP_NMS_FOLD5), data movement: PROC 0 reads this core's S tiles (all its // rows lie in one cell row and one tile column) into CB_S and pushes them; PROC 1 waits for them; both then fold their // rows (nms_fold5_row.inc, prepended); the TRISCs fold the others (nms_fold5_trisc.cpp). // RT args: s_addr, p_addr, y0. CT args: PROC, cb_s, WC, TCOLS, SW, PAD, ROWS, NT, S accessor. #include #include "api/dataflow/dataflow_api.h" void kernel_main() { const uint32_t s_addr = get_arg_val(0); const uint32_t p_addr = get_arg_val(1); const uint32_t y0 = get_arg_val(2); constexpr uint32_t PROC = get_compile_time_arg_val(0); constexpr uint32_t cb_s = get_compile_time_arg_val(1); constexpr uint32_t WC = get_compile_time_arg_val(2); constexpr uint32_t TCOLS = get_compile_time_arg_val(3); constexpr uint32_t SW = get_compile_time_arg_val(4); constexpr uint32_t PAD = get_compile_time_arg_val(5); constexpr uint32_t ROWS = get_compile_time_arg_val(6); constexpr uint32_t NT = get_compile_time_arg_val(7); constexpr auto s_args = TensorAccessorArgs<8>(); if constexpr (PROC == 0) { const auto s = TensorAccessor(s_args, s_addr, 2048); const uint32_t cy = y0 >> 3, i = y0 & 7; const uint32_t tc = (i * 8) >> 5; const uint32_t r0 = cy * WC; const uint32_t tr0 = r0 >> 5, tr1 = (r0 + WC - 1) >> 5; cb_reserve_back(cb_s, NT); const uint32_t l1 = get_write_ptr(cb_s); for (uint32_t tr = tr0; tr <= tr1; ++tr) { noc_async_read(s.get_noc_addr(tr * TCOLS + tc), l1 + (tr - tr0) * 2048, 2048); } noc_async_read_barrier(); cb_push_back(cb_s, NT); } cb_wait_front(cb_s, NT); fold5_rows(PROC, get_read_ptr(cb_s), p_addr, y0); }