superpoint-p150 / code /kernels /sp_nms /nms_fold5_dm.cpp
changh95's picture
Optimized build (2026-10-03): trace 3.73 -> 0.50 ms
c699c4c verified
Raw History Blame Contribute Delete
1.97 kB
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// SuperPoint NMS fold on 5 RISCs per core (SP_NMS_FOLD5), data movement: PROC 0 reads this core's S tiles (all its
// rows lie in one cell row and one tile column) into CB_S and pushes them; PROC 1 waits for them; both then fold their
// rows (nms_fold5_row.inc, prepended); the TRISCs fold the others (nms_fold5_trisc.cpp).
// RT args: s_addr, p_addr, y0. CT args: PROC, cb_s, WC, TCOLS, SW, PAD, ROWS, NT, S accessor.
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
const uint32_t s_addr = get_arg_val<uint32_t>(0);
const uint32_t p_addr = get_arg_val<uint32_t>(1);
const uint32_t y0 = get_arg_val<uint32_t>(2);
constexpr uint32_t PROC = get_compile_time_arg_val(0);
constexpr uint32_t cb_s = get_compile_time_arg_val(1);
constexpr uint32_t WC = get_compile_time_arg_val(2);
constexpr uint32_t TCOLS = get_compile_time_arg_val(3);
constexpr uint32_t SW = get_compile_time_arg_val(4);
constexpr uint32_t PAD = get_compile_time_arg_val(5);
constexpr uint32_t ROWS = get_compile_time_arg_val(6);
constexpr uint32_t NT = get_compile_time_arg_val(7);
constexpr auto s_args = TensorAccessorArgs<8>();
if constexpr (PROC == 0) {
const auto s = TensorAccessor(s_args, s_addr, 2048);
const uint32_t cy = y0 >> 3, i = y0 & 7;
const uint32_t tc = (i * 8) >> 5;
const uint32_t r0 = cy * WC;
const uint32_t tr0 = r0 >> 5, tr1 = (r0 + WC - 1) >> 5;
cb_reserve_back(cb_s, NT);
const uint32_t l1 = get_write_ptr(cb_s);
for (uint32_t tr = tr0; tr <= tr1; ++tr) {
noc_async_read(s.get_noc_addr(tr * TCOLS + tc), l1 + (tr - tr0) * 2048, 2048);
}
noc_async_read_barrier();
cb_push_back(cb_s, NT);
}
cb_wait_front(cb_s, NT);
fold5_rows<WC, SW, PAD, ROWS>(PROC, get_read_ptr(cb_s), p_addr, y0);
}