Download code/kernels/sp_nms/nms_fold5_dm.cpp from changh95/superpoint-p150: direct link, hf CLI and curl.
- Browser
- Download file 1.97 kB
-
https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/nms_fold5_dm.cpp
- Command line
-
hf download hf://changh95/superpoint-p150/code/kernels/sp_nms/nms_fold5_dm.cpp
-
curl -L -o nms_fold5_dm.cpp https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/nms_fold5_dm.cpp
1.97 kB
| // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. | |
| // SPDX-License-Identifier: Apache-2.0 | |
| // | |
| // SuperPoint NMS fold on 5 RISCs per core (SP_NMS_FOLD5), data movement: PROC 0 reads this core's S tiles (all its | |
| // rows lie in one cell row and one tile column) into CB_S and pushes them; PROC 1 waits for them; both then fold their | |
| // rows (nms_fold5_row.inc, prepended); the TRISCs fold the others (nms_fold5_trisc.cpp). | |
| // RT args: s_addr, p_addr, y0. CT args: PROC, cb_s, WC, TCOLS, SW, PAD, ROWS, NT, S accessor. | |
| void kernel_main() { | |
| const uint32_t s_addr = get_arg_val<uint32_t>(0); | |
| const uint32_t p_addr = get_arg_val<uint32_t>(1); | |
| const uint32_t y0 = get_arg_val<uint32_t>(2); | |
| constexpr uint32_t PROC = get_compile_time_arg_val(0); | |
| constexpr uint32_t cb_s = get_compile_time_arg_val(1); | |
| constexpr uint32_t WC = get_compile_time_arg_val(2); | |
| constexpr uint32_t TCOLS = get_compile_time_arg_val(3); | |
| constexpr uint32_t SW = get_compile_time_arg_val(4); | |
| constexpr uint32_t PAD = get_compile_time_arg_val(5); | |
| constexpr uint32_t ROWS = get_compile_time_arg_val(6); | |
| constexpr uint32_t NT = get_compile_time_arg_val(7); | |
| constexpr auto s_args = TensorAccessorArgs<8>(); | |
| if constexpr (PROC == 0) { | |
| const auto s = TensorAccessor(s_args, s_addr, 2048); | |
| const uint32_t cy = y0 >> 3, i = y0 & 7; | |
| const uint32_t tc = (i * 8) >> 5; | |
| const uint32_t r0 = cy * WC; | |
| const uint32_t tr0 = r0 >> 5, tr1 = (r0 + WC - 1) >> 5; | |
| cb_reserve_back(cb_s, NT); | |
| const uint32_t l1 = get_write_ptr(cb_s); | |
| for (uint32_t tr = tr0; tr <= tr1; ++tr) { | |
| noc_async_read(s.get_noc_addr(tr * TCOLS + tc), l1 + (tr - tr0) * 2048, 2048); | |
| } | |
| noc_async_read_barrier(); | |
| cb_push_back(cb_s, NT); | |
| } | |
| cb_wait_front(cb_s, NT); | |
| fold5_rows<WC, SW, PAD, ROWS>(PROC, get_read_ptr(cb_s), p_addr, y0); | |
| } | |