Download code/kernels/sp_nms/kp_compact3.cpp from changh95/superpoint-p150: direct link, hf CLI and curl.
- Browser
- Download file 6.16 kB
-
https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/kp_compact3.cpp
- Command line
-
hf download hf://changh95/superpoint-p150/code/kernels/sp_nms/kp_compact3.cpp
-
curl -L -o kp_compact3.cpp https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/kp_compact3.cpp
6.16 kB
| // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. | |
| // SPDX-License-Identifier: Apache-2.0 | |
| // | |
| // SuperPoint keypoint list on device, DMA-only version of kp_compact2.cpp (same HDR output format). | |
| // The NMS unfold kernel (nms_unfold_kp.cpp, KP_REC) already wrote every candidate as a final 4-word | |
| // header entry into its RISC's slot of the records tensor R (L1 of the NMS core, slot s = 2 core + risc: | |
| // [count, 0, 0, 0] + CAP x 16 B). This kernel runs on both data-movement RISCs of core (0, 0): each | |
| // computes the kept prefix of every slot from the candidate counts (the candidate tensor C, local L1), | |
| // then gathers the entries of its half of the slots with one NoC read per slot straight into the shared | |
| // L1 header image; PROC 1 raises semaphore 0, PROC 0 finishes the header (zero rows up to a multiple of | |
| // 32, counts) and writes it to DRAM (and, KPC_HR, into the descriptor bucket, as kp_compact2.cpp). | |
| // Runtime args: c_addr, hdr_addr, rec_addr, prm_addr (KPC_HR), NB bucket addresses (KPC_HR), then | |
| // NCORE words (noc_x << 16 | noc_y) of the NMS cores (slot s -> core s / 2). | |
| void kernel_main() { | |
| const uint32_t c_addr = get_arg_val<uint32_t>(0); | |
| const uint32_t hdr_addr = get_arg_val<uint32_t>(1); | |
| const uint32_t rec_addr = get_arg_val<uint32_t>(2); | |
| constexpr uint32_t cb_scratch = get_compile_time_arg_val(0); | |
| constexpr uint32_t NSLOT = get_compile_time_arg_val(1); | |
| constexpr uint32_t CAP = get_compile_time_arg_val(2); | |
| constexpr uint32_t KMAX = get_compile_time_arg_val(3); | |
| constexpr uint32_t PROC = get_compile_time_arg_val(4); | |
| constexpr uint32_t SLOT_BYTES = (CAP + 1) * 4; | |
| constexpr uint32_t REC_BYTES = (CAP + 1) * 16; | |
| constexpr uint32_t HDR_WORDS = 16 + 4 * KMAX; | |
| constexpr auto hdr_args = TensorAccessorArgs<5>(); | |
| const auto hdracc = TensorAccessor(hdr_args, hdr_addr, HDR_WORDS * 4); | |
| constexpr auto prm_args = TensorAccessorArgs<hdr_args.next_compile_time_args_offset()>(); | |
| constexpr auto bk_args = TensorAccessorArgs<prm_args.next_compile_time_args_offset()>(); | |
| constexpr uint32_t ROWB = KPC_C * 4; | |
| constexpr uint32_t NB = KMAX / KPC_BSTEP; | |
| constexpr uint32_t CORE_ARG0 = 4 + NB; | |
| constexpr uint32_t CORE_ARG0 = 3; | |
| const uint32_t hdr_l1 = get_write_ptr(cb_scratch); | |
| const uint32_t prm_l1 = hdr_l1 + HDR_WORDS * 4; | |
| if constexpr (PROC == 0) { | |
| const auto prmacc = TensorAccessor(prm_args, get_arg_val<uint32_t>(3), 64); | |
| noc_async_read(prmacc.get_noc_addr(0), prm_l1, 64); | |
| } | |
| // PROC 0 owns slots [0, NSLOT/2), PROC 1 [NSLOT/2, NSLOT): counts into a local (RISC-private, fast) | |
| // array, PROC 0 publishes its kept sum (semaphore 1) = PROC 1's first entry index | |
| constexpr uint32_t HALF = NSLOT / 2; | |
| const uint32_t s_begin = PROC == 0 ? 0 : HALF, s_end = PROC == 0 ? HALF : NSLOT; | |
| volatile tt_l1_ptr uint32_t* xch = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(hdr_l1 + HDR_WORDS * 4 + 64); | |
| uint16_t cnts[NSLOT - HALF]; | |
| uint32_t total = 0, overflow = 0, sum = 0; | |
| for (uint32_t s = s_begin; s < s_end; ++s) { | |
| uint32_t cnt = reinterpret_cast<volatile uint32_t*>(c_addr + s * SLOT_BYTES)[0]; | |
| total += cnt; | |
| if (cnt > CAP) { | |
| overflow = 1; | |
| cnt = CAP; | |
| } | |
| cnts[s - s_begin] = (uint16_t)cnt; | |
| sum += cnt; | |
| } | |
| volatile tt_l1_ptr uint32_t* sem1 = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(1)); | |
| uint32_t n = 0; | |
| if constexpr (PROC == 0) { | |
| xch[0] = sum; | |
| *sem1 = 1; | |
| } else { | |
| noc_semaphore_wait(sem1, 1); | |
| *sem1 = 0; | |
| n = xch[0] < KMAX ? xch[0] : KMAX; | |
| } | |
| const uint32_t out_l1 = hdr_l1 + 64; | |
| for (uint32_t s = s_begin; s < s_end && n < KMAX; ++s) { | |
| uint32_t m = cnts[s - s_begin]; | |
| if (m == 0) { | |
| continue; | |
| } | |
| if (m > KMAX - n) { | |
| m = KMAX - n; | |
| } | |
| const uint32_t xy = get_arg_val<uint32_t>(CORE_ARG0 + (s >> 1)); | |
| const uint64_t src = get_noc_addr(xy >> 16, xy & 0xFFFF, rec_addr + (s & 1) * REC_BYTES + 16); | |
| noc_async_read(src, out_l1 + 16 * n, 16 * m); | |
| n += m; | |
| } | |
| noc_async_read_barrier(); | |
| volatile tt_l1_ptr uint32_t* sem = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(0)); | |
| if constexpr (PROC == 1) { | |
| xch[1] = total; | |
| xch[2] = overflow; | |
| xch[3] = sum; | |
| *sem = 1; | |
| return; | |
| } else { | |
| noc_semaphore_wait(sem, 1); | |
| *sem = 0; | |
| total += xch[1]; | |
| overflow |= xch[2]; | |
| const uint32_t ksum = sum + xch[3]; | |
| const uint32_t kept = ksum < KMAX ? ksum : KMAX; | |
| uint32_t* hdr = reinterpret_cast<uint32_t*>(hdr_l1); | |
| uint32_t* out = hdr + 16; | |
| const uint32_t nr = (kept + 31) & ~31u; | |
| for (uint32_t k = 4 * kept; k < 4 * nr; ++k) { | |
| out[k] = 0; | |
| } | |
| hdr[0] = total; | |
| hdr[1] = overflow; | |
| hdr[2] = kept; | |
| uint32_t b = kept == 0 ? 0 : (kept + KPC_BSTEP - 1) / KPC_BSTEP - 1; | |
| const uint32_t spec = reinterpret_cast<volatile uint32_t*>(prm_l1)[2]; | |
| if (spec > b) { | |
| b = spec; | |
| } | |
| if (b > NB - 1) { | |
| b = NB - 1; | |
| } | |
| hdr[3] = b; | |
| noc_async_write(hdr_l1, hdracc.get_noc_addr(0), (16 + 4 * nr) * 4); | |
| { | |
| const auto bacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + b), ROWB); | |
| const uint32_t bytes = (16 + 4 * nr) * 4; | |
| for (uint32_t p = 0, off = 0; off < bytes; ++p, off += ROWB) { | |
| const uint32_t sz = bytes - off < ROWB ? bytes - off : ROWB; | |
| noc_async_write(hdr_l1 + off, bacc.get_noc_addr(p), sz); | |
| } | |
| if (b != spec && spec < NB) { | |
| const auto sacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + spec), ROWB); | |
| noc_async_write(hdr_l1, sacc.get_noc_addr(0), 64); | |
| } | |
| } | |
| noc_async_write_barrier(); | |
| } | |
| } | |