Download code/kernels/sp_nms/nms_unfold_kp.cpp from changh95/superpoint-p150: direct link, hf CLI and curl.
- Browser
- Download file 2.27 kB
-
https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/nms_unfold_kp.cpp
- Command line
-
hf download hf://changh95/superpoint-p150/code/kernels/sp_nms/nms_unfold_kp.cpp
-
curl -L -o nms_unfold_kp.cpp https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/nms_unfold_kp.cpp
2.27 kB
| // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. | |
| // SPDX-License-Identifier: Apache-2.0 | |
| // | |
| // SuperPoint NMS, step 3 + keypoint candidate compaction (data movement only, ttnn.generic_op). | |
| // M = 9x9 window max in strip layout [H, SW, 32] (this core's L1 shard), P = padded strip scores | |
| // [H, SW + 2*PAD, 32] (this core's L1 shard). Writes the natural dense NMS map row by row: | |
| // out[y, x] = P[y, xw + PAD, l] if it equals M[y, xw, l] else 0, x = l*SW + xw | |
| // (bit-identical to where(s == maxpool9x9(s), s, 0) for s >= 0). out: [H, W] ROW_MAJOR interleaved. | |
| // | |
| // Additionally every RISC appends the keypoint candidates of its rows, in raster order, to its own | |
| // slot of the candidate tensor C [NSLOT, CAP + 1] uint32 (one DRAM page per slot): | |
| // C[slot, 0] = number of candidates (may exceed CAP -> overflow, the host falls back), | |
| // C[slot, 1 + k] = (bf16 score bits << 16) | ((y - y_first) * W + x) | |
| // with the host's extract_keypoints rule: BORDER <= y < H - BORDER, BORDER <= x < W - BORDER and | |
| // float(score) > threshold <=> int16(score bits) > int16(THR) (THR = fp32 bits of the threshold | |
| // >> 16; exact for bf16 scores, see postprocess._bf16_threshold_bits). | |
| // THR and BORDER are per-request values (taxonomy class C): read from the 64-byte parameter tensor | |
| // PRM = [THR, BORDER, 0...] (uint32, DRAM) that the host rewrites only when they change. | |
| // KP_REC (records for the DMA-only compaction kp_compact3.cpp): every RISC also writes its candidates | |
| // as final 4-word keypoint-header entries into its slot of the records tensor R (L1, this core's shard, | |
| // slot = PROC: 16 B count word + CAP x 16 B): ((y << 16) | x, score bits, (y0c*wc << 16) | y1c*wc, | |
| // (x0c << 16) | x1c); the cell indices of the bilinear taps are q = floor((2v - 7)(c - 1) / (16c - 9)), | |
| // v0c = clamp(q), v1c = clamp(q + 1) (the host asserts this equals postprocess.SampleTables). | |
| // Extra RT arg 9: records address. Defines: KP_WC, KP_HC (cells per row / column). | |
| // (the unfold itself is nms_unfold_fn.inc, prepended to this source by models/tt/nms_kernels.py) | |
| void kernel_main() { | |
| constexpr uint32_t SW = get_compile_time_arg_val(1); | |
| unfold_rows<0, get_compile_time_arg_val(0), SW * 16>(reinterpret_cast<const uint32_t*>(get_arg_val<uint32_t>(0)), 0); | |
| } | |