File size: 6,156 Bytes
c699c4c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 | // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// SuperPoint keypoint list on device, DMA-only version of kp_compact2.cpp (same HDR output format).
// The NMS unfold kernel (nms_unfold_kp.cpp, KP_REC) already wrote every candidate as a final 4-word
// header entry into its RISC's slot of the records tensor R (L1 of the NMS core, slot s = 2 core + risc:
// [count, 0, 0, 0] + CAP x 16 B). This kernel runs on both data-movement RISCs of core (0, 0): each
// computes the kept prefix of every slot from the candidate counts (the candidate tensor C, local L1),
// then gathers the entries of its half of the slots with one NoC read per slot straight into the shared
// L1 header image; PROC 1 raises semaphore 0, PROC 0 finishes the header (zero rows up to a multiple of
// 32, counts) and writes it to DRAM (and, KPC_HR, into the descriptor bucket, as kp_compact2.cpp).
// Runtime args: c_addr, hdr_addr, rec_addr, prm_addr (KPC_HR), NB bucket addresses (KPC_HR), then
// NCORE words (noc_x << 16 | noc_y) of the NMS cores (slot s -> core s / 2).
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
const uint32_t c_addr = get_arg_val<uint32_t>(0);
const uint32_t hdr_addr = get_arg_val<uint32_t>(1);
const uint32_t rec_addr = get_arg_val<uint32_t>(2);
constexpr uint32_t cb_scratch = get_compile_time_arg_val(0);
constexpr uint32_t NSLOT = get_compile_time_arg_val(1);
constexpr uint32_t CAP = get_compile_time_arg_val(2);
constexpr uint32_t KMAX = get_compile_time_arg_val(3);
constexpr uint32_t PROC = get_compile_time_arg_val(4);
constexpr uint32_t SLOT_BYTES = (CAP + 1) * 4;
constexpr uint32_t REC_BYTES = (CAP + 1) * 16;
constexpr uint32_t HDR_WORDS = 16 + 4 * KMAX;
constexpr auto hdr_args = TensorAccessorArgs<5>();
const auto hdracc = TensorAccessor(hdr_args, hdr_addr, HDR_WORDS * 4);
#ifdef KPC_HR
constexpr auto prm_args = TensorAccessorArgs<hdr_args.next_compile_time_args_offset()>();
constexpr auto bk_args = TensorAccessorArgs<prm_args.next_compile_time_args_offset()>();
constexpr uint32_t ROWB = KPC_C * 4;
constexpr uint32_t NB = KMAX / KPC_BSTEP;
constexpr uint32_t CORE_ARG0 = 4 + NB;
#else
constexpr uint32_t CORE_ARG0 = 3;
#endif
const uint32_t hdr_l1 = get_write_ptr(cb_scratch);
#ifdef KPC_HR
const uint32_t prm_l1 = hdr_l1 + HDR_WORDS * 4;
if constexpr (PROC == 0) {
const auto prmacc = TensorAccessor(prm_args, get_arg_val<uint32_t>(3), 64);
noc_async_read(prmacc.get_noc_addr(0), prm_l1, 64);
}
#endif
// PROC 0 owns slots [0, NSLOT/2), PROC 1 [NSLOT/2, NSLOT): counts into a local (RISC-private, fast)
// array, PROC 0 publishes its kept sum (semaphore 1) = PROC 1's first entry index
constexpr uint32_t HALF = NSLOT / 2;
const uint32_t s_begin = PROC == 0 ? 0 : HALF, s_end = PROC == 0 ? HALF : NSLOT;
volatile tt_l1_ptr uint32_t* xch = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(hdr_l1 + HDR_WORDS * 4 + 64);
uint16_t cnts[NSLOT - HALF];
uint32_t total = 0, overflow = 0, sum = 0;
for (uint32_t s = s_begin; s < s_end; ++s) {
uint32_t cnt = reinterpret_cast<volatile uint32_t*>(c_addr + s * SLOT_BYTES)[0];
total += cnt;
if (cnt > CAP) {
overflow = 1;
cnt = CAP;
}
cnts[s - s_begin] = (uint16_t)cnt;
sum += cnt;
}
volatile tt_l1_ptr uint32_t* sem1 = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(1));
uint32_t n = 0;
if constexpr (PROC == 0) {
xch[0] = sum;
*sem1 = 1;
} else {
noc_semaphore_wait(sem1, 1);
*sem1 = 0;
n = xch[0] < KMAX ? xch[0] : KMAX;
}
const uint32_t out_l1 = hdr_l1 + 64;
for (uint32_t s = s_begin; s < s_end && n < KMAX; ++s) {
uint32_t m = cnts[s - s_begin];
if (m == 0) {
continue;
}
if (m > KMAX - n) {
m = KMAX - n;
}
const uint32_t xy = get_arg_val<uint32_t>(CORE_ARG0 + (s >> 1));
const uint64_t src = get_noc_addr(xy >> 16, xy & 0xFFFF, rec_addr + (s & 1) * REC_BYTES + 16);
noc_async_read(src, out_l1 + 16 * n, 16 * m);
n += m;
}
noc_async_read_barrier();
volatile tt_l1_ptr uint32_t* sem = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(0));
if constexpr (PROC == 1) {
xch[1] = total;
xch[2] = overflow;
xch[3] = sum;
*sem = 1;
return;
} else {
noc_semaphore_wait(sem, 1);
*sem = 0;
total += xch[1];
overflow |= xch[2];
const uint32_t ksum = sum + xch[3];
const uint32_t kept = ksum < KMAX ? ksum : KMAX;
uint32_t* hdr = reinterpret_cast<uint32_t*>(hdr_l1);
uint32_t* out = hdr + 16;
const uint32_t nr = (kept + 31) & ~31u;
for (uint32_t k = 4 * kept; k < 4 * nr; ++k) {
out[k] = 0;
}
hdr[0] = total;
hdr[1] = overflow;
hdr[2] = kept;
#ifdef KPC_HR
uint32_t b = kept == 0 ? 0 : (kept + KPC_BSTEP - 1) / KPC_BSTEP - 1;
const uint32_t spec = reinterpret_cast<volatile uint32_t*>(prm_l1)[2];
if (spec > b) {
b = spec;
}
if (b > NB - 1) {
b = NB - 1;
}
hdr[3] = b;
#endif
noc_async_write(hdr_l1, hdracc.get_noc_addr(0), (16 + 4 * nr) * 4);
#ifdef KPC_HR
{
const auto bacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + b), ROWB);
const uint32_t bytes = (16 + 4 * nr) * 4;
for (uint32_t p = 0, off = 0; off < bytes; ++p, off += ROWB) {
const uint32_t sz = bytes - off < ROWB ? bytes - off : ROWB;
noc_async_write(hdr_l1 + off, bacc.get_noc_addr(p), sz);
}
if (b != spec && spec < NB) {
const auto sacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + spec), ROWB);
noc_async_write(hdr_l1, sacc.get_noc_addr(0), 64);
}
}
#endif
noc_async_write_barrier();
}
}
|