superpoint-p150 / code /kernels /sp_nms /sample_fused_reader.cpp
changh95's picture
Optimized build (2026-10-03): trace 3.73 -> 0.50 ms
c699c4c verified
Raw History Blame Contribute Delete
5.43 kB
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// SuperPoint descriptor sampling, fused op, reader (RISCV_0). One generic_op replaces the former
// gather op + 4 fp32 multiplies + 3 fp32 adds + untilize op.
// Unit u = (tile row tr = u / NQ, channel chunk q = u % NQ): keypoints k = 32*tr .. 32*tr+31 (HDR,
// kp_compact.cpp layout), channels CPU*q .. CPU*q+CPU-1.
// Data are kept ROW-MAJOR inside "pseudo tiles": a 2 KB bf16 page holds KPT = 1024/CPU keypoints x
// CPU channels in plain row-major order. Elementwise SFPU math does not care about the element
// order (unpack/pack keep it), so no tilize is needed. Page (p, t) (p = 0..KT-1 keypoint group,
// t = 0..3 tap) of CB_G holds G_t[k, c] = D[cell_t(k), CPU*q + c] for k = 32*tr + p*KPT + i.
// Units whose tile row is at or beyond n = HDR[2] push their pages unfilled (the compute kernel
// processes a fixed unit count; the writer drops those rows). Pure copies -> exact.
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
const uint32_t d_addr = get_arg_val<uint32_t>(0);
const uint32_t hdr_addr = get_arg_val<uint32_t>(1);
const uint32_t nunits = get_arg_val<uint32_t>(2);
constexpr uint32_t cb_g = get_compile_time_arg_val(0);
constexpr uint32_t cb_scratch = get_compile_time_arg_val(1);
constexpr uint32_t C = get_compile_time_arg_val(2);
constexpr uint32_t KV = get_compile_time_arg_val(3);
constexpr uint32_t CPU = get_compile_time_arg_val(4); // channels per unit
constexpr uint32_t NQ = C / CPU;
constexpr uint32_t KPT = 1024 / CPU; // keypoints per pseudo tile
constexpr uint32_t KT = 32 / KPT; // pseudo tiles per tap
constexpr auto d_args = TensorAccessorArgs<5>();
constexpr auto hdr_args = TensorAccessorArgs<d_args.next_compile_time_args_offset()>();
const auto dacc = TensorAccessor(d_args, d_addr, C * 2);
const auto hacc = TensorAccessor(hdr_args, hdr_addr, (16 + 4 * KV) * 4);
#ifdef SF_SPLIT
// weight-page fill split (one unit per core): this RISC fills the tap weights of keypoints
// 0..SF_SPLIT-1 of the unit, the writer the rest; local semaphore 0 tells the writer it is done.
constexpr auto wt_args = TensorAccessorArgs<hdr_args.next_compile_time_args_offset()>();
const uint32_t wtab_addr = get_arg_val<uint32_t>(3 + nunits);
const auto wtacc = TensorAccessor(wt_args, wtab_addr, SF_W * 16);
volatile tt_l1_ptr uint32_t* fill_sem = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(0));
#endif
const uint32_t hdr0 = get_write_ptr(cb_scratch); // HDR[0..15] (64 B)
const uint32_t kps = hdr0 + 64; // 32 HDR keypoint entries (512 B)
noc_async_read(hacc.get_noc_addr(0), hdr0, 64);
noc_async_read_barrier();
const uint32_t n = reinterpret_cast<volatile uint32_t*>(hdr0)[2];
uint32_t cur_tr = 0xFFFFFFFF;
for (uint32_t ui = 0; ui < nunits; ++ui) {
const uint32_t u = get_arg_val<uint32_t>(3 + ui);
const uint32_t tr = u / NQ, q = u % NQ;
cb_reserve_back(cb_g, 4 * KT);
if (tr * 32 < n) {
if (tr != cur_tr) {
noc_async_read(hacc.get_noc_addr(0) + (16 + 128 * tr) * 4, kps, 512);
noc_async_read_barrier();
cur_tr = tr;
}
const uint32_t* kp = reinterpret_cast<const uint32_t*>(kps);
const uint32_t g0 = get_write_ptr(cb_g);
for (uint32_t t = 0; t < 4; ++t) {
const uint32_t xs = (t & 1) ? 0 : 16, ys = (t >> 1) ? 0 : 16;
for (uint32_t r = 0; r < 32; ++r) {
const uint32_t cell = ((kp[4 * r + 2] >> ys) & 0xFFFF) + ((kp[4 * r + 3] >> xs) & 0xFFFF);
const uint32_t page = (r / KPT) * 4 + t;
noc_async_read(dacc.get_noc_addr(cell) + q * CPU * 2, g0 + page * 2048 + (r % KPT) * CPU * 2, CPU * 2);
}
}
#ifdef SF_SPLIT
const uint32_t wblk = kps + 512;
if (ui == 0) {
for (uint32_t r = 0; r < SF_SPLIT; ++r) {
const uint32_t yx = kp[4 * r];
const uint32_t y = yx >> 16, x = yx & 0xFFFF;
noc_async_read(wtacc.get_noc_addr(y) + ((x * 16) & ~63u), wblk + r * 64, 64);
}
}
#endif
noc_async_read_barrier();
#ifdef SF_SPLIT
if (ui == 0) {
uint32_t* w0 = reinterpret_cast<uint32_t*>(get_write_ptr(SF_CBW)); // the writer's (only) CB_W slot
for (uint32_t r = 0; r < SF_SPLIT; ++r) {
const uint32_t x = kp[4 * r] & 0xFFFF;
const uint32_t* wv = reinterpret_cast<const uint32_t*>(wblk + r * 64 + ((x * 16) & 63));
const uint32_t p = r / KPT, i = r % KPT;
for (uint32_t t = 0; t < 4; ++t) {
const uint32_t v = wv[t];
uint32_t* d = w0 + (p * 4 + t) * 1024 + i * CPU;
for (uint32_t c = 0; c < CPU; c += 8) {
d[c] = v; d[c + 1] = v; d[c + 2] = v; d[c + 3] = v;
d[c + 4] = v; d[c + 5] = v; d[c + 6] = v; d[c + 7] = v;
}
}
}
*fill_sem = 1;
}
#endif
}
cb_push_back(cb_g, 4 * KT);
}
}