Download code/kernels/sp_nms/sample_pipe_compute.cpp from changh95/superpoint-p150: direct link, hf CLI and curl.
- Browser
- Download file 4.81 kB
-
https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/sample_pipe_compute.cpp
- Command line
-
hf download hf://changh95/superpoint-p150/code/kernels/sp_nms/sample_pipe_compute.cpp
-
curl -L -o sample_pipe_compute.cpp https://huggingface.co/changh95/superpoint-p150/resolve/main/code/kernels/sp_nms/sample_pipe_compute.cpp
4.81 kB
| // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc. | |
| // SPDX-License-Identifier: Apache-2.0 | |
| // | |
| // SuperPoint descriptor sampling, pipelined variant (SP_SF_PIPE=1), compute: per keypoint group (order: the core's first | |
| // unit 1, 2, 3, 0 with group 0's weights from CB_W0, further units 0..3) | |
| // OUT = ((G0*W0 + G1*W1) + G2*W2) + G3*W3 (fp32 in DEST, SFPU mul/add) | |
| // exactly the operations and order of sample_fused_compute.cpp; W_t is built in DST from the compact weight page by | |
| // SFPU register copies (sf_replicate). | |
| // DST slot WC holds compact page t / 2 (block (t % 2) * 8 + i = 4 DST rows = SFPU rows 2 b, 2 b + 1, every element | |
| // w(i, t)); write w(i, t) into the 4 SFPU rows (8 DST rows = 128 elements = keypoint i's channels) of keypoint i in slot WS. | |
| template <uint32_t WC, uint32_t WS> | |
| inline void sf_replicate(uint32_t t) { | |
| for (uint32_t i = 0; i < 8; ++i) { | |
| sfpi::vFloat w = sfpi::dst_reg[WC * 32 + ((t & 1) * 8 + i) * 2]; | |
| sfpi::dst_reg[WS * 32 + 4 * i + 0] = w; | |
| sfpi::dst_reg[WS * 32 + 4 * i + 1] = w; | |
| sfpi::dst_reg[WS * 32 + 4 * i + 2] = w; | |
| sfpi::dst_reg[WS * 32 + 4 * i + 3] = w; | |
| } | |
| } | |
| // SF_WC16: the compact block holds w(i, t) only in DST row 4 b (16 words); load that 4-row group (even columns) into | |
| // LREG0..3 and transpose (SFPTRANSP: LREG k row j <-> LREG j row k) so LREG0 holds row 0 of all four = w in every lane, | |
| // then store it to the 4 SFPU rows (DST rows 8 i .. 8 i + 7, both column parities) of keypoint i. Register moves only. | |
| template <uint32_t WC, uint32_t WS> | |
| inline void sf_replicate16(uint32_t t) { | |
| for (uint32_t i = 0; i < 8; ++i) { | |
| const uint32_t src = WC * 64 + 4 * ((t & 1) * 8 + i); | |
| TT_SFPLOAD(p_sfpu::LREG0, InstrModLoadStore::DEFAULT, ADDR_MOD_7, src); | |
| TT_SFPLOAD(p_sfpu::LREG1, InstrModLoadStore::DEFAULT, ADDR_MOD_7, src); | |
| TT_SFPLOAD(p_sfpu::LREG2, InstrModLoadStore::DEFAULT, ADDR_MOD_7, src); | |
| TT_SFPLOAD(p_sfpu::LREG3, InstrModLoadStore::DEFAULT, ADDR_MOD_7, src); | |
| TTI_SFPTRANSP(0, 0, 0, 0); | |
| const uint32_t dst = WS * 64 + 8 * i; | |
| TT_SFPSTORE(p_sfpu::LREG0, InstrModLoadStore::DEFAULT, ADDR_MOD_7, dst); | |
| TT_SFPSTORE(p_sfpu::LREG0, InstrModLoadStore::DEFAULT, ADDR_MOD_7, dst + 2); | |
| TT_SFPSTORE(p_sfpu::LREG0, InstrModLoadStore::DEFAULT, ADDR_MOD_7, dst + 4); | |
| TT_SFPSTORE(p_sfpu::LREG0, InstrModLoadStore::DEFAULT, ADDR_MOD_7, dst + 6); | |
| } | |
| } | |
| void kernel_main() { | |
| const uint32_t nunits = get_arg_val<uint32_t>(0); | |
| constexpr uint32_t cb_g = get_compile_time_arg_val(0); | |
| constexpr uint32_t cb_w = get_compile_time_arg_val(1); | |
| constexpr uint32_t cb_o = get_compile_time_arg_val(2); | |
| constexpr uint32_t cb_w0 = get_compile_time_arg_val(4); | |
| unary_op_init_common(cb_g, cb_o); | |
| for (uint32_t ui = 0; ui < nunits; ++ui) { | |
| for (uint32_t gi = 0; gi < SF_KT; ++gi) { | |
| const uint32_t g = ui == 0 ? ((gi + 1) % SF_KT) : gi; | |
| const uint32_t cbw = (ui == 0 && g == 0) ? cb_w0 : cb_w; | |
| cb_wait_front(cb_g, 4); | |
| cb_wait_front(cbw, 2); | |
| tile_regs_acquire(); | |
| for (uint32_t t = 0; t < 4; ++t) { | |
| const uint32_t dg = t == 0 ? 0 : 1; | |
| if ((t & 1) == 0) { | |
| copy_tile_to_dst_init_short_with_dt(cb_g, cbw); | |
| copy_tile(cbw, t >> 1, 3); | |
| } | |
| copy_tile_to_dst_init_short_with_dt(cbw, cb_g); | |
| copy_tile(cb_g, t, dg); | |
| mul_binary_tile_init(); | |
| MATH((_llk_math_eltwise_sfpu_start_(0))); | |
| if (dg == 0) { | |
| MATH((sf_replicate16<3, 1>(t))); | |
| } else { | |
| MATH((sf_replicate16<3, 2>(t))); | |
| } | |
| if (dg == 0) { | |
| MATH((sf_replicate<3, 1>(t))); | |
| } else { | |
| MATH((sf_replicate<3, 2>(t))); | |
| } | |
| MATH((_llk_math_eltwise_sfpu_done_())); | |
| mul_binary_tile(dg, dg + 1, dg); | |
| if (t > 0) { | |
| add_binary_tile_init(); | |
| add_binary_tile(0, 1, 0); | |
| } | |
| } | |
| tile_regs_commit(); | |
| tile_regs_wait(); | |
| cb_reserve_back(cb_o, 1); | |
| pack_tile(0, cb_o); | |
| cb_push_back(cb_o, 1); | |
| tile_regs_release(); | |
| cb_pop_front(cb_g, 4); | |
| cb_pop_front(cbw, 2); | |
| } | |
| } | |
| } | |