File size: 6,156 Bytes
c699c4c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// SuperPoint keypoint list on device, DMA-only version of kp_compact2.cpp (same HDR output format).
// The NMS unfold kernel (nms_unfold_kp.cpp, KP_REC) already wrote every candidate as a final 4-word
// header entry into its RISC's slot of the records tensor R (L1 of the NMS core, slot s = 2 core + risc:
// [count, 0, 0, 0] + CAP x 16 B). This kernel runs on both data-movement RISCs of core (0, 0): each
// computes the kept prefix of every slot from the candidate counts (the candidate tensor C, local L1),
// then gathers the entries of its half of the slots with one NoC read per slot straight into the shared
// L1 header image; PROC 1 raises semaphore 0, PROC 0 finishes the header (zero rows up to a multiple of
// 32, counts) and writes it to DRAM (and, KPC_HR, into the descriptor bucket, as kp_compact2.cpp).
// Runtime args: c_addr, hdr_addr, rec_addr, prm_addr (KPC_HR), NB bucket addresses (KPC_HR), then
// NCORE words (noc_x << 16 | noc_y) of the NMS cores (slot s -> core s / 2).
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"

void kernel_main() {
    const uint32_t c_addr = get_arg_val<uint32_t>(0);
    const uint32_t hdr_addr = get_arg_val<uint32_t>(1);
    const uint32_t rec_addr = get_arg_val<uint32_t>(2);

    constexpr uint32_t cb_scratch = get_compile_time_arg_val(0);
    constexpr uint32_t NSLOT = get_compile_time_arg_val(1);
    constexpr uint32_t CAP = get_compile_time_arg_val(2);
    constexpr uint32_t KMAX = get_compile_time_arg_val(3);
    constexpr uint32_t PROC = get_compile_time_arg_val(4);
    constexpr uint32_t SLOT_BYTES = (CAP + 1) * 4;
    constexpr uint32_t REC_BYTES = (CAP + 1) * 16;
    constexpr uint32_t HDR_WORDS = 16 + 4 * KMAX;
    constexpr auto hdr_args = TensorAccessorArgs<5>();
    const auto hdracc = TensorAccessor(hdr_args, hdr_addr, HDR_WORDS * 4);
#ifdef KPC_HR
    constexpr auto prm_args = TensorAccessorArgs<hdr_args.next_compile_time_args_offset()>();
    constexpr auto bk_args = TensorAccessorArgs<prm_args.next_compile_time_args_offset()>();
    constexpr uint32_t ROWB = KPC_C * 4;
    constexpr uint32_t NB = KMAX / KPC_BSTEP;
    constexpr uint32_t CORE_ARG0 = 4 + NB;
#else
    constexpr uint32_t CORE_ARG0 = 3;
#endif

    const uint32_t hdr_l1 = get_write_ptr(cb_scratch);
#ifdef KPC_HR
    const uint32_t prm_l1 = hdr_l1 + HDR_WORDS * 4;
    if constexpr (PROC == 0) {
        const auto prmacc = TensorAccessor(prm_args, get_arg_val<uint32_t>(3), 64);
        noc_async_read(prmacc.get_noc_addr(0), prm_l1, 64);
    }
#endif

    // PROC 0 owns slots [0, NSLOT/2), PROC 1 [NSLOT/2, NSLOT): counts into a local (RISC-private, fast)
    // array, PROC 0 publishes its kept sum (semaphore 1) = PROC 1's first entry index
    constexpr uint32_t HALF = NSLOT / 2;
    const uint32_t s_begin = PROC == 0 ? 0 : HALF, s_end = PROC == 0 ? HALF : NSLOT;
    volatile tt_l1_ptr uint32_t* xch = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(hdr_l1 + HDR_WORDS * 4 + 64);
    uint16_t cnts[NSLOT - HALF];
    uint32_t total = 0, overflow = 0, sum = 0;
    for (uint32_t s = s_begin; s < s_end; ++s) {
        uint32_t cnt = reinterpret_cast<volatile uint32_t*>(c_addr + s * SLOT_BYTES)[0];
        total += cnt;
        if (cnt > CAP) {
            overflow = 1;
            cnt = CAP;
        }
        cnts[s - s_begin] = (uint16_t)cnt;
        sum += cnt;
    }
    volatile tt_l1_ptr uint32_t* sem1 = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(1));
    uint32_t n = 0;
    if constexpr (PROC == 0) {
        xch[0] = sum;
        *sem1 = 1;
    } else {
        noc_semaphore_wait(sem1, 1);
        *sem1 = 0;
        n = xch[0] < KMAX ? xch[0] : KMAX;
    }
    const uint32_t out_l1 = hdr_l1 + 64;
    for (uint32_t s = s_begin; s < s_end && n < KMAX; ++s) {
        uint32_t m = cnts[s - s_begin];
        if (m == 0) {
            continue;
        }
        if (m > KMAX - n) {
            m = KMAX - n;
        }
        const uint32_t xy = get_arg_val<uint32_t>(CORE_ARG0 + (s >> 1));
        const uint64_t src = get_noc_addr(xy >> 16, xy & 0xFFFF, rec_addr + (s & 1) * REC_BYTES + 16);
        noc_async_read(src, out_l1 + 16 * n, 16 * m);
        n += m;
    }
    noc_async_read_barrier();
    volatile tt_l1_ptr uint32_t* sem = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_semaphore(0));
    if constexpr (PROC == 1) {
        xch[1] = total;
        xch[2] = overflow;
        xch[3] = sum;
        *sem = 1;
        return;
    } else {
        noc_semaphore_wait(sem, 1);
        *sem = 0;
        total += xch[1];
        overflow |= xch[2];
        const uint32_t ksum = sum + xch[3];
        const uint32_t kept = ksum < KMAX ? ksum : KMAX;
        uint32_t* hdr = reinterpret_cast<uint32_t*>(hdr_l1);
        uint32_t* out = hdr + 16;
        const uint32_t nr = (kept + 31) & ~31u;
        for (uint32_t k = 4 * kept; k < 4 * nr; ++k) {
            out[k] = 0;
        }
        hdr[0] = total;
        hdr[1] = overflow;
        hdr[2] = kept;
#ifdef KPC_HR
        uint32_t b = kept == 0 ? 0 : (kept + KPC_BSTEP - 1) / KPC_BSTEP - 1;
        const uint32_t spec = reinterpret_cast<volatile uint32_t*>(prm_l1)[2];
        if (spec > b) {
            b = spec;
        }
        if (b > NB - 1) {
            b = NB - 1;
        }
        hdr[3] = b;
#endif
        noc_async_write(hdr_l1, hdracc.get_noc_addr(0), (16 + 4 * nr) * 4);
#ifdef KPC_HR
        {
            const auto bacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + b), ROWB);
            const uint32_t bytes = (16 + 4 * nr) * 4;
            for (uint32_t p = 0, off = 0; off < bytes; ++p, off += ROWB) {
                const uint32_t sz = bytes - off < ROWB ? bytes - off : ROWB;
                noc_async_write(hdr_l1 + off, bacc.get_noc_addr(p), sz);
            }
            if (b != spec && spec < NB) {
                const auto sacc = TensorAccessor(bk_args, get_arg_val<uint32_t>(4 + spec), ROWB);
                noc_async_write(hdr_l1, sacc.get_noc_addr(0), 64);
            }
        }
#endif
        noc_async_write_barrier();
    }
}