File size: 1,275 Bytes
c699c4c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
// SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// Descriptor head (1x1 conv + L2 norm + untilize, models/tt/desc_head.py), PROC 0: publish the local
// input shard (CB_IN) and the core's L1-resident copy of the 1x1 weights + bias (CB_W, both bound to
// sharded tensors), and build the reduce scaler tile (1.0 in row 0 of each face).
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"

void kernel_main() {
    constexpr uint32_t cb_in = get_compile_time_arg_val(0);
    constexpr uint32_t cb_one = get_compile_time_arg_val(1);
    constexpr uint32_t NT = get_compile_time_arg_val(2);
    constexpr uint32_t cb_w = get_compile_time_arg_val(3);
    constexpr uint32_t NW = get_compile_time_arg_val(4);
    cb_reserve_back(cb_in, NT);
    cb_push_back(cb_in, NT);
    cb_reserve_back(cb_w, NW);
    cb_push_back(cb_w, NW);
    cb_reserve_back(cb_one, 1);
    volatile tt_l1_ptr uint32_t* p = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_write_ptr(cb_one));
    for (uint32_t i = 0; i < 512; ++i) {
        p[i] = 0;
    }
    for (uint32_t f = 0; f < 4; ++f) {
        for (uint32_t j = 0; j < 8; ++j) {
            p[f * 128 + j] = 0x3F803F80u;
        }
    }
    (void)p[511];
    cb_push_back(cb_one, 1);
}