File size: 1,275 Bytes
c699c4c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 | // SPDX-FileCopyrightText: © 2026 Tenstorrent USA, Inc.
// SPDX-License-Identifier: Apache-2.0
//
// Descriptor head (1x1 conv + L2 norm + untilize, models/tt/desc_head.py), PROC 0: publish the local
// input shard (CB_IN) and the core's L1-resident copy of the 1x1 weights + bias (CB_W, both bound to
// sharded tensors), and build the reduce scaler tile (1.0 in row 0 of each face).
#include <stdint.h>
#include "api/dataflow/dataflow_api.h"
void kernel_main() {
constexpr uint32_t cb_in = get_compile_time_arg_val(0);
constexpr uint32_t cb_one = get_compile_time_arg_val(1);
constexpr uint32_t NT = get_compile_time_arg_val(2);
constexpr uint32_t cb_w = get_compile_time_arg_val(3);
constexpr uint32_t NW = get_compile_time_arg_val(4);
cb_reserve_back(cb_in, NT);
cb_push_back(cb_in, NT);
cb_reserve_back(cb_w, NW);
cb_push_back(cb_w, NW);
cb_reserve_back(cb_one, 1);
volatile tt_l1_ptr uint32_t* p = reinterpret_cast<volatile tt_l1_ptr uint32_t*>(get_write_ptr(cb_one));
for (uint32_t i = 0; i < 512; ++i) {
p[i] = 0;
}
for (uint32_t f = 0; f < 4; ++f) {
for (uint32_t j = 0; j < 8; ++j) {
p[f * 128 + j] = 0x3F803F80u;
}
}
(void)p[511];
cb_push_back(cb_one, 1);
}
|