Download code/tt_diffusion_planner/tt/kernels/kcat_reader.cpp from changh95/diffusion-planner-p150: direct link, hf CLI and curl.
- Browser
- Download file 935 Bytes
-
https://huggingface.co/changh95/diffusion-planner-p150/resolve/main/code/tt_diffusion_planner/tt/kernels/kcat_reader.cpp
- Command line
-
hf download hf://changh95/diffusion-planner-p150/code/tt_diffusion_planner/tt/kernels/kcat_reader.cpp
-
curl -L -o kcat_reader.cpp https://huggingface.co/changh95/diffusion-planner-p150/resolve/main/code/tt_diffusion_planner/tt/kernels/kcat_reader.cpp
935 Bytes
| // SPDX-License-Identifier: Apache-2.0 | |
| // Split-matmul operand build (tt/kcat_kernel.py), reader (RISCV_0): this core's contiguous range of x tiles. | |
| // CT args: [0] per-core RT-arg count (P3), then the TensorAccessorArgs of x. | |
| // Common RT args: [x_addr]. Per-core RT args: [t0, n] (flattened x tile indices). | |
| constexpr auto x_args = TensorAccessorArgs<1>(); | |
| constexpr uint32_t cb_x = 0; | |
| constexpr uint32_t TB = 4096; | |
| void kernel_main() { | |
| const uint32_t x_addr = get_common_arg_val<uint32_t>(0); | |
| const uint32_t t0 = get_arg_val<uint32_t>(0); | |
| const uint32_t n = get_arg_val<uint32_t>(1); | |
| const auto x = TensorAccessor(x_args, x_addr, TB); | |
| for (uint32_t t = t0; t < t0 + n; ++t) { | |
| cb_reserve_back(cb_x, 1); | |
| noc_async_read(x.get_noc_addr(t), get_write_ptr(cb_x), TB); | |
| noc_async_read_barrier(); | |
| cb_push_back(cb_x, 1); | |
| } | |
| } | |