Download code/tt_moge/kernels/pixshuf_reader.cpp from changh95/moge-2-p150: direct link, hf CLI and curl.
- Browser
- Download file 1.66 kB
-
https://huggingface.co/changh95/moge-2-p150/resolve/main/code/tt_moge/kernels/pixshuf_reader.cpp
- Command line
-
hf download hf://changh95/moge-2-p150/code/tt_moge/kernels/pixshuf_reader.cpp
-
curl -L -o pixshuf_reader.cpp https://huggingface.co/changh95/moge-2-p150/resolve/main/code/tt_moge/kernels/pixshuf_reader.cpp
1.66 kB
| // SPDX-License-Identifier: Apache-2.0 | |
| // Pixel-shuffle reader for a column block of a wider TILE matrix (moge-2 opt kernel, round 9): work unit u = | |
| // (tile row t = u / 2, dy = u % 2) reads the WT tiles [COL0 + dy * WT, +WT) of tile row t of a matrix with | |
| // ROW_TILES tiles per tile row into c_0 (the stream's [HW, (dy, dx, o)] block of the combined level-1 fold | |
| // matmul output). Runtime args: src_addr, num_units, start_unit. Compile args: WT, ROW_TILES, COL0, | |
| // TensorAccessorArgs. | |
| void kernel_main() { | |
| const uint32_t src_addr = get_arg_val<uint32_t>(0); | |
| const uint32_t nunits = get_arg_val<uint32_t>(1); | |
| const uint32_t u0 = get_arg_val<uint32_t>(2); | |
| constexpr uint32_t WT = get_compile_time_arg_val(0); | |
| constexpr uint32_t ROW_TILES = get_compile_time_arg_val(1); | |
| constexpr uint32_t COL0 = get_compile_time_arg_val(2); | |
| constexpr auto src_args = TensorAccessorArgs<3>(); | |
| constexpr uint32_t cb_id = 0; | |
| const uint32_t page_bytes = get_local_cb_interface(cb_id).fifo_page_size; | |
| const auto s = TensorAccessor(src_args, src_addr); | |
| DataflowBuffer dfb(cb_id); | |
| for (uint32_t u = u0; u < u0 + nunits; ++u) { | |
| const uint32_t base = (u >> 1) * ROW_TILES + COL0 + (u & 1) * WT; | |
| dfb.reserve_back(WT); | |
| const uint32_t l = dfb.get_write_ptr(); | |
| for (uint32_t j = 0; j < WT; ++j) { | |
| noc_async_read(s.get_noc_addr(base + j), l + j * page_bytes, page_bytes); | |
| } | |
| noc_async_read_barrier(); | |
| dfb.push_back(WT); | |
| } | |
| } | |