Download src/kernels/cpu/native_expert.cpp from WineryLabs/Winery-Strata: direct link, hf CLI and curl.
- Browser
- Download file 8.07 kB
-
https://huggingface.co/WineryLabs/Winery-Strata/resolve/main/src/kernels/cpu/native_expert.cpp
- Command line
-
hf download hf://WineryLabs/Winery-Strata/src/kernels/cpu/native_expert.cpp
-
curl -L -o native_expert.cpp https://huggingface.co/WineryLabs/Winery-Strata/resolve/main/src/kernels/cpu/native_expert.cpp
8.07 kB
| // src/kernels/cpu/native_expert.cpp - plan v0.3 P6: native (GGUF-form) experts on the CPU through ggml-cpu. | |
| // See the header. Nothing here is Strata arithmetic: the activation quantizers and the row dot products are | |
| // ggml-cpu's, so an IQ expert computes what llama.cpp's CPU backend computes for it. | |
| namespace strata::kernels::cpu { | |
| namespace { | |
| const ggml_type_traits_cpu* traits(int type) { return ggml_get_type_traits_cpu((ggml_type) type); } | |
| void init_once() { | |
| static std::once_flag once; | |
| std::call_once(once, [] { ggml_cpu_init(); }); | |
| } | |
| } // namespace | |
| bool native_experts_available() noexcept { return true; } | |
| bool native_fmt(int gu_type, int d_type, int64_t n_embd, int64_t n_ff, NativeFmt& f, std::string& err) { | |
| init_once(); | |
| const ggml_type_traits_cpu* tg = traits(gu_type); | |
| const ggml_type_traits_cpu* td = traits(d_type); | |
| if (tg == nullptr || tg->vec_dot == nullptr || td == nullptr || td->vec_dot == nullptr) { | |
| err = "native experts: ggml-cpu has no dot product for type " + std::to_string(tg && tg->vec_dot ? d_type : gu_type); | |
| return false; | |
| } | |
| const ggml_type_traits_cpu* ag = traits(tg->vec_dot_type); | |
| const ggml_type_traits_cpu* ad = traits(td->vec_dot_type); | |
| if (ag == nullptr || ag->from_float == nullptr || ad == nullptr || ad->from_float == nullptr) { | |
| err = "native experts: ggml-cpu cannot quantize an activation for this layer"; | |
| return false; | |
| } | |
| if (n_embd % ggml_blck_size((ggml_type) gu_type) || n_ff % ggml_blck_size((ggml_type) d_type) || | |
| n_embd % ggml_blck_size(tg->vec_dot_type) || n_ff % ggml_blck_size(td->vec_dot_type)) { | |
| err = "native experts: expert geometry is not whole blocks"; | |
| return false; | |
| } | |
| f.gu_type = gu_type; | |
| f.d_type = d_type; | |
| f.gu_act = (int) tg->vec_dot_type; | |
| f.d_act = (int) td->vec_dot_type; | |
| f.n_embd = n_embd; | |
| f.n_ff = n_ff; | |
| f.gu_row = ggml_row_size((ggml_type) gu_type, n_embd); | |
| f.d_row = ggml_row_size((ggml_type) d_type, n_ff); | |
| f.up_off = f.gu_row * (size_t) n_ff; | |
| f.down_off = 2 * f.up_off; | |
| f.bytes = f.down_off + f.d_row * (size_t) n_embd; | |
| f.act_bytes = ggml_row_size(tg->vec_dot_type, n_embd); | |
| f.h_bytes = ggml_row_size(td->vec_dot_type, n_ff); | |
| if (f.act_bytes > kNativeActBytes || f.h_bytes > kNativeHBytes) { | |
| err = "native experts: activation larger than the pool's buffers"; | |
| return false; | |
| } | |
| return true; | |
| } | |
| void native_quant_act(const NativeFmt& f, const float* x, void* dst) { | |
| traits(f.gu_act)->from_float(x, dst, f.n_embd); | |
| } | |
| void native_quant_h(const NativeFmt& f, const float* h, void* dst) { | |
| traits(f.d_act)->from_float(h, dst, f.n_ff); | |
| } | |
| void native_gu_rows(const NativeFmt& f, const uint8_t* blob, const void* const* act, int nt, float* const* ff, | |
| int r0, int r1) { | |
| // the multi-token kernels decode the weights once for all tokens: 2.0-2.4x ggml-cpu at three tokens, no faster | |
| // at one (all are bound by the codebook lookups, ~5 GB/s per core), measured by native_expert_parity. AVX-512 | |
| // first, then the AVX-2 one (Zen 2/3, Intel 12th-14th gen). STRATA_NO_IQ512 drops an AVX-512 CPU to the | |
| // AVX-2 kernel, STRATA_NO_IQ256 drops the AVX-2 kernel; ggml-cpu's single-token vec_dot is reached only with | |
| // both set (and on a CPU without AVX-512, STRATA_NO_IQ512 changes nothing). | |
| static const bool avx512 = cpu_avx512_ok() && std::getenv("STRATA_NO_IQ512") == nullptr; | |
| static const bool avx2 = std::getenv("STRATA_NO_IQ256") == nullptr; | |
| // #152: from how many tokens the multi-token kernels run (ggml's vec_dot below that). The default 2 is the | |
| // measured-fastest rule, but a token's expert rows then round differently alone than in a group, so greedy output | |
| // can depend on how many drafts a verify window held. STRATA_IQ_MT_MIN=1 (opt-in, 0.1.30) uses the multi-token | |
| // kernels for every group: output independent of the drafting, at a measured -1..-3% decode on IQ3_S (AVX-512). | |
| static const int mt_min = [] { const char* e = std::getenv("STRATA_IQ_MT_MIN"); return e ? std::atoi(e) : 2; }(); | |
| // Unsloth UD-Q4_K_XL's Q4_K gate/up: the multi-token kernel is bit-exact against ggml's per-token dot (any group | |
| // size, no #152 rule). Opt-in, STRATA_KQ256=1: measured no faster in the engine (a window's expert groups hold | |
| // ~1.4 tokens and the weights stay in L1 across ggml's per-token calls; 1.01-1.13x in native_expert_parity). | |
| static const bool kq = [] { const char* v = std::getenv("STRATA_KQ256"); return v != nullptr && std::atoi(v) != 0; }(); | |
| if (kq && f.gu_type == 12 && nt >= 2) { // one token: ggml's own dot below (the same bits, less overhead) | |
| kq256_gu_rows(f.gu_type, blob, f.gu_row, f.up_off, (int) f.n_embd, act, nt, ff, r0, r1); | |
| return; | |
| } | |
| // A format with only an AVX-2 kernel (IQ4_XS, #415) takes it on AVX-2 CPUs only: an AVX-512 CPU keeps ggml-cpu for | |
| // it, as before (its rows would round differently). Each kernel only for the formats it implements: falling | |
| // through an empty switch would leave ff unwritten instead of falling back to ggml-cpu. | |
| static const bool cpu512 = cpu_avx512_ok(); | |
| if (nt >= mt_min && (iq512_supported(f.gu_type) || (!cpu512 && iq256_supported(f.gu_type)))) { | |
| if (avx512 && iq512_supported(f.gu_type)) { | |
| iq512_gu_rows(f.gu_type, blob, f.gu_row, f.up_off, (int) f.n_embd, act, nt, ff, r0, r1); | |
| return; | |
| } | |
| if (avx2 && iq256_supported(f.gu_type)) { | |
| iq256_gu_rows(f.gu_type, blob, f.gu_row, f.up_off, (int) f.n_embd, act, nt, ff, r0, r1); | |
| return; | |
| } | |
| } | |
| const ggml_vec_dot_t dot = traits(f.gu_type)->vec_dot; | |
| const int n = (int) f.n_embd; | |
| for (int r = r0; r < r1; ++r) { | |
| const uint8_t* gr = blob + (size_t) r * f.gu_row; | |
| const uint8_t* ur = blob + f.up_off + (size_t) r * f.gu_row; | |
| for (int t = 0; t < nt; ++t) { | |
| float g = 0.f, u = 0.f; | |
| dot(n, &g, 0, gr, 0, act[t], 0, 1); | |
| dot(n, &u, 0, ur, 0, act[t], 0, 1); | |
| ff[t][r] = (g / (1.f + std::exp(-g))) * u; | |
| } | |
| } | |
| } | |
| void native_down_rows(const NativeFmt& f, const uint8_t* blob, const void* const* hq, int nt, float* const* out, | |
| int r0, int r1) { | |
| // IQ4_NL down rows: the AVX-2 multi-token kernel decodes the nibbles and absolutises the weights once per | |
| // block instead of once per token; ggml-cpu's dot is single-token. STRATA_NO_IQ4NL falls back to it. | |
| static const bool iq4nl_mt = std::getenv("STRATA_NO_IQ4NL") == nullptr; | |
| static const int mt_min = [] { const char* e = std::getenv("STRATA_IQ_MT_MIN"); return e ? std::atoi(e) : 2; }(); | |
| static const bool kq = [] { const char* v = std::getenv("STRATA_KQ256"); return v != nullptr && std::atoi(v) != 0; }(); | |
| if (kq && nt >= 2 && (f.d_type == 7 || f.d_type == 8)) { // Q5_1 / Q8_0 down: bit-exact, any group size | |
| kq256_rows(f.d_type, blob + f.down_off, f.d_row, (int) f.n_ff, hq, nt, out, r0, r1); | |
| return; | |
| } | |
| if (nt >= mt_min && f.d_type == 20 && iq4nl_mt) { // #152: the same rule as the gate/up rows | |
| iq4nl256_down_rows(blob + f.down_off, f.d_row, (int) f.n_ff, hq, nt, out, r0, r1); | |
| return; | |
| } | |
| const ggml_vec_dot_t dot = traits(f.d_type)->vec_dot; | |
| const int n = (int) f.n_ff; | |
| for (int r = r0; r < r1; ++r) { | |
| const uint8_t* dr = blob + f.down_off + (size_t) r * f.d_row; | |
| for (int t = 0; t < nt; ++t) { | |
| float s = 0.f; | |
| dot(n, &s, 0, dr, 0, hq[t], 0, 1); | |
| out[t][r] = s; | |
| } | |
| } | |
| } | |
| } // namespace strata::kernels::cpu | |