Winery-Strata / src /program /generate.cpp
Zandy-Wandy's picture
🍷 Winery Strata: fork of Niko1221/Strata (Winery Cuvée family, wine-red UI)
bbb6388 verified
Raw History Blame Contribute Delete
456 kB
// src/program/generate.cpp - P2.S6: `strata generate`.
//
// THE DRIVER, and the first program in this project that answers a question. Everything below it is a
// component; this is the thing that composes them into a token:
//
// embed_row(token) -> 48 captured layer graphs (the CPU expert pool behind the doorbell) ->
// lm_head(R) -> sample -> embed_row(next) -> ...
//
// WHAT IT IS NOT. There is no tokenizer here. `pack/full/tokenizer/` and `tools/strata_tokenizer.py` exist,
// and a C++ BPE is Phase 1's deliverable rather than this program's, so the prompt arrives as IDS via
// `--tokens`. That is not a placeholder: it is exactly what Gate C1 needs, because C1 compares logits against
// llama.cpp on the SAME ids, and a tokenizer on only one side of that comparison is a second variable.
//
// AND IT IS PHASE 2, so hit rate is `h = 0` and the number it prints is slow on purpose
// (`phase-2-correct-engine.md:5-9`). What it is FOR is the honest tok/s figure and the logit dump.
#include "strata/core/device.hpp"
#include "strata/core/expert_cache.hpp"
#include "strata/core/conversation_snapshot.hpp"
#include "strata/core/conversation_memory.hpp"
#include "strata/core/coupled_draft.hpp"
#include "strata/core/expert_source.hpp"
#include "strata/core/pinned.hpp"
#include "strata/core/remote_experts.hpp"
#include "strata/core/on_device.hpp"
#include "strata/core/peer_experts.hpp"
#include "strata/core/layer.hpp"
#include "strata/core/layout.hpp"
#include "strata/core/session.hpp"
#include "strata/core/weights.hpp"
#include "strata/kernels/cpu/expert.hpp"
#include "strata/kernels/cpu/pool.hpp"
#include "strata/kernels/cpu/expert_layout.hpp"
#include "strata/kernels/ngram.hpp"
#include "strata/kernels/iq_kernels.hpp"
#include "strata/kernels/s2_expert_grouped.hpp"
#include "strata/kernels/sampler.hpp"
#include "strata/kernels/verify_kernels.hpp"
#include "strata/kernels/shared_expert.hpp"
#include "strata/kernels/native_moe.hpp"
#include "strata/kernels/native_gdn.hpp"
#include "strata/kernels/native_router.hpp"
#include "strata/kernels/native_qsa.hpp"
#include "strata/kernels/native_qsa_indexer.hpp"
#include "strata/kernels/native_rope.hpp"
#include "strata/kernels/mrope.hpp"
#include "strata/kernels/kv_q4.hpp"
#include "strata/kernels/qsa.hpp"
#include "strata/core/native_head.hpp"
#include "strata/core/verify.hpp"
#include "strata/core/mtp.hpp"
#include "strata/prefill/prefill.hpp"
#include "strata/core/native_dense.hpp"
#include "strata/program/logits_selection.hpp"
#include "strata/program/conv_cache.hpp"
#include "strata/spec/draft_policy.hpp"
#include "strata/spec/suffix_drafter.hpp"
#include "strata/kernels/cvec.hpp"
#include "strata/core/progress.hpp"
#include "strata/core/device.hpp"
#include "strata/core/emulate.hpp"
#ifndef NOMINMAX
#define NOMINMAX // gguf_reader.hpp includes windows.h
#endif
#include "strata/artifact/gguf_reader.hpp"
#if defined(_WIN32)
#include <windows.h>
#include <psapi.h>
#include <io.h>
#else
#include <unistd.h>
#include <cerrno>
#endif
#include <cuda_runtime.h>
#include <array>
#include <chrono>
#include <algorithm>
#include <iostream>
#include <future>
#include <thread>
#include <atomic>
#include <condition_variable>
#include <deque>
#include <map>
#include <mutex>
#include <optional>
#include <new>
#include <charconv>
#include <cmath>
#include <limits>
#include <cstdio>
#include <cstdlib>
#include <cstring>
#include <filesystem>
#include <fstream>
#include <sstream>
#include <string>
#include <set>
#include <vector>
namespace {
// Windows' WDDM driver model: native Windows, or WSL2 (its GPU goes through /dev/dxg to the Windows driver). There,
// pinning a large arena into two CUDA contexts leaves WDDM refusing every later allocation (the 5080 + 3090 rig);
// a Linux driver has no such limit (#253: the 8 GiB cap cost a 4090 + 3060 split 3x of its prompt speed).
bool under_wddm() {
#ifdef _WIN32
return true;
#else
static const bool dxg = std::filesystem::exists("/dev/dxg");
return dxg;
#endif
}
// perf-review D-4: the lent slots are refilled with queued copies and one wait; STRATA_REFILL_BLOCKING=1 waits on each
bool refill_blocking() {
static const bool v = std::getenv("STRATA_REFILL_BLOCKING") != nullptr;
return v;
}
using Clock = std::chrono::steady_clock;
// The resident RAM mode and the adaptive tier. A swap copies `in` (held in RAM) into the slot of `out` (held only
// by that slot). Before the slot is overwritten, `out`'s bytes are copied back from it into an exchange buffer, so
// the CPU computes `out` from RAM while the swap is in flight; when the swap has landed, `commit_exchanges` moves
// them into `in`'s place in RAM. The RAM copy then again holds exactly the experts no core slot does, and no swap
// reads the file. Swaps that need no exchange (`out` in the lend region is held in RAM already; or `in` is not) go
// on as before; ones beyond the buffers' room wait for a later round. Runs on the adaptive tier's thread while the
// GPU commits and drafts: the copies back are on its stream, and waited for before the refills are queued.
template <class Swap>
bool resident_stage_swaps(strata::core::FileExpertSource& src, strata::core::ExpertCache& cache,
const std::vector<int32_t>& host_res, int64_t n_expert, std::vector<Swap>& swaps,
cudaStream_t stream) {
if (!src.complement_ready() || swaps.empty()) return true;
struct Staged { int32_t layer, in, out; int64_t q; };
std::vector<Staged> staged;
std::vector<Swap> kept;
kept.reserve(swaps.size());
for (const Swap& s : swaps) {
if (!src.has_resident(s.layer, s.in) || src.has_resident(s.layer, s.out)) { kept.push_back(s); continue; }
const int64_t q = (int64_t) staged.size();
if (q >= src.exchange_capacity()) continue;
const int32_t slot = host_res[(size_t) s.layer * (size_t) n_expert + (size_t) s.out];
if (slot < 0) continue;
if (cudaMemcpyAsync(src.exchange_buffer(q), cache.device_slot(slot),
(size_t) strata::kernels::cpu::expert_layout().blob_bytes(s.layer), cudaMemcpyDeviceToHost,
stream) != cudaSuccess)
return false;
staged.push_back({s.layer, s.in, s.out, q});
kept.push_back(s);
}
if (!staged.empty()) {
if (cudaStreamSynchronize(stream) != cudaSuccess) return false;
for (const Staged& x : staged)
if (!src.stage_exchange(x.layer, x.in, x.out, x.q)) return false;
}
swaps.swap(kept);
return true;
}
struct Options {
std::string pack = "pack/full";
std::vector<int64_t> tokens; // the prompt, PRE-TOKENIZED
int64_t max_new = 16;
int64_t max_context = 4096;
bool greedy = true;
uint64_t seed = 0;
int top_k = 20;
float top_p = 0.95f;
float temperature = 1.0f;
std::string dump_logits; // one line of logits per generated position
int64_t logits_stride = 1; // storage selection; all prompt tokens remain conditioned
/// **THE RESIDUAL, SO THE HEAD CAN BE CHECKED WITHOUT THE LAYERS.**
///
/// C1 fails (LEDGER L116) and the pipeline is `embed -> 48 layers -> head`. Dumping `R` splits it in half:
/// the head is one norm, two bf16 projections and one 794 MB GEMV, all of which can be recomputed in Python
/// from the manifest. If Python agrees with the engine on the same `R`, the head is right and the layers
/// are wrong; if it disagrees, the head is wrong. Nothing else in the engine can be split that cheaply.
std::string ple_gguf; // the ORIGINAL second GGUF shard: the PLE table is not in the pack
bool no_ple = false; // explicit diagnostic ablation; never a normal inference default
bool stream_token = false; // R2.6 experiment: ordered work on the session stream
bool check_logits = false; // optional full-vocabulary finite scan
bool gr_fp32_activations = false; // pinned CUDA single-token BF16 activation contract
bool gr_native_mmvf = false; // pinned projection reduction tree as well as FP32 inputs
bool native_bf16 = false; // SSM gates, router and indexer projections only
bool native_bf16_extra = false; // PLE value and shared expert scalar gate
bool native_ple_key = false; // unchanged Q2_0 key and CUDA Q8_1 activations
bool native_moe_combine = false; // pinned fused CUDA weighted reduction
bool native_gdn = false; // pinned CUDA recurrence and preprocessing
bool native_flash_attn_short = false; // diagnostic pinned attention, context <=256
bool native_qsa_indexer = false; // pinned F16 key cache and F32 pooling
bool native_qsa = false; // pinned F32 QSA norms and gate
/// THE CONTEXT EXTENSION (rope scaling, rope_scaling.hpp). These knobs resolve to ONE process config,
/// set once before `session_init` builds the rope table and the graphs capture the kernels. There is
/// deliberately no per-request form: K sits in the cache POST-RoPE, so one cache must never mix two
/// scalings. Precedence: an EXPLICIT flag over the model file's rope keys over the struct defaults -
/// `none` and `1` are explicit values (the opt-outs), the absent flag is not.
std::string rope_scaling; ///< --rope-scaling none|linear|yarn (llama.cpp's names); empty = the flag is absent
double rope_scale = 0; ///< --rope-scale F: the extension factor; 0 = the flag is absent (the model file's, else 1 = off)
double rope_freq_base = 0; ///< --rope-freq-base N: 0 = the model's (1e7)
double rope_freq_scale = 0; ///< --rope-freq-scale F: the raw ggml knob; 0 = 1/--rope-scale
double yarn_orig_ctx = 0; ///< --yarn-orig-ctx N: 0 = the model's, else 262144
double yarn_ext_factor = -1.0; ///< --yarn-ext-factor F: <0 = auto (1 for yarn, 0 otherwise)
double yarn_attn_factor = 1.0; ///< --yarn-attn-factor F
double yarn_beta_fast = 32.0; ///< --yarn-beta-fast F
double yarn_beta_slow = 1.0; ///< --yarn-beta-slow F
bool native_rope = false; // pinned text-only CUDA rotary arithmetic
bool native_ple_postops = false; // pinned PLE postprojection arithmetic
bool native_router = false; // pinned fused 512-expert top-10 router
bool cpu_oracle_q8_0 = false; // pinned x86 activation scales/codes at both expert stages
std::string native_head_gguf; // native output.weight experiment; same model shard as the pack
std::vector<std::string> native_dense_gguf; // repeat for native GDN/QSA projection shards
/// Every shard of --native's model (strata::gguf_split_paths: the metadata shard first; a missing shard is an
/// error) and of --native-head-gguf's (the same list unless that names another model).
std::vector<std::string> native_shards, native_head_shards;
/// Plan v0.3 P1: the whole native arithmetic set as ONE switch (model shard 1). It enables exactly the
/// combination recorded in bench/results/2026-09-23-attention-ple plus the native indexer, and never the
/// <=256-token attention adapter. It becomes the default once P0 shows it is not slower.
std::string native_preset;
/// The token embedding from this GGUF instead of --native's (tools/embd_bf16_pack.py: BF16 as shipped)
std::string embd_gguf;
/// Plan v0.3 P2: how the n-gram table is read. Direct (default) = unbuffered SSD reads, table never in RAM.
std::string ple_io = "direct";
int64_t ple_row_cache = 1 << 20; ///< bounded row cache (rows of 90 B); 0 disables
int ple_inflight = 256; // the prompt path reads a chunk's rows at once: 64 left the SSD half idle (32K: 303 -> 189 ms)
double ple_delay_us = 0; ///< fault injection: every row read completes no earlier than this
bool ple_sync_submit = false; ///< A/B arm: submit reads on the token thread, no I/O worker
std::string kv = "fp16"; ///< plan v0.3 P7: KV storage, fp16 (default) or int8 (half the VRAM)
int64_t kv_resident = 0; ///< KV streaming: resident cells per QSA layer (0: all in VRAM)
std::string dump_residual;
/// The head input, `bb.mixed`. It exists so the head can be SPLIT: steps 1-4 (the per-stream norm, the two
/// bf16 projections and the stream mean) recompute cheaply in Python, and only the 794 MB GEMV does not.
std::string dump_mixed;
/// One residual snapshot per layer per position: `n_layers * hc * n_embd` floats per position, appended in
/// position order. This is the C1 BISECTION LADDER - it is what `llama-debug --tensor-filter l_last` prints
/// for the reference, so the first layer whose `sum` diverges is the layer that holds the bug. It needs the
/// captured path (the expert pool only exists there), so it is refused with `--no-capture`.
std::string dump_layers;
/// `2 * n_embd + 2 * hc` floats per layer per position: the attention half's block output, the MoE half's,
/// and the two injection vectors. It separates `linear_attn_out-<l>` from `ffn_out-<l>`, which the residual
/// ladder cannot. **CAPTURED INTO THE LAYER GRAPHS**, so it must be armed before `session_capture`.
std::string dump_halves;
/// P0.S8's routing trace, and a prerequisite the Phase 3 plan names explicitly. One record per layer per
/// position: `int32 layer, int32 k, k int32 ids, k float weights`. It is what a hit-rate curve for a
/// candidate VRAM expert cache is computed from, and it needs no new kernels - the doorbell already
/// publishes exactly this much to pinned memory.
std::string dump_routing;
bool no_capture = false; // run the layers directly instead of replaying graphs
bool no_pool = false; // skip the CPU expert pool: the GPU-only floor
bool sync_every_layer = false;
/// Per-stage CUDA-event timings inside the layer halves. `--no-capture` only: an event recorded inside a
/// stream capture is silently dropped, so the captured path cannot carry this.
bool stage_timing = false;
/// Launch the 48 captured `pre` graphs back to back with no host work between them and report the pure GPU
/// time per token. This is the only measurement that separates host-bound from GPU-bound, because the
/// stage events include every gap where the GPU waited for the host.
bool graph_only = false;
bool gpu_only_full = false; ///< R0.3: pre + post + head, the true per-token GPU floor
int pool_workers = 0; ///< R2.2: 0 = "all physical cores minus the host's"; >0 overrides
/// #272: the pool's core layout; `all` (the default) is the layout it always had, auto / p-cores are opt-in
strata::kernels::cpu::PoolAffinity pool_affinity = strata::kernels::cpu::PoolAffinity::All;
/// R2.2's first half, as an A/B arm. **ON by default**, because the measurement that justifies it is the
/// pool's own drain: 33.7 GB/s against 5/6 x 44.14 = 36.8 for five workers, on a machine whose sixth core
/// is reserved for a host thread that has nothing to do while the drain runs.
bool no_host_worker = false;
bool coupled_draft = strata::core::coupled_draft_env(); ///< Coupled draft sampling for MTP drafter under sampling
bool mmap_experts = false; ///< R2.1: opt OUT of the resident arena, back to MapViewOfFile
std::string shared_expert_arena; ///< Linux: optional file backing for the resident arena shared by processes
bool resident_cpu_experts = false; ///< mmap-backed static-cache misses copied into ordinary RAM
/// `--resident-experts` (the low-RAM PC's resident mode, chosen by setup): `--resident-cpu-experts` with the copy
/// page-locked when the driver allows (else locked in the working set), 4 GiB of RAM headroom, and plain mmap
/// (with a warning) when even the experts no slot holds do not fit.
bool resident_pin = false;
uint64_t resident_headroom = 8ull << 30;
bool resident_soft = false;
bool resident_cpu_explicit = false; ///< #384: --resident-cpu-experts given by itself (not only implied)
/// CS-T `--resident-budget-gib N`: the resident mode with a RAM budget - the N GiB of experts the GPU cache does
/// not hold that the expert profile ranks hottest are copied into RAM, the rest are read from the files in place
/// (the GGUF shards when the pack has no experts.bin). 0 = the whole complement (--resident-experts).
uint64_t resident_budget = 0;
/// R4: slots of VRAM-resident experts. **0 = off, and off is the default.**
/// **THE COMMENT THAT USED TO BE HERE WAS FALSE AND ROUND 328 MEASURED IT.** It said "the cache has no
/// consumer yet - `moe_hit_grouped_s2` does not exist - so switching it on costs the fill traffic and
/// saves nothing". The kernel exists (`src/kernels/cuda/s2_expert_grouped.cu`), it is wired at line ~660
/// via `expert_hit_run`, and switching the cache on **does** move work off the CPU pool: the drain fell
/// **19.076 -> 10.312 ms/token** at 4096 per-layer slots, for **-2.7 ms/token** end to end. What was
/// true is that the ADMISSION POLICY gave every slot to the first position, which is why the earlier
/// measurement found nothing - see `expert_cache_per_layer`.
int expert_cache = 0;
std::array<int, 3> expert_cache_remote{}; ///< CUDA1..3 slots; CUDA0 keeps dense/state/MTP
std::string expert_cache_remote_placement = "stripe"; ///< stripe experts or assign complete layers to CUDA1..3
bool expert_cache_cpu_order = false;
/// **R4.2g. ROUND 328 MEASURED THAT THE GLOBAL ADMISSION POLICY CANNOT WORK, AND THIS IS THE FIX.**
/// The default policy hands out slots in arrival order from one counter shared by all 48 layers, so the
/// first `n_slots` distinct pairs - about 26 LAYERS OF POSITION 0 - take every slot and hits are confined
/// to them. Measured at 256 slots: **1781 of 60000 = 2.97%**, against **21.4%** for 8 slots per layer and
/// **70.4%** for 64, from `Memory/cache_allocation.py` on the same run's routing. Off by default.
bool expert_cache_per_layer = false;
/// Multi-GPU: a second expert tier on CUDA device `peer_device` (-1 = off), `peer_reserve_mib` left free
/// there, `peer_slots` caps its size (0 = as many as fit), `peer_adapt_swaps` per adaptive round
/// (-1 = adapt_swaps).
int peer_device = -1;
int peer_reserve_mib = 600;
int64_t peer_slots = 0;
int peer_adapt_swaps = -1;
/// the prompt path's rows per layer the peer computes: -1 = half of chunk x top-k, 0 = the prompt path stays on
/// the primary
int64_t peer_prefill_rows = -1;
/// The PLE gather's prefetch, as an A/B arm. The gather measured 2.10-2.61 ms/token because its sixteen
/// row reads are sixteen SEPARATE page faults into a 26.8 GB mapping; see `ple_prefetch_enable`.
bool no_ple_prefetch = false;
/// R4.2e: a `profile.bin` from `tools/make_profile.py`. **When given, it decides residency instead of the
/// compulsory-miss policy**, which is the whole point: a profile ranked by routing frequency over a whole
/// trace is what the plan's `h = 0.6447` refers to, and compulsory-miss measured 0.4864 because it fills
/// with whatever the prompt touched FIRST. Empty means no profile.
std::string expert_profile;
/// #477 (--serve, opt-in): where to save what the adaptive tier learned, as a profile `--expert-profile` reads
/// (the resident experts first, then the routing counted since the start); on QUIT and every
/// `expert_profile_save_min` minutes between requests. Empty (the default): nothing is counted or written.
std::string expert_profile_save;
double expert_profile_save_min = 10.0;
/// R4.2d: **ON by default**, because the measurement is unambiguous and the alternative is known-broken.
/// Without it, 17 of 10,562 layers had the hit work done when the pool returned; with it, 9,190. The
/// A/B arm is `--no-hit-poke`.
bool no_hit_poke = false;
/// R0.9: capture each layer as THREE graphs and time them from outside the capture, which is the only
/// valid way to get a per-stage table on the real graph. Prints and exits; it is a measurement, not a run.
bool gpu_stages = false;
bool stats = false;
bool shared_late = false; ///< plan v0.3 P3 A/B: shared expert inside post[l] (old order)
bool keep_canonical = false; ///< plan v0.3 P1 A/B: load canonical copies of natively served tensors
bool no_token_graph = false; ///< plan v0.3 P3 A/B: two graphs per layer instead of one per token
bool no_fused_gr = false; ///< plan v0.3 P3 A/B: the six-kernel native gr_read + separate gr_write
bool no_fast_attn = false; ///< plan v0.3 P3 A/B: gather + one-block-per-head QSA attention
bool no_publish_kernel = false; ///< plan v0.3 P3 A/B: memcpy nodes for the doorbell and QSA step
bool no_fused_gdn = false; ///< plan v0.3 P3 A/B: llama.cpp-layout GDN step + separate out norm
bool no_fast_select = false; ///< plan v0.3 P7 A/B: FP64 row scores + bit-serial cell top-k
/// Plan v0.3 P4: `--expert-cache auto` sizes the VRAM tier from what is free after the weights, the session
/// and the KV state, minus this reserve for the graphs, the hit scratch and the head.
int vram_reserve_mib = 700;
bool vram_reserve_given = false; ///< --vram-reserve-mib on the command line (#496: no smaller automatic reserve)
/// Plan v0.3 P5: batched prompt processing in chunks of this many tokens (0 = the token path).
int64_t prefill_chunk = 0;
/// `--prefill auto`: the largest chunk (up to 8192) whose buffers the expert cache can lend. Every expert a chunk
/// routes to is streamed once per chunk, so a bigger chunk streams fewer bytes per token (the "ubatch" effect).
bool prefill_auto = false;
/// #282, opt-in: the largest chunk `--prefill auto` may take - 8192 by default; `--prefill auto:16384` or
/// `auto:32768` (or STRATA_PREFILL_AUTO_MAX) lets it go further, never past the context
int64_t prefill_auto_max = 8192;
bool no_split_rows = false; ///< plan v0.3 P4 A/B: one whole expert per pool thread
/// Plan v0.3 P5: the prompt path borrows the top expert-cache slots for its buffers and refills them after
/// the prompt (default); `--no-prefill-borrow` reserves the buffers' VRAM for the whole session instead.
bool no_prefill_borrow = false;
/// Plan v0.3 P5 validation: batch only positions [0, P) and run the rest of the prompt through the token path
/// (teacher-forced), so the logits of positions >= P - which depend on the batched state - can be scored
/// against the oracle at many positions. 0 = the whole prompt but the last position.
int64_t prefill_until = 0;
/// Plan v0.3 P6: after every processed position, append the residual after the last layer (hc x n_embd
/// floats, the MTP draft head's input) to this file. Token path only.
std::string dump_final_r;
/// Plan v0.3 P6: speculative decoding with a verify window of this many tokens (the last accepted token and
/// spec-1 drafts); 0 = plain decode. `spec_oracle` drafts from a token file (the expected continuation, for
/// the exactness test); `spec_corrupt` N > 0 replaces every Nth draft with a wrong token.
int spec = 0;
std::string spec_oracle;
int spec_corrupt = 0;
/// Plan v0.3 P6: the MTP draft layer's runtime directory (tools/mtp_rt.py); drafts come from it.
std::string mtp;
int64_t mtp_window = 32768; ///< the draft layer attends to the last N cells (0 = every cell)
/// Plan v0.3 P6: the share (0..1) of each layer's distinct missed experts the GPU reads over PCIe from the
/// pinned arena while the CPU computes the rest (verify windows).
double pcie_frac = -1.0; ///< < 0: the model's default (0.2 direct for the Q2_0 pack, 0.55 DMA for native packs)
std::string pcie_mode = "auto"; ///< auto | dma | kernel | direct
/// Plan v0.3 P6: every `adapt_every` rounds, swap up to `adapt_swaps` of the most-routed missing experts into
/// the VRAM tier in place of the least-routed resident ones (decayed counts). 0 = static residency.
int adapt_every = 4;
float adapt_decay = 0.7f; ///< the usage counts are multiplied by this after each adaptation (--adapt-decay)
/// Plan v0.3 P6: a draft enters the verify window only while every draft before it (and itself) has at least
/// this probability under the draft layer; 0 = always --spec-1 drafts.
double spec_min_p = 0.0;
/// Stop when the model emits an end-of-turn token (<|endoftext|> 248044, <|im_end|> 248046, or --eos-ids).
bool stop_eos = false;
std::vector<int64_t> eos_ids = {248044, 248046};
bool spec_split = false; ///< opt-in split verify window (the overlap study: exact, ~7% slower)
/// --serve, multi-GPU layer split: "K" or "K1,K2,.." (the first layer of each later stage) or "auto" (placed
/// from each GPU's free VRAM); empty = one GPU
std::string layer_split;
/// the later stages' devices "D1,D2,.." (default: the next visible GPUs; "0" with one K: both stages on this
/// GPU, sharing everything - the bit-exact A/B of the hand-off)
std::string split_device;
/// --split-skip-if-fits (opt-in; the server passes it for a config's "split_skip_if_fits"): with
/// --layer-split auto, run on CUDA0 alone when it holds every profiled expert pair plus the whole session (KV),
/// the drafter and the reserve - a split then only adds hand-offs (two RDNA4 cards: 1,384 vs 1,794 tok/s at 4K)
bool split_skip_if_fits = false;
/// Plan v0.3 P8: stay resident and take requests on stdin (see the --serve block in main).
bool serve = false;
/// The vision path: keep a per-cell (t, h, w) rotary position table so --serve can take GENI requests.
bool vision = false;
int adapt_swaps = 96;
/// --serve: how many conversation checkpoints to keep between requests (0 = every request reads its whole
/// prompt again, the v0.1.2 behaviour). One is the GDN recurrence of the 36 layers, the QSA indexer tails and
/// the PLE history (~118 MB of host RAM); the KV cache itself is positional and stays where it is.
int prompt_cache = 6;
int64_t conversation_cache_mib = 0; // opt-in host RAM for independent conversations
int conversation_cache_slots = 4;
int64_t conversation_cache_min_free_mib = 2560;
/// --serve: also keep a checkpoint every N freshly read prompt tokens (0 = only at the last turn boundary)
int64_t prompt_cache_every = 16384;
/// --serve: a prompt read from token 0 is also checkpointed at its first turn boundary - the end of the system
/// prompt, which every chat of the same client shares - when that is at least N tokens (0 = never)
int64_t prompt_cache_root = 2048;
/// --serve: the token that opens a chat turn (<|im_start|>). The last one in a prompt is where the chat's
/// history ends and the new assistant turn begins, which is the checkpoint the next request can reuse.
int64_t turn_token = 248045;
/// --serve: a text part of the prompt of at most N tokens (a chat message, the assistant header, a short tool
/// result) goes through the verify windows, S tokens at a time, instead of the batched prompt path (0 = always
/// the batched path)
int64_t short_read = 64;
/// The suffix drafter (prompt lookup): when the text being written repeats an earlier stretch of the context (code
/// edits, quoted input, tool-call JSON) by at least this many tokens, the window may be filled with what followed
/// it there instead of the MTP's drafts, where the MTP's own first guess agrees and the draft policy expects it to
/// pay (strata/spec/draft_policy.hpp). On by default; 0 = MTP only.
int suffix_draft = 3;
/// The MTP's own window cap (0 = --spec): with --spec 6 --mtp-max-t 4 the long windows come from suffix matches.
int mtp_max_t = 0;
/// A control vector on the residual stream (strata/kernels/cvec.hpp), with llama.cpp's flags: the
/// `experimental-speed-projection` profile passes `--control-vector-scaled FILE:1.0 --control-vector-layer-range
/// 4 44 --cvec-mode project --cvec-dir per-layer`. None by default; --serve switches a loaded one per request.
std::vector<std::pair<std::string, float>> cvec_files;
int cvec_first = -1, cvec_last = -1; ///< llama.cpp's defaults: 1 .. the last layer
int cvec_mode = 1; ///< 0 = project, 1 = add (llama.cpp's default)
int cvec_single = -1; ///< --cvec-dir single:L (project mode): layer L's direction everywhere
};
void usage() {
std::fprintf(stderr,
"strata generate --pack DIR --tokens \"1,2,3\" [options]\n"
"\n"
" --pack DIR the pack directory (default pack/full)\n"
" --tokens LIST the prompt as comma-separated token IDS (required)\n"
" --tokens-file PATH pretokenized prompt, commas or whitespace (alternative to --tokens)\n"
" --ple-gguf PATH required PLE table (original second GGUF shard); with --native, the model's\n"
" shard that holds per_layer_token_embd.weight when not given\n"
" --no-ple explicit diagnostic ablation of the PLE layer\n"
" --ple-io direct|mmap|ram n-gram table reads (plan v0.3 P2). direct (default): unbuffered SSD\n"
" reads, the table never enters RAM or the file cache; mmap: A/B arm;\n"
" ram: mmap with the whole table locked in RAM at start (Linux/macOS)\n"
" --ple-row-cache N bounded cache of fetched rows, 90 B each (default 1048576; 0 = off)\n"
" --ple-inflight N outstanding SSD reads (default 256)\n"
" --ple-delay-us U fault injection: each row read completes no earlier than U us\n"
" --ple-sync-submit A/B arm: submit table reads on the token thread (default: an I/O thread)\n"
" --kv fp16|int8 KV storage (plan v0.3 P7): int8 codes + fp16 scale per 64 values, half the\n"
" VRAM; default fp16 until gate G-C accepts int8\n"
" --kv q4_0 4-bit K/V after a Hadamard rotation (PR #21): half of int8's memory,\n"
" slightly lower precision (see bench/results/2026-09-27-kv-q4)\n"
" --kv k8v4 hybrid: INT8 K (exact attention scores) + rotated Q4_0 V, 816 B/cell\n"
" (vs int8's 1,056); not with --kv-resident\n"
" --kv-resident N KV streaming: keep N cells of each QSA layer in VRAM (min 20480) and the\n"
" whole K/V in pinned RAM; the freed VRAM goes to expert slots. 0 (default):\n"
" all of it in VRAM. A context of N cells or fewer is not streamed\n"
" --stream-token enqueue token work on the session stream (experimental)\n"
" --check-logits copy and check all logits in the stream-token path\n"
" --gr-fp32-activations experimental CUDA-oracle GR activation precision\n"
" --gr-native-mmvf experimental pinned GR norm/projections; implies FP32 activations\n"
" --native-bf16 experimental CUDA-oracle SSM/router/indexer BF16 projections\n"
" --native-bf16-extra experimental CUDA-oracle PLE/shared gate BF16 projections\n"
" --native-ple-key experimental native PLE key; requires --native-dense-gguf\n"
" --native-moe-combine experimental pinned CUDA routed/shared combination\n"
" --native-gdn experimental pinned CUDA GDN norms/gates/recurrence\n"
" --native-flash-attn-short diagnostic pinned vector attention; --max-context <=256\n"
" --native-qsa-indexer experimental pinned indexer key cache and pooling\n"
" --native-qsa experimental pinned QSA normalization and output gate\n"
" --native-rope experimental pinned text-only CUDA rotary arithmetic\n"
" --native-ple-postops experimental pinned PLE postprojection arithmetic\n"
" --native-router experimental pinned CUDA 512-expert top-10 routing\n"
" --cpu-oracle-q8-0 experimental pinned CPU expert quantization and dot reduction\n"
" --native SHARD1 every full-context native path at once (plan v0.3 P1): stream-token,\n"
" GR MMVF, BF16, head, dense + PLE key, MoE combine, GDN, router, QSA,\n"
" indexer, RoPE, PLE postops, and the CPU q8_0 contract unless the\n"
" expert cache is on. Individual --native-* flags stay for A/B.\n"
" --native-head-gguf PATH native Q5_K head from model shard 1; requires --stream-token\n"
" --embd-gguf PATH the token embedding from this GGUF instead of --native's (tools/embd_bf16_pack.py:\n"
" BF16 as the checkpoint ships it; mapped host memory, no VRAM)\n"
" --native-dense-gguf PATH native GDN/QSA/shared projections; repeat for each source model shard\n"
" --expert-cache-cpu-order experimental GPU expert reduction matching CPU order\n"
" --max-new N tokens to generate (default 16)\n"
" --max-context N KV/state capacity (default 4096)\n"
" --rope-scaling T extend the context past the trained one: none, linear\n"
" (position interpolation) or yarn - llama.cpp's types and names.\n"
" Default: the model file's rope keys, else none. Fixed at startup:\n"
" K in the cache is post-RoPE, so one run one scaling\n"
" --rope-scale F the extension factor for linear/yarn (default: the model file's\n"
" factor, else 1 = off)\n"
" --rope-freq-base N the raw ggml knobs: the frequency base (0 = the model's 1e7) and\n"
" --rope-freq-scale F the angle shrink (0 = 1/--rope-scale)\n"
" --yarn-orig-ctx N the trained context the correction targets (0 = 262144)\n"
" --yarn-ext-factor F --yarn-attn-factor F --yarn-beta-fast F --yarn-beta-slow F\n"
" YaRN's knobs; defaults: -1 (auto: 1 for yarn), 1, 32, 1\n"
" --greedy argmax (the default)\n"
" --seed S enable sampling with this Philox seed\n"
" --top-k N --top-p F --temperature F\n"
" --dump-logits PATH write one line of raw logits per position\n"
" --logits-stride N store every Nth row plus final input (default 1); N>1 requires --max-new 1\n"
" --dump-residual PATH write the final R (hc x n_embd, f32) for head bisection\n"
" --dump-layers PATH write R after EVERY layer, per position: the C1 bisection ladder\n"
" --dump-halves PATH write both halves' block_out and inject per layer: the half bisection\n"
" --dump-routing PATH write the routed expert ids and weights per layer per position (P0.S8)\n"
" --no-capture run the layers directly instead of replaying graphs\n"
" --shared-late A/B: shared expert after the CPU pool (default: overlapped with it)\n"
" --keep-canonical A/B: also load canonical copies of natively served tensors (more VRAM)\n"
" --vision --serve takes images too (GENI requests; embeddings from strata-vision)\n"
" --prompt-cache N --serve: keep N conversation checkpoints between requests (default 6, ~118 MB\n"
" of RAM each; 0 = read every prompt from the start)\n"
" --conversation-cache-mib N --serve: RAM budget for parked conversations (default 0 = off)\n"
" --conversation-cache-slots N --serve: at most N parked conversations (default 4)\n"
" --conversation-cache-min-free-mib N --serve: physical RAM floor when parking (default 2560)\n"
" --prompt-cache-every N --serve: also checkpoint every N fresh prompt tokens (default 16384, 0 = off)\n"
" --turn-token ID --serve: the token that opens a chat turn (default 248045, <|im_start|>)\n"
" --short-read N --serve: read at most N fresh text tokens through the decode windows instead\n"
" of the batched prompt path (default 64, 0 = off)\n"
" --suffix-draft N prompt lookup: draft from an earlier repeat of the last N+ tokens of context\n"
" when it pays (default 3; 0 = MTP only)\n"
" --mtp-max-t M cap the MTP's windows at M tokens (0 = --spec; longer ones come from suffixes)\n"
" --control-vector-scaled FILE:SCALE[,...] a control vector GGUF on the residual stream (llama.cpp's\n"
" format; --control-vector FILE = scale 1). --serve: requests switch it (cvec=0|1)\n"
" --control-vector-layer-range A B the layers it follows (inclusive; default 1 .. the last)\n"
" --cvec-mode add|project h += s v (default) or h -= s (h.v) v with v unit\n"
" --cvec-dir per-layer|single:L each layer's own direction (default) or layer L's everywhere (project)\n"
" --no-token-graph A/B: two graphs per layer (the host launches each) instead of one per token\n"
" --no-fused-gr A/B: the six-kernel hyper-connection read and a separate write (native)\n"
" --prefill CHUNK batched prompt processing in chunks of CHUNK tokens (needs --native); auto =\n"
" the largest chunk up to 8192 whose buffers the expert cache can lend;\n"
" auto:16384 / auto:32768 (or STRATA_PREFILL_AUTO_MAX) allow bigger ones\n"
" --no-pool skip the CPU expert pool (the GPU-only floor)\n"
" --sync-every-layer debug: synchronise after every layer\n"
" --ple-gguf PATH the n-gram/PLE shard. WITHOUT IT LAYER 1's PLE IS SILENTLY SKIPPED,\n"
" which changes every number downstream - pass it for any real run\n"
" --dump-mixed PATH write the post-attention residual (n_embd, f32)\n"
" --stage-timing per-stage KERNEL-COUNT shares. NOT a time profile: an uncaptured\n"
" event interval includes host gaps, so run with --gpu-only-full first\n"
" --graph-only MEASURE: replay the 48 `pre` graphs only. OMITS the 48 `post` graphs\n"
" and the LM head, so it is NOT the GPU floor (R0.3, Memory/ERRORS.md A4)\n"
" --gpu-only-full MEASURE: replay pre+post for all 48 layers plus the LM head, no pool.\n"
" THE TRUE PER-TOKEN GPU FLOOR. Quote this one, not --graph-only.\n"
" --stats print the per-stage breakdown\n"
" --gpu-stages R0.9: capture the layer as three graphs (mixer / ffn+router / post)\n"
" and time them from OUTSIDE the capture. The per-stage table on the\n"
" real graph that --stage-timing cannot give. Prints and exits.\n"
" --expert-profile P R4.2e: pre-load the VRAM tier from a `profile.bin` (see\n"
" tools/make_profile.py) instead of admitting on first use.\n"
" --expert-profile-save P --serve, #477: save what the adaptive tier learned (the experts in\n"
" VRAM, then the routing counted since the start) as a profile at P, on\n"
" QUIT and every --expert-profile-save-every MIN minutes (default 10;\n"
" 0 = on QUIT only) between requests; start from it with --expert-profile P\n"
" --no-hit-poke R4.2d's A/B arm. The hit path pokes the driver once right after its\n"
" launch so the GPU starts while the CPU pool runs; without it the work\n"
" waits for the next driver entry and does not overlap at all.\n"
" --no-ple-prefetch A/B arm: read the PLE table's sixteen rows one at a time, instead of\n"
" issuing them in one PrefetchVirtualMemory call.\n"
" --expert-cache N R4: keep N expert blobs resident in VRAM and compute their rows on the\n"
" GPU via `moe_hit_grouped_s2`. DEFAULT 0. Measured at 4096 slots\n"
" with --expert-cache-per-layer: 54.4%% hits, CPU pool drain 19.1 -> 10.3\n"
" ms/token, -2.7 ms/token end to end.\n"
" --expert-cache-device1 N pre-fill N experts on CUDA1 (experimental)\n"
" --expert-cache-device2 N pre-fill N more experts on CUDA2\n"
" --expert-cache-device3 N pre-fill N more experts on CUDA3\n"
" --expert-cache-remote-placement stripe|layer distribute expert ranks or whole\n"
" layers across CUDA1..3 (default: stripe)\n"
" --peer-device N a second GPU as an adaptive expert-cache tier (rows over P2P; it also\n"
" computes its experts' prompt rows). Not with --layer-split or\n"
" --expert-cache-device1..3. Default output unchanged without it.\n"
" --peer-reserve-mib N VRAM the peer tier leaves free on its card (default 600)\n"
" --peer-slots N expert slots on the peer (default: what fits)\n"
" --peer-adapt-swaps N peer cache swaps per adaptation step (default: --adapt-swaps)\n"
" --peer-prefill-rows N prompt rows per layer the peer computes (default half of chunk x top-k;\n"
" 0 = the prompt path stays on the primary)\n"
" --expert-cache-per-layer R4.2g: give each layer its OWN slots instead of letting the first\n"
" position take all of them. The default policy fills in arrival order\n"
" from one shared counter, so 256 slots went to ~26 layers of position 0\n"
" and measured **2.97%%**. Per-layer, the same routing gives 21.4%% at 8\n"
" slots/layer and 70.4%% at 64.\n"
" --no-host-worker R2.2: the A/B arm. By default the HOST THREAD joins the drain, so the\n"
" pool is six threads on six cores instead of five plus an idle core;\n"
" this flag restores the five-worker form for comparison on `pool phases`.\n"
" --pool-workers N R2.2: CPU expert pool worker count. Default 0 = every physical core\n"
" except the one the host loop spins on (with --pool-affinity auto or\n"
" p-cores on a hybrid CPU: P-cores minus 1). A sweep is how the pool's\n"
" deviation from `cpu_s2` is attributed.\n"
" --pool-affinity MODE Worker CPU affinity: all (default: one worker per physical core, as\n"
" always), auto (hybrid CPUs: P-cores first, then their SMT siblings,\n"
" then E-cores) or p-cores (P-cores and their siblings only).\n"
" --coupled-draft enable coupled draft sampling for MTP drafter under sampling (STRATA_SPEC_COUPLED)\n"
" --no-coupled-draft disable coupled draft sampling (propose argmax drafts)\n"
" --mmap-experts R2.1: opt OUT of the resident expert arena, back to MapViewOfFile.\n"
" The A/B arm: the mmap's rate depends on the OS page cache holding\n"
" 34 GB, and measured 71.97 vs 34.78 ms/token cold vs warm.\n"
" --shared-expert-arena FILE Linux: back the resident arena with one MAP_SHARED file.\n"
" Put this file on /dev/shm, not ordinary SSD storage.\n"
" A small header binds an existing backing file to the same pack.\n"
" --resident-cpu-experts with mmap and a static profile, keep the experts the GPU cache does not\n"
" hold resident in ordinary RAM (and the prompt path's lend region as far as\n"
" RAM allows); adaptive swaps exchange them, so none is read from the file again.\n"
" --resident-experts the low-RAM PC's resident mode (setup): --mmap-experts --resident-cpu-experts\n"
" with the copy page-locked when possible, 4 GiB headroom, plain mmap if it\n"
" does not fit. Same answers as --mmap-experts for the same placement.\n");
}
bool parse_i64_list(const char* s, std::vector<int64_t>& out, std::string& err) {
out.clear();
std::string text(s);
for (char& c : text) if (c == ',') c = ' ';
std::istringstream input(text);
std::string token;
while (input >> token) {
int32_t id = 0;
const auto parsed = std::from_chars(token.data(), token.data() + token.size(), id);
if (parsed.ec != std::errc{} || parsed.ptr != token.data() + token.size() || id < 0) {
err = "invalid token id: expected an integer in [0, 2147483647]";
out.clear();
return false;
}
out.push_back(id);
}
if (out.empty()) { err = "token list was empty"; return false; }
return true;
}
/// The pool's adapter plus the wall-clock it spent, so the report can say how much of the token was the CPU.
struct Drive {
strata::core::ExpertDispatch d;
double cpu_ms = 0;
int64_t calls = 0;
/// THE ROUTING TRACE, which is P0.S8 and a stated prerequisite of Phase 3. `drive_pool` is called once
/// per layer from the main loop - the workers live inside `expert_pool_dispatch` - so a single FILE* here
/// needs no locking. `d.layers` is the CURRENT layer on entry (the adapter increments it as it walks the
/// blob), which is why the layer index comes from there rather than from a counter of our own.
std::FILE* routing = nullptr;
};
void drive_pool(void* user, const float* x_f, const int32_t* ids, const float* weights, int64_t n_embd, int64_t k,
float* out) {
Drive* t = (Drive*) user;
const Clock::time_point a = Clock::now();
strata::core::expert_pool_dispatch(&t->d, x_f, ids, weights, n_embd, k, out);
t->cpu_ms += std::chrono::duration<double, std::milli>(Clock::now() - a).count();
++t->calls;
// THE ROUTING TRACE. Written AFTER the dispatch so the layer index is still this layer's: `d.layers` is
// advanced by the adapter as it consumes the blob, and reading it after the call is the same value the
// dispatch used. Record = int32 layer, int32 k, k int32 ids, k float weights.
if (t->routing != nullptr) {
// **`d.layers` HAS ALREADY BEEN ADVANCED BY THE TIME THIS RUNS, AND THE FIRST TRACE WAS OFF BY ONE
// BECAUSE OF IT.** The adapter walks the blob by incrementing `d.layers` as it consumes each layer's
// experts, so after the dispatch it holds the NEXT layer's index. `tools/make_profile.py` caught it
// with a bounds check when the trace turned out to span 1..48 instead of 0..47. The hit-rate CURVE was
// unaffected - it is a per-layer split, and shifting every layer by one preserves both metrics - but
// anything keyed on the layer index, which is exactly what a cache profile is, would have been wrong.
const int32_t layer_idx = (int32_t) (t->d.layers - 1);
if (layer_idx < 0 || layer_idx >= 48) {
std::fprintf(stderr, "strata generate: the routing trace saw layer %d, outside 0..47\n", layer_idx);
return;
}
const int32_t rec[2] = {layer_idx, (int32_t) k};
std::fwrite(rec, sizeof rec, 1, t->routing);
std::fwrite(ids, sizeof(int32_t), (size_t) k, t->routing);
std::fwrite(weights, sizeof(float), (size_t) k, t->routing);
}
}
/// Plan v0.3 P6: the pool for a verify window.
void drive_pool_multi(void* user, const float* x_f, const int32_t* ids, int64_t n_tok, int64_t k, float* out,
int64_t layer) {
Drive* t = (Drive*) user;
t->d.layers = layer;
const Clock::time_point a = Clock::now();
strata::core::expert_pool_dispatch_multi(t->d, x_f, ids, n_tok, k, out);
t->cpu_ms += std::chrono::duration<double, std::milli>(Clock::now() - a).count();
++t->calls;
// the routing trace for the serve path: the same record format drive_pool writes (layer, k, ids, weights),
// one record per token. The multi dispatch fuses the router weights into the kernel and does not surface
// them, so records carry unit weights: tools/make_profile.py ranks pairs by routed frequency, which is the
// signal that matters; a one-shot --dump-routing run records true weights if a weighted ranking is wanted.
if (t->routing != nullptr && layer >= 0 && layer < 48) {
for (int64_t tok = 0; tok < n_tok; ++tok) {
const int32_t rec[2] = {(int32_t) layer, (int32_t) k};
std::fwrite(rec, sizeof rec, 1, t->routing);
std::fwrite(ids + tok * k, sizeof(int32_t), (size_t) k, t->routing);
static const float one[64] = {}; // k <= 64 in a verify window; zeros read as unit weights
std::fwrite(one, sizeof(float), (size_t) k, t->routing);
}
}
}
/// Layer split: every verify stage shares one Drive (its counters, usage and failure flags); the GPU plan, the expert
/// cache and the PCIe share the pool uses for a layer are those of the stage that runs it.
struct SplitDrive {
static constexpr int kMax = 8;
Drive* base = nullptr;
int n = 0; ///< stages
int64_t end[kMax] = {}; ///< stage i runs the layers from end[i - 1] (0) below end[i]
strata::core::GpuPlanSink* plan[kMax] = {};
const uint8_t* cache_base[kMax] = {};
const uint64_t* cache_slot_off[kMax] = {};
int pcie_num[kMax] = {};
};
void drive_pool_split(void* user, const float* x_f, const int32_t* ids, int64_t n_tok, int64_t k, float* out,
int64_t layer) {
SplitDrive* s = (SplitDrive*) user;
int st = 0;
while (st + 1 < s->n && layer >= s->end[st]) ++st;
Drive& d = *s->base;
d.d.plan = s->plan[st];
d.d.cache_base = s->cache_base[st];
d.d.cache_slot_off = s->cache_slot_off[st];
d.d.pcie_num = s->pcie_num[st];
drive_pool_multi(s->base, x_f, ids, n_tok, k, out, layer);
}
/// Layer split across GPUs: a later stage on its own device, with its own copy of the dense weights, a session, an
/// expert cache for its layers, a verify window and a prompt path; the last one also holds the head (the drafter
/// lives on its device too).
struct GpuStage {
int dev = 0;
int64_t lb = 0, le = 0;
double pcie_frac = 0.0;
strata::core::WeightTable wt;
strata::core::NativeDense dense;
strata::core::NativeHead head;
strata::core::SessionState ss;
cudaStream_t stream = nullptr;
strata::core::ExpertCache cache;
std::vector<std::pair<int32_t, int32_t>> profile; ///< its layers' share of the profile, hottest first
int32_t* d_res = nullptr; ///< the residency table on its device
strata::core::Verifier ver;
strata::prefill::Prefill sp;
cudaStream_t adapt_stream = nullptr;
cudaEvent_t adapt_ev = nullptr;
bool adapt_live = false; ///< swaps of this request are in flight on it
int32_t* mrope = nullptr; ///< --vision: the image-position table on its device
};
// ---- issue #31: what the watchdog prints before it stops a stalled engine
struct MemSample {
unsigned long long faults = 0, rss_mib = 0, avail_mib = 0, commit_mib = 0; // faults: hard (Linux) / all (Windows)
};
MemSample mem_sample() {
MemSample m;
#if defined(_WIN32)
PROCESS_MEMORY_COUNTERS pmc{};
if (K32GetProcessMemoryInfo(GetCurrentProcess(), &pmc, sizeof pmc)) {
m.faults = pmc.PageFaultCount;
m.rss_mib = pmc.WorkingSetSize >> 20;
m.commit_mib = pmc.PagefileUsage >> 20;
}
MEMORYSTATUSEX ms{};
ms.dwLength = sizeof ms;
if (GlobalMemoryStatusEx(&ms)) m.avail_mib = ms.ullAvailPhys >> 20;
#else
if (std::FILE* f = std::fopen("/proc/self/stat", "r")) {
char buf[4096];
const size_t n = std::fread(buf, 1, sizeof buf - 1, f);
buf[n] = 0;
std::fclose(f);
const char* s = std::strrchr(buf, ')'); // fields after the command name: 3 state ... 12 majflt
for (int field = 2; s && field < 12; ++field) s = std::strchr(s + 1, ' ');
if (s) m.faults = std::strtoull(s + 1, nullptr, 10);
}
auto kb = [](const char* path, const char* key) -> unsigned long long {
unsigned long long v = 0;
if (std::FILE* f = std::fopen(path, "r")) {
char line[256];
const size_t kl = std::strlen(key);
while (std::fgets(line, sizeof line, f))
if (std::strncmp(line, key, kl) == 0) { v = std::strtoull(line + kl, nullptr, 10); break; }
std::fclose(f);
}
return v;
};
m.rss_mib = kb("/proc/self/status", "VmRSS:") >> 10;
m.commit_mib = kb("/proc/self/status", "VmSwap:") >> 10;
m.avail_mib = kb("/proc/meminfo", "MemAvailable:") >> 10;
#endif
return m;
}
// the stage the watchdog names: "<where> <detail>", and the prompt chunk a batched read is in (#251)
std::string stage_text() {
const strata::core::Progress& p = strata::core::progress();
std::string s = std::string(p.where.load()) + " " + std::to_string((long long) p.detail.load());
if (const int64_t c = p.chunk.load(); c >= 0) s += " of the prompt chunk from token " + std::to_string((long long) c);
return s;
}
void stall_report(std::FILE* f, uint64_t layers_during) {
strata::core::Progress& p = strata::core::progress();
std::fprintf(f, "strata serve: stall report (engine %s): stage \"%s\" for %lld s; %llu layers served since the "
"last finished step (0 = stopped, more = slow)\n", STRATA_VERSION, stage_text().c_str(),
(long long) ((strata::core::progress_now_ms() - p.since_ms.load()) / 1000),
(unsigned long long) layers_during);
for (int pass = 0; pass < 2; ++pass) {
if (pass == 1) {
std::this_thread::sleep_for(std::chrono::seconds(2));
std::fprintf(f, " 2 s later:\n");
}
if (auto fn = strata::core::diag_pool_fn().load()) fn(f);
if (auto fn = strata::core::diag_verify_fn().load()) fn(f);
const MemSample m = mem_sample();
std::fprintf(f, " memory: %llu MiB resident, %llu MiB %s, %llu MiB RAM available; %llu %s\n", m.rss_mib,
m.commit_mib,
#if defined(_WIN32)
"committed", m.avail_mib, m.faults, "page faults so far"
#else
"in swap", m.avail_mib, m.faults, "major page faults so far"
#endif
);
std::fflush(f);
}
#if defined(_WIN32)
// every thread's stack, to read against this build's PDB: a few MB beside the engine's working directory
if (HMODULE dbg = LoadLibraryA("dbghelp.dll")) {
using Fn = BOOL(WINAPI*)(HANDLE, DWORD, HANDLE, int, void*, void*, void*);
if (auto write = (Fn) GetProcAddress(dbg, "MiniDumpWriteDump")) {
char path[64];
std::snprintf(path, sizeof path, "strata-stall-%lu.dmp", (unsigned long) GetCurrentProcessId());
HANDLE h = CreateFileA(path, GENERIC_WRITE, 0, nullptr, CREATE_ALWAYS, FILE_ATTRIBUTE_NORMAL, nullptr);
if (h != INVALID_HANDLE_VALUE) {
const int kThreadInfo = 0x1000; // MiniDumpWithThreadInfo; MiniDumpNormal = 0
const BOOL ok = write(GetCurrentProcess(), GetCurrentProcessId(), h, kThreadInfo, nullptr, nullptr, nullptr);
CloseHandle(h);
char full[MAX_PATH];
if (!GetFullPathNameA(path, MAX_PATH, full, nullptr)) std::snprintf(full, sizeof full, "%s", path);
std::fprintf(f, " %s the thread stacks to %s (attach it to the issue)\n", ok ? "wrote" : "could not write",
full);
}
}
}
#endif
}
/// STRATA_TRACE=1: the VRAM left at a step of the startup (finds what fills the card after the cache is sized)
void mem_mark(const char* where) {
static const bool on = std::getenv("STRATA_TRACE") != nullptr;
if (!on) return;
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
std::fprintf(stderr, "strata trace: %lld MiB free after %s\n", (long long) (free_b >> 20), where);
}
/// #463's A/B: STRATA_ADAPT_NOWAIT=1 lets a verify window start before the adaptive tier's copies have landed (0.1.37)
bool adapt_nowait() {
static const bool v = [] { const char* e = std::getenv("STRATA_ADAPT_NOWAIT"); return e && e[0] == '1'; }();
return v;
}
int argmax(const std::vector<float>& v) {
int best = 0;
for (size_t i = 1; i < v.size(); ++i)
if (v[i] > v[best]) best = (int) i;
return best;
}
// ---- --serve's conversation cache. A chat or an agent sends the whole conversation again with every request, and
// reading it again is what made a long session wait minutes for every turn. What a sequence leaves behind splits in
// two, and only one half needs copying:
// * POSITIONAL state - the KV cache of the 12 QSA layers and their pooled indexer keys, the draft layer's KV. A
// cell is written once for its position and read only by later positions (the block scores take `dead` for the
// block being filled, never its pooled row), so rewinding to a position just means writing from there again.
// * RUNNING state - the 36 GDN recurrences and conv histories, each QSA layer's indexer tail (the unfinished
// block's raw keys) and the PLE's normalized history. Each describes "everything so far" and cannot be
// rewound, so a checkpoint is a copy of exactly these: ~118 MB, the same set the verifier snapshots to roll
// back rejected drafts.
// A checkpoint is only valid while the positional cells below it still hold ITS tokens, so the serve loop keeps
// just the checkpoints that are prefixes of the tokens the session holds now.
using ImgKey = strata::core::ConversationImageKey;
using ConvCheckpoint = strata::core::ConversationCheckpoint;
uint64_t fnv1a(const void* data, size_t n, uint64_t h = 1469598103934665603ull) {
const uint8_t* p = (const uint8_t*) data;
for (size_t i = 0; i < n; ++i) { h ^= p[i]; h *= 1099511628211ull; }
return h;
}
using ConvStateSizes = strata::core::ConversationStateSizes;
ConvStateSizes conv_state_sizes(const strata::core::ModelGeometry& g, const strata::core::SessionState& ss) {
ConvStateSizes z;
std::string error;
// a split stage's session owns only its layer range's state (see SessionState's carve note); the geometry
// and the carve have already passed engine validation
strata::core::conversation_session_sizes(g, ss, z, error);
return z;
}
/// Copies the running state out (this session's carve only). The caller has synchronized the device.
bool checkpoint_save(ConvCheckpoint& c, const strata::core::SessionState& ss, const strata::core::ModelGeometry& g) {
std::string error;
if (strata::core::conversation_checkpoint_save(c, ss, g, error)) return true;
std::fprintf(stderr, "strata serve: checkpoint save: %s\n", error.c_str()); // the caller's ERR has no reason
return false;
}
/// Puts a checkpoint's running state back; the positional cells below it are the caller's to guarantee.
bool checkpoint_restore(const ConvCheckpoint& c, strata::core::SessionState& ss, const strata::core::ModelGeometry& g) {
std::string error;
if (strata::core::conversation_checkpoint_restore(c, ss, g, error)) return true;
std::fprintf(stderr, "strata serve: checkpoint restore: %s\n", error.c_str());
return false;
}
// --control-vector-scaled: llama.cpp's `common_control_vector_load` (every file's `direction.<l>` times its scale,
// summed; layer 0 has none) and `llama_adapter_cvec::apply` with the projection-mode patch (project: the unit
// direction and its norm as the scale), into the tables `cvec_upload` takes. `summary` is what INFO reports.
bool load_control_vectors(const Options& o, const strata::core::ModelGeometry& g, std::string& summary, std::string& err) {
const int64_t L = g.n_layers, N = g.n_embd;
std::vector<float> data((size_t) (L * N), 0.0f);
std::vector<bool> have((size_t) L, false);
for (const auto& [path, scale] : o.cvec_files) {
try {
strata::GgufFile f(path);
const strata::MetaValue* arch = f.get("general.architecture");
if (arch == nullptr || arch->s != "controlvector") {
err = path + ": not a control vector GGUF (general.architecture is not 'controlvector')";
return false;
}
const strata::MetaValue* hint = f.get("controlvector.model_hint");
if (hint != nullptr && hint->s != "qwen4exp")
std::fprintf(stderr, "strata generate: %s was made for '%s', not qwen4exp\n", path.c_str(), hint->s.c_str());
int found = 0;
for (const strata::TensorInfo& t : f.tensors()) {
if (t.name.rfind("direction.", 0) != 0) continue;
const long l = std::strtol(t.name.c_str() + 10, nullptr, 10);
if (l < 1 || l >= L) continue; // layer 0 has no vector; past the model is ignored, as in llama.cpp
if (t.type != 0 || t.elements() != (uint64_t) N) {
err = path + ": " + t.name + " must be " + std::to_string((long long) N) + " f32";
return false;
}
const float* src = reinterpret_cast<const float*>(f.tensor_data(t));
for (int64_t j = 0; j < N; ++j) data[(size_t) (l * N + j)] += scale * src[j];
have[(size_t) l] = true;
++found;
}
if (found == 0) { err = path + ": no direction.<layer> tensors"; return false; }
} catch (const std::exception& e) {
err = e.what();
return false;
}
}
const int first = o.cvec_first <= 0 ? 1 : o.cvec_first;
const int last = (o.cvec_last <= 0 || o.cvec_last >= L) ? (int) L - 1 : o.cvec_last;
const int single = o.cvec_mode == 0 ? o.cvec_single : -1;
if (single >= 0 && (single >= L || !have[(size_t) single])) {
err = "--cvec-dir single:" + std::to_string(single) + ": the vector has no direction for that layer";
return false;
}
std::vector<float> dir((size_t) (L * N), 0.0f), s((size_t) L, 0.0f);
int steered = 0;
for (int64_t l = first; l <= last; ++l) {
const int64_t src = single >= 0 ? single : l;
if (!have[(size_t) src]) continue;
const float* d = data.data() + (size_t) (src * N);
if (o.cvec_mode == 0) {
double nrm = 0.0;
for (int64_t j = 0; j < N; ++j) nrm += (double) d[j] * d[j];
nrm = std::sqrt(nrm);
if (nrm <= 0.0) continue;
s[(size_t) l] = (float) nrm;
for (int64_t j = 0; j < N; ++j) dir[(size_t) (l * N + j)] = (float) (d[j] / nrm);
} else {
s[(size_t) l] = 1.0f;
std::copy(d, d + N, dir.begin() + (size_t) (l * N));
}
++steered;
}
if (steered == 0) { err = "the control vector has no direction in layers " + std::to_string(first) + ".." + std::to_string(last); return false; }
if (!strata::kernels::cvec_upload(dir, s, o.cvec_mode, first, last, N, g.hc, err)) return false;
summary = std::string(o.cvec_mode == 0 ? "project" : "add") + ":" + std::to_string(first) + "-" + std::to_string(last) +
(single >= 0 ? ":single" + std::to_string(single) : "");
// the line llama.cpp's patched build prints, so a log shows the same thing
std::fprintf(stderr, "strata generate: control vector mode = %s, dir = %s, layers %d..%d (%d steered)\n",
o.cvec_mode == 0 ? "project" : "add", single >= 0 ? "single" : "per-layer", first, last, steered);
return true;
}
// The effective host->device bandwidth of the PCIe link: copies from pinned host memory, as the expert arena's
// reads are. The native default share (0.55) was measured on x16 links (~26-28 GB/s); a x8 card in a x8 slot
// carries about half of that. Returns < 0 when the probe cannot run (then the caller keeps the default).
//
// #485: a single timed burst read an x16 PCIe 4 link (RTX A3000 laptop) at 18.5, 6.9 and 5.8 GB/s in three starts -
// a link still in a low-power state, other DMA, a context not yet up to speed - and the low reading set the share.
// Such a disturbance only ever slows a copy: nothing makes a correctly timed copy faster than the link carries. So
// the same 1 GiB is now copied as four bursts of 256 MiB, timed one by one, and the fastest is the link's figure (the
// median would still follow a disturbance that lasts through half the bursts). Each burst takes ~10 ms on an x16
// PCIe 4 link, so the probe takes no longer than the one 1 GiB burst did. `samples`, when given, gets every
// burst's reading for the log.
double probe_pcie_h2d_gbps(std::string* samples = nullptr) {
constexpr size_t kBytes = 256ull << 20;
constexpr int kBursts = 4;
void* h = nullptr;
void* d = nullptr;
if (cudaMallocHost(&h, kBytes) != cudaSuccess) return -1.0;
if (cudaMalloc(&d, kBytes) != cudaSuccess) {
cudaFreeHost(h);
return -1.0;
}
std::memset(h, 0, kBytes); // fault the pages in before timing
cudaMemcpyAsync(d, h, kBytes, cudaMemcpyHostToDevice); // warmup: context up, copy engine primed
float ms[kBursts] = {};
#if defined(STRATA_USE_HIP) && defined(_WIN32)
// Windows HIP: the events do not bracket the copies there (an RX 6800 read 3,300-26,000 GB/s, so every link kept
// the x16 share), so each burst is timed on the host between two synchronizes: 256 MiB takes ~10 ms, so the
// synchronize around it hardly matters
bool ok = cudaDeviceSynchronize() == cudaSuccess;
for (int b = 0; b < kBursts && ok; ++b) {
const Clock::time_point t0 = Clock::now();
cudaMemcpyAsync(d, h, kBytes, cudaMemcpyHostToDevice);
ok = cudaDeviceSynchronize() == cudaSuccess;
ms[b] = (float) std::chrono::duration<double, std::milli>(Clock::now() - t0).count();
}
#else
// the bursts run back to back on the stream, an event between each two
cudaEvent_t ev[kBursts + 1];
int n_ev = 0;
while (n_ev <= kBursts && cudaEventCreate(&ev[n_ev]) == cudaSuccess) ++n_ev;
bool ok = n_ev == kBursts + 1;
if (ok) {
cudaEventRecord(ev[0]);
for (int b = 0; b < kBursts; ++b) {
cudaMemcpyAsync(d, h, kBytes, cudaMemcpyHostToDevice);
cudaEventRecord(ev[b + 1]);
}
ok = cudaEventSynchronize(ev[kBursts]) == cudaSuccess;
for (int b = 0; b < kBursts && ok; ++b) ok = cudaEventElapsedTime(&ms[b], ev[b], ev[b + 1]) == cudaSuccess;
}
for (int i = 0; i < n_ev; ++i) cudaEventDestroy(ev[i]);
#endif
double bw = -1.0;
if (samples != nullptr) samples->clear();
for (int b = 0; b < kBursts && ok; ++b) {
const double s = ms[b] > 0.01f ? (double) kBytes / (ms[b] * 1e-3) / 1e9 : -1.0;
bw = std::max(bw, s);
if (samples != nullptr) {
char buf[32];
std::snprintf(buf, sizeof(buf), "%s%.1f", b == 0 ? "" : " ", s);
*samples += buf;
}
}
if (!ok) (void) cudaGetLastError();
cudaFree(d);
cudaFreeHost(h);
return bw;
}
// The PCIe share of the missed experts for a link measured at `gbps`: `base` (the share measured on x16 links) from
// 20 GB/s up, and below that in proportion to the bandwidth, so the time the link spends on its share stays about
// what the x16 share costs. Continuous (#485): before, 19.9 GB/s gave 0.42 and 20.0 the full 0.55 (and 4.0 GB/s
// gave 0.08, 3.9 none), so a reading near either edge moved the share by a quarter of its range. From 20 GB/s up
// (an x16 PCIe 4/5 link: ~26-28 GB/s) the share is unchanged.
double pcie_frac_for_gbps(double gbps, double base) {
return gbps <= 0.0 ? base : base * std::min(1.0, gbps / 20.0);
}
} // namespace
int main(int argc, char** argv) {
// **UNBUFFERED, BECAUSE THE INTERESTING OUTPUT IS THE OUTPUT BEFORE A CRASH.** `stdout` redirected to a
// pipe or a file is block-buffered, so a program that dies loses every line it had already printed - which
// turns "it crashed at step 7" into "it crashed somewhere", and the difference is a debugging session.
std::setvbuf(stdout, nullptr, _IONBF, 0);
// Load every CUDA kernel when the context is created, before the expert cache takes the free VRAM. With the
// default lazy loading, a kernel first used mid-prompt (MMQ for IQ3_XXS at 64K+ on a 12 GB card) found no VRAM
// left for its code and the engine ended ("out of memory: cudaFuncSetAttribute"). Costs ~30 MB of VRAM.
if (std::getenv("CUDA_MODULE_LOADING") == nullptr) {
#if defined(_WIN32)
_putenv_s("CUDA_MODULE_LOADING", "EAGER");
#else
setenv("CUDA_MODULE_LOADING", "EAGER", 0);
#endif
}
Options o;
bool have_tokens = false;
bool have_logits_stride = false;
for (int i = 1; i < argc; ++i) {
const std::string a = argv[i];
auto next = [&](const char* what) -> const char* {
if (i + 1 >= argc) { std::fprintf(stderr, "%s needs a value\n", what); std::exit(2); }
return argv[++i];
};
bool parsed = true;
if (a == "--help" || a == "-h") { usage(); return 0; }
else if (a == "--pack") o.pack = next("--pack");
else if (a == "--tokens") {
if (have_tokens) { std::fprintf(stderr, "supply one token input only\n"); return 2; }
std::string e;
if (!parse_i64_list(next("--tokens"), o.tokens, e)) { std::fprintf(stderr, "%s\n", e.c_str()); return 2; }
have_tokens = true;
}
else if (a == "--tokens-file") {
if (have_tokens) { std::fprintf(stderr, "supply one token input only\n"); return 2; }
std::ifstream input(next("--tokens-file"));
if (!input) { std::fprintf(stderr, "cannot open token file\n"); return 2; }
std::string text((std::istreambuf_iterator<char>(input)), std::istreambuf_iterator<char>());
if (input.bad()) { std::fprintf(stderr, "cannot read token file\n"); return 2; }
std::string error;
if (text.find('\0') != std::string::npos || !parse_i64_list(text.c_str(), o.tokens, error)) {
std::fprintf(stderr, "malformed token file: %s\n", error.c_str()); return 2;
}
have_tokens = true;
}
else if (a == "--max-new") o.max_new = std::atoll(next("--max-new"));
else if (a == "--max-context") o.max_context = std::atoll(next("--max-context"));
else if (a == "--rope-scaling") o.rope_scaling = next("--rope-scaling");
else if (a == "--rope-scale") o.rope_scale = std::atof(next("--rope-scale"));
else if (a == "--rope-freq-base") o.rope_freq_base = std::atof(next("--rope-freq-base"));
else if (a == "--rope-freq-scale") o.rope_freq_scale = std::atof(next("--rope-freq-scale"));
else if (a == "--yarn-orig-ctx") o.yarn_orig_ctx = std::atof(next("--yarn-orig-ctx"));
else if (a == "--yarn-ext-factor") o.yarn_ext_factor = std::atof(next("--yarn-ext-factor"));
else if (a == "--yarn-attn-factor") o.yarn_attn_factor = std::atof(next("--yarn-attn-factor"));
else if (a == "--yarn-beta-fast") o.yarn_beta_fast = std::atof(next("--yarn-beta-fast"));
else if (a == "--yarn-beta-slow") o.yarn_beta_slow = std::atof(next("--yarn-beta-slow"));
else if (a == "--greedy") o.greedy = true;
else if (a == "--seed") { o.seed = (uint64_t) std::atoll(next("--seed")); o.greedy = false; }
else if (a == "--top-k") o.top_k = std::atoi(next("--top-k"));
else if (a == "--top-p") o.top_p = (float) std::atof(next("--top-p"));
else if (a == "--temperature") o.temperature = (float) std::atof(next("--temperature"));
else if (a == "--dump-logits") o.dump_logits = next("--dump-logits");
else if (a == "--logits-stride") {
if (have_logits_stride) { std::fprintf(stderr, "--logits-stride must be supplied only once\n"); return 2; }
if (!strata::program::logits_selection::parse_stride(next("--logits-stride"), o.logits_stride)) {
std::fprintf(stderr, "--logits-stride requires a positive decimal int64\n"); return 2;
}
have_logits_stride = true;
}
else if (a == "--dump-residual") o.dump_residual = next("--dump-residual");
else if (a == "--dump-mixed") o.dump_mixed = next("--dump-mixed");
else if (a == "--dump-layers") o.dump_layers = next("--dump-layers");
else if (a == "--dump-halves") o.dump_halves = next("--dump-halves");
else if (a == "--dump-routing") o.dump_routing = next("--dump-routing");
else if (a == "--ple-gguf") o.ple_gguf = next("--ple-gguf");
else if (a == "--no-ple") o.no_ple = true;
else if (a == "--ple-io") o.ple_io = next("--ple-io");
else if (a == "--ple-row-cache") o.ple_row_cache = std::atoll(next("--ple-row-cache"));
else if (a == "--ple-inflight") o.ple_inflight = std::atoi(next("--ple-inflight"));
else if (a == "--ple-delay-us") o.ple_delay_us = std::atof(next("--ple-delay-us"));
else if (a == "--ple-sync-submit") o.ple_sync_submit = true;
else if (a == "--kv") o.kv = next("--kv");
else if (a == "--kv-resident") o.kv_resident = std::atoll(next("--kv-resident"));
else if (a == "--stream-token") o.stream_token = true;
else if (a == "--check-logits") o.check_logits = true;
else if (a == "--gr-fp32-activations") o.gr_fp32_activations = true;
else if (a == "--gr-native-mmvf") o.gr_native_mmvf = true;
else if (a == "--native-bf16") o.native_bf16 = true;
else if (a == "--native-bf16-extra") o.native_bf16_extra = true;
else if (a == "--native-ple-key") o.native_ple_key = true;
else if (a == "--native-moe-combine") o.native_moe_combine = true;
else if (a == "--native-gdn") o.native_gdn = true;
else if (a == "--native-flash-attn-short") o.native_flash_attn_short = true;
else if (a == "--native-qsa-indexer") o.native_qsa_indexer = true;
else if (a == "--native-qsa") o.native_qsa = true;
else if (a == "--native-rope") o.native_rope = true;
else if (a == "--native-ple-postops") o.native_ple_postops = true;
else if (a == "--native-router") o.native_router = true;
else if (a == "--cpu-oracle-q8-0") o.cpu_oracle_q8_0 = true;
else if (a == "--native") o.native_preset = next("--native");
else if (a == "--native-head-gguf") o.native_head_gguf = next("--native-head-gguf");
else if (a == "--embd-gguf") o.embd_gguf = next("--embd-gguf");
else if (a == "--native-dense-gguf") o.native_dense_gguf.push_back(next("--native-dense-gguf"));
else if (a == "--no-capture") o.no_capture = true;
else if (a == "--no-pool") o.no_pool = true;
else if (a == "--sync-every-layer") o.sync_every_layer = true;
else if (a == "--stage-timing") o.stage_timing = true;
else if (a == "--graph-only") o.graph_only = true;
else if (a == "--gpu-only-full") o.gpu_only_full = true;
else if (a == "--pool-workers") o.pool_workers = std::atoi(next("--pool-workers"));
else if (a == "--pool-affinity") {
const std::string v = next("--pool-affinity");
if (v == "auto") o.pool_affinity = strata::kernels::cpu::PoolAffinity::Auto;
else if (v == "p-cores" || v == "pcores") o.pool_affinity = strata::kernels::cpu::PoolAffinity::PCores;
else if (v == "all") o.pool_affinity = strata::kernels::cpu::PoolAffinity::All;
else {
std::fprintf(stderr, "strata generate: unknown --pool-affinity value '%s' (expected auto, p-cores, or all)\n", v.c_str());
return 2;
}
}
else if (a == "--no-host-worker") o.no_host_worker = true;
else if (a == "--no-ple-prefetch") o.no_ple_prefetch = true;
else if (a == "--coupled-draft") o.coupled_draft = true;
else if (a == "--no-coupled-draft") o.coupled_draft = false;
else parsed = false;
// The chain continues here in a second statement: one chain of 120+ `else if` passed MSVC's limit of 128
// nested blocks (C1061). The order of the tests and what each does are unchanged.
if (!parsed) {
if (a == "--expert-cache") {
const std::string v = next("--expert-cache");
o.expert_cache = (v == "auto") ? -1 : std::atoi(v.c_str());
}
else if (a == "--expert-cache-device1") o.expert_cache_remote[0] = std::atoi(next("--expert-cache-device1"));
else if (a == "--expert-cache-device2") o.expert_cache_remote[1] = std::atoi(next("--expert-cache-device2"));
else if (a == "--expert-cache-device3") o.expert_cache_remote[2] = std::atoi(next("--expert-cache-device3"));
else if (a == "--expert-cache-remote-placement")
o.expert_cache_remote_placement = next("--expert-cache-remote-placement");
else if (a == "--vram-reserve-mib") { o.vram_reserve_mib = std::atoi(next("--vram-reserve-mib")); o.vram_reserve_given = true; }
else if (a == "--prefill") {
const std::string v = next("--prefill");
o.prefill_auto = v == "auto" || v.rfind("auto:", 0) == 0;
// #282: auto:N (N = 16384 or 32768) or STRATA_PREFILL_AUTO_MAX lets auto take chunks above 8192
const char* env_max = std::getenv("STRATA_PREFILL_AUTO_MAX");
const long long want_max = v.rfind("auto:", 0) == 0 ? std::atoll(v.c_str() + 5)
: (o.prefill_auto && env_max != nullptr ? std::atoll(env_max) : 8192);
o.prefill_auto_max = want_max >= 32768 ? 32768 : want_max >= 16384 ? 16384 : 8192;
o.prefill_chunk = o.prefill_auto ? o.prefill_auto_max : std::atoll(v.c_str());
}
else if (a == "--no-split-rows") o.no_split_rows = true;
else if (a == "--no-prefill-borrow") o.no_prefill_borrow = true;
else if (a == "--prefill-until") o.prefill_until = std::atoll(next("--prefill-until"));
else if (a == "--dump-final-r") o.dump_final_r = next("--dump-final-r");
else if (a == "--spec") o.spec = std::atoi(next("--spec"));
else if (a == "--spec-oracle") o.spec_oracle = next("--spec-oracle");
else if (a == "--spec-corrupt") o.spec_corrupt = std::atoi(next("--spec-corrupt"));
else if (a == "--mtp") o.mtp = next("--mtp");
else if (a == "--mtp-window") o.mtp_window = std::atoll(next("--mtp-window"));
else if (a == "--pcie-frac") o.pcie_frac = std::atof(next("--pcie-frac"));
else if (a == "--adapt-every") o.adapt_every = std::atoi(next("--adapt-every"));
else if (a == "--adapt-decay") o.adapt_decay = (float) std::atof(next("--adapt-decay"));
else if (a == "--spec-min-p") o.spec_min_p = std::atof(next("--spec-min-p"));
else if (a == "--stop-eos") o.stop_eos = true;
else if (a == "--spec-split") o.spec_split = true;
else if (a == "--layer-split") o.layer_split = next("--layer-split");
else if (a == "--split-device") o.split_device = next("--split-device");
else if (a == "--split-skip-if-fits") o.split_skip_if_fits = true;
else if (a == "--pcie-mode") o.pcie_mode = next("--pcie-mode");
else if (a == "--serve") o.serve = true;
else if (a == "--vision") o.vision = true;
else if (a == "--prompt-cache") o.prompt_cache = std::max(0, std::atoi(next("--prompt-cache")));
else if (a == "--conversation-cache-mib" || a == "--conversation-cache-slots" ||
a == "--conversation-cache-min-free-mib") {
const std::string value = next(a.c_str());
int64_t number = 0;
const auto result = std::from_chars(value.data(), value.data() + value.size(), number);
const int64_t limit = a == "--conversation-cache-slots" ? INT32_MAX : INT64_MAX / (1024 * 1024);
if (result.ec != std::errc{} || result.ptr != value.data() + value.size() || number < 0 || number > limit) {
std::fprintf(stderr, "%s needs a nonnegative integer within range\n", a.c_str());
return 2;
}
if (a == "--conversation-cache-mib") o.conversation_cache_mib = number;
else if (a == "--conversation-cache-min-free-mib") o.conversation_cache_min_free_mib = number;
else o.conversation_cache_slots = (int) number;
}
else if (a == "--prompt-cache-every") o.prompt_cache_every = std::max(0LL, std::atoll(next("--prompt-cache-every")));
else if (a == "--prompt-cache-root") o.prompt_cache_root = std::max(0LL, std::atoll(next("--prompt-cache-root")));
else if (a == "--turn-token") o.turn_token = std::atoll(next("--turn-token"));
else if (a == "--short-read") o.short_read = std::max(0LL, std::atoll(next("--short-read")));
else if (a == "--suffix-draft") o.suffix_draft = std::max(0, std::atoi(next("--suffix-draft")));
else if (a == "--mtp-max-t") o.mtp_max_t = std::max(0, std::atoi(next("--mtp-max-t")));
else if (a == "--control-vector") o.cvec_files.push_back({next("--control-vector"), 1.0f});
else if (a == "--control-vector-scaled") {
// FILE:SCALE, comma-separated; the LAST colon splits, so a Windows path (C:\...) keeps its drive
std::stringstream list(next("--control-vector-scaled"));
std::string item;
while (std::getline(list, item, ',')) {
const size_t colon = item.rfind(':');
char* end = nullptr;
const float sc = colon == std::string::npos ? 0.0f : std::strtof(item.c_str() + colon + 1, &end);
if (colon == std::string::npos || colon == 0 || end == item.c_str() + colon + 1 || *end != '\0') {
std::fprintf(stderr, "--control-vector-scaled: expected FILE:SCALE, got '%s'\n", item.c_str());
return 2;
}
o.cvec_files.push_back({item.substr(0, colon), sc});
}
}
else if (a == "--control-vector-layer-range") {
o.cvec_first = std::atoi(next("--control-vector-layer-range"));
o.cvec_last = std::atoi(next("--control-vector-layer-range"));
}
else if (a == "--cvec-mode") {
const std::string m = next("--cvec-mode");
if (m == "project") o.cvec_mode = 0;
else if (m == "add") o.cvec_mode = 1;
else { std::fprintf(stderr, "--cvec-mode: add or project, got '%s'\n", m.c_str()); return 2; }
}
else if (a == "--cvec-dir") {
const std::string d = next("--cvec-dir");
if (d == "per-layer") o.cvec_single = -1;
else if (d.rfind("single:", 0) == 0) o.cvec_single = std::atoi(d.c_str() + 7);
else { std::fprintf(stderr, "--cvec-dir: per-layer or single:L, got '%s'\n", d.c_str()); return 2; }
}
else if (a == "--no-spec-split") o.spec_split = false;
else if (a == "--eos-ids") {
std::string e;
if (!parse_i64_list(next("--eos-ids"), o.eos_ids, e)) { std::fprintf(stderr, "--eos-ids: %s\n", e.c_str()); return 2; }
o.stop_eos = true;
}
else if (a == "--adapt-swaps") o.adapt_swaps = std::atoi(next("--adapt-swaps"));
else if (a == "--expert-cache-cpu-order") o.expert_cache_cpu_order = true;
else if (a == "--expert-cache-per-layer") o.expert_cache_per_layer = true;
else if (a == "--peer-device") o.peer_device = std::atoi(next("--peer-device"));
else if (a == "--peer-reserve-mib") o.peer_reserve_mib = std::atoi(next("--peer-reserve-mib"));
else if (a == "--peer-slots") o.peer_slots = std::atoll(next("--peer-slots"));
else if (a == "--peer-adapt-swaps") o.peer_adapt_swaps = std::atoi(next("--peer-adapt-swaps"));
else if (a == "--peer-prefill-rows") o.peer_prefill_rows = std::atoll(next("--peer-prefill-rows"));
else if (a == "--no-hit-poke") o.no_hit_poke = true;
else if (a == "--expert-profile") o.expert_profile = next("--expert-profile");
else if (a == "--expert-profile-save") o.expert_profile_save = next("--expert-profile-save");
else if (a == "--expert-profile-save-every")
o.expert_profile_save_min = std::atof(next("--expert-profile-save-every"));
else if (a == "--gpu-stages") o.gpu_stages = true;
else if (a == "--mmap-experts") o.mmap_experts = true;
else if (a == "--shared-expert-arena") o.shared_expert_arena = next("--shared-expert-arena");
else if (a == "--resident-cpu-experts") o.resident_cpu_experts = o.resident_cpu_explicit = true;
else if (a == "--resident-experts") {
o.mmap_experts = o.resident_cpu_experts = o.resident_pin = o.resident_soft = true;
o.resident_headroom = 4ull << 30;
// A/B arms: STRATA_RESIDENT_PIN=0 keeps the copy pageable (the --resident-cpu-experts form);
// STRATA_RESIDENT_HEADROOM_GIB=N leaves N GiB of the available RAM free instead of 4
if (const char* v = std::getenv("STRATA_RESIDENT_PIN"); v != nullptr && std::string(v) == "0")
o.resident_pin = false;
if (const char* v = std::getenv("STRATA_RESIDENT_HEADROOM_GIB"); v != nullptr && std::atof(v) >= 0.0)
o.resident_headroom = (uint64_t) (std::atof(v) * 1073741824.0);
}
else if (a == "--resident-budget-gib") {
const double gib = std::atof(next("--resident-budget-gib"));
if (!(gib > 0.0)) { std::fprintf(stderr, "strata generate: --resident-budget-gib needs N > 0\n"); return 2; }
o.resident_budget = (uint64_t) (gib * 1073741824.0);
o.mmap_experts = o.resident_cpu_experts = o.resident_pin = true;
o.resident_headroom = 4ull << 30;
if (const char* v = std::getenv("STRATA_RESIDENT_PIN"); v != nullptr && std::string(v) == "0")
o.resident_pin = false;
if (const char* v = std::getenv("STRATA_RESIDENT_HEADROOM_GIB"); v != nullptr && std::atof(v) >= 0.0)
o.resident_headroom = (uint64_t) (std::atof(v) * 1073741824.0);
}
else if (a == "--stats") o.stats = true;
else if (a == "--shared-late") o.shared_late = true;
else if (a == "--keep-canonical") o.keep_canonical = true;
else if (a == "--no-token-graph") o.no_token_graph = true;
else if (a == "--no-fused-gr") o.no_fused_gr = true;
else if (a == "--no-fast-attn") o.no_fast_attn = true;
else if (a == "--no-publish-kernel") o.no_publish_kernel = true;
else if (a == "--no-fused-gdn") o.no_fused_gdn = true;
else if (a == "--no-fast-select") o.no_fast_select = true;
else {
// An unknown flag is an ERROR and not a warning: a typo'd `--max-neww` that silently generated 16
// tokens would look like a working run.
std::fprintf(stderr, "unknown argument: %s\n", a.c_str());
usage();
return 2;
}
}
}
strata::core::set_coupled_draft(o.coupled_draft);
strata::core::set_peer_portable(o.peer_device >= 1); // multi-GPU: the Portable flag on mapped host buffers only with a peer device (before any allocation)
if (o.serve && o.conversation_cache_mib > 0 && (o.prompt_cache == 0 || o.conversation_cache_slots == 0))
std::fprintf(stderr, "strata serve: warning: conversation caching is disabled by %s\n",
o.prompt_cache == 0 ? "--prompt-cache 0" : "--conversation-cache-slots 0");
if (o.conversation_cache_mib > 0 && o.conversation_cache_slots > 0 && o.prompt_cache > 0 && !o.layer_split.empty()) {
std::fprintf(stderr, "strata serve: conversation parking does not yet support --layer-split; disable parking with --conversation-cache-mib 0\n");
return 2;
}
// Layer split (multi-GPU): the later stages run layers [K_i, K_i+1) on their own GPUs (--split-device, default
// the next visible ones); "auto" places the K from each GPU's free VRAM once the weights are in (below). Across
// GPUs, not yet: KV streaming, images, control vectors, the helper caches (--expert-cache-remote), and lending
// cache slots to the prompt path (each stage's prompt path has its own buffers).
// --pcie-frac given: that one share is every stage's (the stages' own link probes are skipped), so a split over
// a fast and a slow link cannot set the two apart from the command line (#485)
const bool pcie_given = o.pcie_frac >= 0.0;
std::vector<int64_t> split_at;
std::vector<int> split_devs;
bool split_auto = false, split_same = false;
if (!o.layer_split.empty()) {
int n_dev = 1;
if (cudaGetDeviceCount(&n_dev) != cudaSuccess || n_dev < 1) n_dev = 1;
cudaGetLastError();
auto ints = [](const std::string& str, auto& out) -> bool {
using V = typename std::decay_t<decltype(out)>::value_type;
size_t a = 0;
while (a < str.size()) {
size_t b = str.find(',', a);
if (b == std::string::npos) b = str.size();
const std::string t = str.substr(a, b - a);
if (t.empty() || t.find_first_not_of("0123456789") != std::string::npos) return false;
out.push_back((V) std::atoll(t.c_str()));
a = b + 1;
}
return !out.empty();
};
split_auto = o.layer_split == "auto";
bool ok = o.serve && (split_auto || ints(o.layer_split, split_at));
if (ok && !o.split_device.empty()) ok = ints(o.split_device, split_devs);
else if (ok)
for (int d = 1; d < n_dev && (split_auto || split_devs.size() < split_at.size()); ++d) split_devs.push_back(d);
if (ok && !split_auto && split_devs.empty() && split_at.size() == 1) split_devs.push_back(0); // one GPU
split_same = ok && split_devs.size() == 1 && split_devs[0] == 0 && !split_auto;
if (ok && split_auto && split_devs.empty()) {
std::fprintf(stderr, "strata generate: --layer-split auto: one GPU visible, so no split\n");
o.layer_split.clear();
split_auto = false;
} else if (ok) {
ok = (split_auto || split_at.size() == split_devs.size()) && split_devs.size() < (size_t) SplitDrive::kMax;
for (size_t i = 0; ok && i < split_at.size(); ++i) ok = split_at[i] >= 2 && (i == 0 || split_at[i] > split_at[i - 1]);
for (size_t i = 0; ok && !split_same && i < split_devs.size(); ++i) {
ok = split_devs[i] > 0 && split_devs[i] < n_dev;
for (size_t j = 0; ok && j < i; ++j) ok = split_devs[i] != split_devs[j];
}
}
if (!ok) {
std::fprintf(stderr, "strata generate: --layer-split K[,K2..]|auto needs --serve, rising K from 2, and one "
"distinct GPU per K in --split-device (1..%d; or 0 with one K: the same GPU)\n", n_dev - 1);
return 2;
}
}
bool multi_gpu = !split_devs.empty() && !split_same; // cleared by --split-skip-if-fits before any stage loads
bool split_own_auto = false; // #340: the split keeps own prompt buffers by its rule (not --no-prefill-borrow)
if (o.mmap_experts && !o.shared_expert_arena.empty()) {
std::fprintf(stderr, "strata generate: --shared-expert-arena backs the resident arena and cannot be used with --mmap-experts\n");
return 2;
}
if (o.resident_cpu_experts && (!o.mmap_experts || o.expert_profile.empty())) {
std::fprintf(stderr, "strata generate: --resident-cpu-experts requires --mmap-experts and a static --expert-profile\n");
return 2;
}
const bool remote_caches = o.expert_cache_remote[0] > 0 || o.expert_cache_remote[1] > 0 ||
o.expert_cache_remote[2] > 0;
if (o.resident_cpu_experts && !o.layer_split.empty() && !remote_caches && o.resident_soft &&
!o.resident_cpu_explicit && o.resident_budget == 0) {
// #364 #384: setup's --resident-experts with a layer split (--gpus at start, or a config edited by hand) runs
// as the plain mmap mode - the placement those users measured 1.3-1.6x faster than one GPU - instead of
// refusing. Exactly --mmap-experts: nothing else reads these flags (the headroom only sizes the copy).
std::fprintf(stderr, "strata generate: WARNING: the resident RAM mode (--resident-experts) does not support a "
"layer split yet: the experts the GPUs do not hold are read through the OS file cache "
"(--mmap-experts), and RAM may fill up during long prompts\n");
o.resident_cpu_experts = o.resident_pin = o.resident_soft = false;
o.resident_headroom = 8ull << 30;
}
if (o.resident_cpu_experts && (!o.layer_split.empty() || remote_caches)) {
std::fprintf(stderr, "strata generate: --resident-cpu-experts does not support layer splits or remote expert caches\n");
return 2;
}
// the helper-GPU expert caches (--expert-cache-remote, docs/SECOND_GPU.md): CUDA1..3 on one GPU; with a layer
// split, the visible GPUs no stage runs on, in order
int remote_dev[3] = {1, 2, 3};
if (multi_gpu) {
if (o.expert_profile.empty()) {
std::fprintf(stderr, "strata generate: a layer split across GPUs needs --expert-profile\n");
return 2;
}
// A STAGE'S PROMPT PATH BORROWS FROM THAT STAGE'S OWN EXPERT CACHE. This used to set
// `no_prefill_borrow = true` - "each stage's prompt path has its own buffers" - which is true, but it is
// a reason to give each stage its own LOAN, not a reason to make every stage withhold a chunk-sized
// reserve from its cache for the whole session. Forced on, it also collapsed `--prefill auto` to 2048
// (below) and skipped the lend arm, so a stage paid for its prompt buffers twice over: once in VRAM it
// never got back, once in the smaller chunk. At `--prefill 5524` that reserve is 3.8 GiB per stage,
// more than either 8 GB card had - which is how adding two GPUs to the two 12 GB ones lost 260K.
// The loan is the tail of the stage's own cache (see `PfPart` in the serve block); outside the prompt
// that tail is expert cache, so a large chunk costs a stage nothing permanent.
// #340 - BUT A LOAN IS NOT FREE PER REQUEST: every stage streams the lent experts during the prompt and
// copies them back after it, so on cards that hold (nearly) every expert of their layers a 2K prompt read
// 37% slower than 0.1.29's own buffers (2x RX 9070 XT / R9700: 1570 -> 990 tok/s; 2x A5000 in #340: -35%).
// So a split keeps 0.1.29's own buffers (2048-token chunks, priced into the split search) when they are a
// small part of every card - at most 12% of its VRAM (16 GB and larger cards) - and borrows on smaller cards,
// where the reserve would cost the cache (and the context) the paragraph above is about. Measured with own
// buffers on the 9070 XT + R9700: 2K 1588 tok/s, 16K 2006 (0.1.29 1568 / 1935, 0.1.31 993 / 1852).
// OPT-IN (the owner's choice for 0.1.32): own buffers change which experts a full card keeps resident, so the
// split's output differs from 0.1.31's; the default borrows as 0.1.31 did - with the concurrent refill and the
// 96-blob split ring that alone gave 2K +27%, 16K +5% over 0.1.31, decode unchanged, output identical.
// STRATA_SPLIT_OWN=auto: the 12% rule above; 1: own buffers on any cards; unset/0: borrow.
// --split-skip-if-fits keeps the loans (it can fall back to one GPU, which must stay as it is).
const char* own_env = std::getenv("STRATA_SPLIT_OWN");
if (!o.no_prefill_borrow && !o.split_skip_if_fits && own_env != nullptr && own_env[0] != '0') {
const char* v = std::string(own_env) == "auto" ? nullptr : own_env;
bool own = v ? v[0] == '1' : true;
// the buffers of a 2048-token chunk (or the --prefill one) + a 96-blob ring (as split_pf_mib prices them)
const int64_t own_chunk = o.prefill_auto || o.prefill_chunk <= 0 ? 2048 : o.prefill_chunk;
const int64_t reserve_mib = 160 + (own_chunk * 680) / 1024 + 96 * 4;
std::string why;
if (!v) {
std::vector<int> devs = {0};
for (const int d : split_devs) devs.push_back(d);
for (const int d : devs) {
cudaDeviceProp prop{};
if (cudaGetDeviceProperties(&prop, d) != cudaSuccess) { cudaGetLastError(); own = false; break; }
const int64_t total_mib = (int64_t) (prop.totalGlobalMem >> 20);
if (reserve_mib * 100 > 12 * total_mib) {
own = false;
why = "CUDA" + std::to_string(d) + " has " + std::to_string(total_mib) + " MiB";
break;
}
}
}
if (own) {
o.no_prefill_borrow = true;
split_own_auto = true;
std::fprintf(stderr, "strata generate: layer split: every stage keeps its own prompt buffers (~%lld MiB "
"each, %lld-token chunks)%s\n", (long long) reserve_mib, (long long) own_chunk,
v ? " (STRATA_SPLIT_OWN=1)" : "");
} else if (!why.empty()) {
std::fprintf(stderr, "strata generate: layer split: the prompt paths borrow from the caches (%s)\n",
why.c_str());
}
}
int n_vis = 1;
if (cudaGetDeviceCount(&n_vis) != cudaSuccess || n_vis < 1) n_vis = 1;
cudaGetLastError();
int next_free = 1;
for (int r = 0; r < 3; ++r) {
if (o.expert_cache_remote[(size_t) r] <= 0) continue;
while (next_free < n_vis &&
std::find(split_devs.begin(), split_devs.end(), next_free) != split_devs.end()) ++next_free;
if (next_free >= n_vis) {
std::fprintf(stderr, "strata generate: --expert-cache-remote with a layer split needs a GPU that runs no "
"stage (%d visible, %zu used by the split)\n", n_vis, split_devs.size() + 1);
return 2;
}
remote_dev[r] = next_free++;
}
std::string devs;
for (const int d : split_devs) devs += (devs.empty() ? "" : ",") + std::to_string(d);
std::fprintf(stderr, "strata generate: layer split across %zu GPUs: CUDA0, then CUDA%s (split %s)\n",
split_devs.size() + 1, devs.c_str(), o.layer_split.c_str());
}
#if defined(STRATA_USE_HIP)
{
// every GPU this run uses must be an architecture the binary has code for (a gfx1100 build on a gfx1201
// card would otherwise fail later with "invalid device function")
std::vector<int> used{0};
if (multi_gpu) used.insert(used.end(), split_devs.begin(), split_devs.end());
for (int r = 0; r < 3; ++r)
if (o.expert_cache_remote[(size_t) r] > 0) used.push_back(remote_dev[r]);
for (const int d : used) {
if (const std::string why = strata::core::gpu_arch_problem(d); !why.empty()) {
std::fprintf(stderr, "strata generate: %s\n", why.c_str());
return 1;
}
}
}
#endif
if (o.prefill_auto && (o.no_prefill_borrow || o.expert_profile.empty())) {
o.prefill_auto = false; // nothing to lend from: the buffers are reserved for the session, so keep them small
o.prefill_chunk = 2048;
}
if (!have_tokens && o.serve) { // plan v0.3 P8: requests bring their own tokens
o.tokens = {248045};
o.max_new = 1;
have_tokens = true;
o.stop_eos = true;
}
if (!have_tokens) {
std::fprintf(stderr, "strata generate: --tokens is required (this build has no tokenizer; see the "
"header of src/program/generate.cpp)\n");
usage();
return 2;
}
if ((o.ple_io != "direct" && o.ple_io != "mmap" && o.ple_io != "ram") || o.ple_row_cache < 0 || o.ple_inflight < 1 ||
o.ple_inflight > 1024 || !(o.ple_delay_us >= 0)) {
std::fprintf(stderr, "strata generate: invalid --ple-io/--ple-row-cache/--ple-inflight/--ple-delay-us\n");
return 2;
}
#if defined(_WIN32)
if (o.ple_io == "ram") {
std::fprintf(stderr, "strata generate: --ple-io ram is not available on Windows (no mlock); use --ple-io mmap\n");
return 2;
}
#endif
if (o.kv == "q4") o.kv = "q4_0";
if (o.kv != "fp16" && o.kv != "int8" && o.kv != "q4_0" && o.kv != "k8v4") {
std::fprintf(stderr, "strata generate: --kv must be fp16, int8, q4_0 or k8v4\n");
return 2;
}
strata::core::qsa_set_kv_int8(o.kv == "int8");
strata::core::qsa_set_kv_q4(o.kv == "q4_0"); // PR #21: 4-bit codes after a Hadamard rotation (kv_q4.hpp)
// STRATA_KV_ROT=1: INT8 K/V through the Hadamard rotation --kv q4_0 already uses. Opt-in: first-token KL to
// fp16 K/V improved on an NVFP4 pack (0.0066 -> 0.0051) but not on IQ2_XS (0.0022 -> 0.0054)
const char* kv_rot = std::getenv("STRATA_KV_ROT");
strata::core::qsa_set_kv_int8_rotate(kv_rot != nullptr && kv_rot[0] == '1');
if (kv_rot != nullptr && kv_rot[0] == '1' && o.kv == "int8")
std::fprintf(stderr, "strata generate: STRATA_KV_ROT=1: INT8 K/V through the Hadamard rotation (opt-in)\n");
strata::core::qsa_set_kv_hybrid(o.kv == "k8v4"); // K8V4: INT8 K + rotated Q4_0 V, 816 B/cell
if (o.kv_resident < 0) {
std::fprintf(stderr, "strata generate: --kv-resident must be >= 0\n");
return 2;
}
if (o.kv == "k8v4" && o.kv_resident > 0) {
std::fprintf(stderr, "strata generate: --kv k8v4 does not support --kv-resident streaming (yet)\n");
return 2;
}
strata::core::qsa_set_kv_resident(o.kv_resident);
// Prompt lookup (the suffix drafter, on by default): the MTP keeps its --spec windows and a lookup window may be
// up to 2 tokens longer; the draft policy (strata/spec/draft_policy.hpp) takes one only where it pays. Code
// edits +6-11%, ordinary text unchanged (bench/results/2026-09-27-spec). --suffix-draft 0 turns it off.
if (o.suffix_draft > 0 && o.spec >= 2 && o.mtp_max_t == 0) {
o.mtp_max_t = o.spec;
o.spec = std::min(o.spec + 2, 8); // kVerifyMaxT
}
strata::core::layer_set_shared_early(!o.shared_late);
if (!o.native_preset.empty()) {
try {
// every shard of the model (<name>-0000N-of-0000M.gguf beside --native). A missing shard is an error
// here: it used to be skipped, leaving a model with some tensors absent and a later error, or none.
o.native_shards = strata::gguf_split_paths(o.native_preset);
// --ple-gguf defaults to the shard that holds the PLE table, found by name: shard 2 of the ISTA files
// and of Unsloth's UD-Q4_K_XL, shard 1 of Swift's
if (o.ple_gguf.empty() && !o.no_ple) {
const strata::GgufModel model(o.native_shards);
size_t at = 0;
if (model.find("per_layer_token_embd.weight", &at) != nullptr) o.ple_gguf = o.native_shards[at];
}
} catch (const std::exception& e) {
std::fprintf(stderr, "strata generate: --native %s: %s\n", o.native_preset.c_str(), e.what());
return 2;
}
if (o.no_ple || o.ple_gguf.empty()) {
std::fprintf(stderr, "strata generate: --native requires --ple-gguf (the PLE key is native too), and no "
"shard of the model holds per_layer_token_embd.weight\n");
return 2;
}
o.stream_token = true;
o.gr_native_mmvf = true;
o.native_bf16 = o.native_bf16_extra = true;
o.native_ple_key = o.native_moe_combine = o.native_gdn = o.native_router = true;
o.native_qsa = o.native_qsa_indexer = o.native_rope = o.native_ple_postops = true;
if (o.native_head_gguf.empty()) o.native_head_gguf = o.native_preset;
if (o.native_dense_gguf.empty()) {
// every shard of the model (<name>-0000N-of-0000M.gguf beside --native), then the PLE shard: a split
// may put any layer in any shard (Swift's GGUFs: layers 13-47 in shard 2, the PLE table in shard 1)
o.native_dense_gguf = o.native_shards;
// a PLE-only table (tools/ple_fp8_pack.py: architecture strata-ple) holds no projections
bool ple_only = false;
try {
strata::GgufFile pg(o.ple_gguf);
if (const strata::MetaValue* v = pg.get("general.architecture")) ple_only = v->s == "strata-ple";
} catch (const std::exception&) {}
if (!ple_only &&
std::find(o.native_dense_gguf.begin(), o.native_dense_gguf.end(), o.ple_gguf) == o.native_dense_gguf.end())
o.native_dense_gguf.push_back(o.ple_gguf);
}
// Plan v0.3 (24 Sep): the CPU experts stay on the VNNI kernel. The llama.cpp-CPU-exact q8_0 contract
// cost 27.0 vs 17.2 ms/token of pool time and G-C does not need it; `--cpu-oracle-q8-0` still selects it.
}
if (o.logits_stride > 1 && (o.max_new != 1 || o.dump_logits.empty())) {
std::fprintf(stderr, "strata generate: --logits-stride > 1 requires --max-new 1 and --dump-logits\n");
return 2;
}
if (o.no_ple && !o.ple_gguf.empty()) {
std::fprintf(stderr, "strata generate: --no-ple and --ple-gguf are mutually exclusive\n");
return 2;
}
if (o.native_ple_postops && o.no_ple) {
std::fprintf(stderr, "strata generate: --native-ple-postops requires PLE enabled\n");
return 2;
}
if (!o.no_ple && o.ple_gguf.empty()) {
std::fprintf(stderr, "strata generate: --ple-gguf is required; --no-ple explicitly enables a diagnostic ablation\n");
return 2;
}
// P7 audit: positions, cells and pooled-block indices are cast to int32 on the device path.
if (o.max_context > 2147483647LL - 8) {
std::fprintf(stderr, "strata generate: --max-context must be below 2^31\n");
return 2;
}
if (o.max_new <= 0 || o.max_context <= 0 || o.max_new > o.max_context ||
o.tokens.size() > (size_t) (o.max_context - o.max_new)) {
std::fprintf(stderr, "strata generate: positive --max-new and --max-context must fit the prompt and generation\n");
return 2;
}
// THE ROPE KNOBS (rope_scaling.hpp). Anything invalid dies here, at second zero, rather than becoming a
// NaN angle inside one of the twelve QSA layers. Only the RANGES are checked - the config itself is
// resolved after the model file has had its say, right before session_init.
strata::kernels::RopeScaling rope_cfg; // type filled here; the rest at the resolution below
{
using RST = strata::kernels::RopeScalingType;
// an absent --rope-scaling (the empty default) leaves the type to the model file's rope keys,
// resolved below; anything present must be one of the three names
if (o.rope_scaling == "none") rope_cfg.type = RST::None;
else if (o.rope_scaling == "linear") rope_cfg.type = RST::Linear;
else if (o.rope_scaling == "yarn") rope_cfg.type = RST::YaRN;
else if (!o.rope_scaling.empty()) {
std::fprintf(stderr, "strata generate: --rope-scaling must be none, linear or yarn (got '%s')\n",
o.rope_scaling.c_str());
return 2;
}
// every knob FINITE first: `atof("nan")` is NaN, and a NaN passes every range comparison below
for (const double v : {o.rope_scale, o.rope_freq_base, o.rope_freq_scale, o.yarn_orig_ctx, o.yarn_ext_factor,
o.yarn_attn_factor, o.yarn_beta_fast, o.yarn_beta_slow})
if (!std::isfinite(v)) {
std::fprintf(stderr, "strata generate: a rope scaling knob is not a finite number (%g)\n", v);
return 2;
}
// 0 is the absent default; an explicit factor must extend, not shrink
if (o.rope_scale != 0 && o.rope_scale < 1.0) {
std::fprintf(stderr, "strata generate: --rope-scale %g must be >= 1 (it extends the context, not shrinks it)\n",
o.rope_scale);
return 2;
}
if (o.rope_freq_base != 0 && o.rope_freq_base <= 1.0) {
std::fprintf(stderr, "strata generate: --rope-freq-base must be a base above 1 (0 = the model's)\n");
return 2;
}
if (o.rope_freq_scale < 0 || o.yarn_orig_ctx < 0 || o.yarn_ext_factor < -1.0 || o.yarn_attn_factor <= 0 ||
o.yarn_beta_fast <= 0 || o.yarn_beta_slow <= 0) {
std::fprintf(stderr, "strata generate: invalid rope scaling knob (see usage: --yarn-ext-factor <0 = auto, "
"--yarn-orig-ctx 0 = default, the rest positive)\n");
return 2;
}
}
if (!std::isfinite(o.temperature) || o.temperature < 0 || !std::isfinite(o.top_p) ||
o.top_p <= 0 || o.top_p > 1 || o.top_k < 0 || o.expert_cache < -1 ||
o.pool_workers < 0 || std::any_of(o.expert_cache_remote.begin(), o.expert_cache_remote.end(),
[](int slots) { return slots < 0; }) ||
(o.expert_cache_remote[1] > 0 && o.expert_cache_remote[0] == 0) ||
(o.expert_cache_remote[2] > 0 && o.expert_cache_remote[1] == 0)) {
std::fprintf(stderr, "strata generate: invalid sampling or resource parameter\n");
return 2;
}
if (o.expert_cache_remote_placement != "stripe" && o.expert_cache_remote_placement != "layer") {
std::fprintf(stderr, "strata generate: --expert-cache-remote-placement must be stripe or layer\n");
return 2;
}
// the peer tier is the second card's only user: a layer split or a remote expert cache would put a second engine
// part (and a second copy of the same experts) on it
if (o.peer_device >= 1 && (!o.layer_split.empty() || o.expert_cache_remote[0] > 0)) {
std::fprintf(stderr, "strata generate: --peer-device cannot be combined with %s\n",
!o.layer_split.empty() ? "--layer-split (use one or the other)"
: "--expert-cache-device1..3 (the peer tier already caches experts there)");
return 2;
}
if (o.native_flash_attn_short && o.max_context > 256) {
std::fprintf(stderr, "strata generate: --native-flash-attn-short requires --max-context <=256\n");
return 2;
}
if (o.native_flash_attn_short && (o.gpu_only_full || o.graph_only || o.gpu_stages)) {
std::fprintf(stderr, "strata generate: --native-flash-attn-short requires the normal decode loop for status validation\n");
return 2;
}
// The whole-model graph measurements replay every layer through CUDA0's session, which a layer split carves to
// CUDA0's own range - the answer would read another stage's state. Refused here rather than at the call, so
// the reason is visible before 55 GB is loaded.
if (multi_gpu && (o.gpu_only_full || o.gpu_stages)) {
std::fprintf(stderr, "strata generate: --gpu-only-full and --gpu-stages replay the whole model through one "
"session, which a layer split does not have; run them without --layer-split\n");
return 2;
}
if (!o.native_head_gguf.empty()) {
try {
o.native_head_shards = o.native_head_gguf == o.native_preset ? o.native_shards
: strata::gguf_split_paths(o.native_head_gguf);
} catch (const std::exception& e) {
std::fprintf(stderr, "strata generate: --native-head-gguf %s: %s\n", o.native_head_gguf.c_str(), e.what());
return 2;
}
}
if (o.native_ple_key && (o.native_dense_gguf.empty() || o.no_ple)) {
std::fprintf(stderr, "strata generate: --native-ple-key requires PLE and --native-dense-gguf\n");
return 2;
}
if (o.cpu_oracle_q8_0 && (o.expert_cache != 0 || !o.expert_profile.empty())) {
std::fprintf(stderr, "strata generate: --cpu-oracle-q8-0 cannot be combined with --expert-cache or --expert-profile until the GPU expert contract matches\n");
return 2;
}
// **BEFORE ANYTHING ELSE.** The CPU expert kernel is AVX-512 (VNNI + VBMI) and its translation unit is
// compiled `/arch:AVX512`, so on a CPU without those features it does not fail - it executes an illegal
// instruction at some unpredictable token. Refusing at second zero is the whole point of P2.S3's check.
strata::kernels::cpu::expert_set_oracle_q8_0(o.cpu_oracle_q8_0);
// ... and nothing runs on a CPU without AVX2: every CPU expert kernel is AVX2 at least (the AVX-512 ones are
// chosen above it), and so is ggml-cpu in the release build, which the native pack's layout load initializes
// next. Refused here, by name, rather than an illegal instruction in the first expert.
if (!strata::kernels::cpu::cpu_avx2_ok()) {
std::fprintf(stderr, "strata generate: this CPU (%s) does not support AVX2 with FMA and F16C, which every CPU "
"expert kernel needs; Strata runs on Intel Haswell (2013), AMD Zen (2017) or newer\n",
strata::kernels::cpu::cpu_name().c_str());
return 2;
}
std::string err;
if (!o.native_head_gguf.empty() && !o.stream_token) {
std::fprintf(stderr, "--native-head-gguf requires --stream-token\n");
return 2;
}
// Plan v0.3 P6: where the experts live. A native pack (tools/iq_pack.py: the IQ2_XS / IQ3_XXS files) keeps
// every quantized tensor in its GGUF form, so it needs --native (the dense projections, head and embedding
// come from the model file) and runs its experts in verify windows only (--spec).
{
const strata::core::ModelGeometry g0;
if (!strata::kernels::cpu::expert_layout_load(o.pack, g0.n_layers, g0.n_expert, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
// Every layer's formats must have GPU expert kernels and a prompt-path dequantizer, checked here, before
// anything is allocated: an unsupported down type used to exit from inside the first verify window, and
// an unsupported dequant type left the prompt path's fp16 buffer unwritten.
const auto& lay = strata::kernels::cpu::expert_layout();
for (int64_t l = 0; lay.native && l < (int64_t) lay.fmt.size(); ++l) {
const auto& f = lay.fmt[(size_t) l];
if (!strata::kernels::native_expert_supported(f.gu_type, f.d_type, f.n_embd, f.n_ff)) {
std::fprintf(stderr, "strata generate: layer %lld's experts are %s/%s (ggml types %d/%d), which this "
"engine has no GPU kernels for\n", (long long) l,
strata::ggml_type_name((uint32_t) f.gu_type), strata::ggml_type_name((uint32_t) f.d_type),
f.gu_type, f.d_type);
return 1;
}
}
}
const bool native_pack = strata::kernels::cpu::expert_layout().native;
// STRATA_EARLY_REMOTE_CONTEXTS=1: create EVERY secondary context here, like CUDA1's. Under WSL2 the driver's
// pinned/mapped host budget (dxg gpadl, ~1 GiB) is spent by CUDA0's weights and MTP before the later loop runs,
// and a new context then fails with cudaErrorMemoryAllocation (CUDA2: "cudaSetDevice(2) failed: out of memory").
const char* early_env = std::getenv("STRATA_EARLY_REMOTE_CONTEXTS");
const bool early_remote = early_env && early_env[0] == '1';
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0 && (r == 0 || early_remote)) {
// Keep CUDA1's proven startup order: initialise its context before
// allocating GPU0 weights or mapping the large host expert arena.
double free_gib = 0;
if (!strata::core::RemoteExperts::preflight(remote_dev[r], free_gib, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: CUDA%d context ready, %.2f GiB free before expert arena registration\n",
remote_dev[r], free_gib);
}
// plan v0.3 P6: the PCIe share of the missed experts, measured per kind of pack (the paper, finding on PCIe).
// PR #44: a x8 link carries half of what the native default assumes - the GPU's SMs read that share over the
// link (the copy kernel, since 0.1.14), so on a slower link it must shrink or the window waits for it. The
// real H2D bandwidth is probed once; from 20 GB/s up (x16 PCIe 4/5) the measured default stays. The canonical
// pack's 0.2 was never measured against the link, so it is left alone. `--calibrate` measures it outright.
if (o.pcie_frac < 0.0) {
const double base = native_pack ? 0.55 : 0.2;
std::string bursts;
const double bw = native_pack ? probe_pcie_h2d_gbps(&bursts) : -1.0;
if (!native_pack) {
o.pcie_frac = base;
} else if (bw > 0.0) {
o.pcie_frac = pcie_frac_for_gbps(bw, base);
std::fprintf(stderr, "strata generate: PCIe probe: %.1f GB/s host->device (best of %s) -> pcie_frac %.2f "
"(default %.2f)\n", bw, bursts.c_str(), o.pcie_frac, base);
} else {
o.pcie_frac = base;
std::fprintf(stderr, "strata generate: PCIe probe failed -> pcie_frac default %.2f\n", base);
}
}
// the canonical Q2_0 pack's CPU kernels are AVX-512 only; a native pack runs on AVX2 CPUs as well
if (!native_pack) strata::kernels::cpu::cpu_require_expert_support();
else if (!strata::kernels::cpu::cpu_avx512_ok())
std::fprintf(stderr, "strata generate: this CPU has no AVX-512: the expert kernels run on %s "
"(multi-token for the i-quant gate/up rows)\n",
std::getenv("STRATA_NO_IQ256") == nullptr ? "AVX-2" : "ggml-cpu vec_dot (STRATA_NO_IQ256 set)");
strata::core::ModelGeometry g; // canonical defaults; the model file overrides the MoE shape below
int64_t K = 10;
// THE ROPE CONFIG RESOLVES HERE, BEFORE ANY WEIGHT MOVES - the CLI and the model file have both spoken,
// and `session_init` below builds the rope table from it and captures the kernels reading its constants
// (rope_scaling.hpp); the only hard constraint is "set before that", and dying on a bad rope key beats
// scanning gigabytes of shards first. Precedence: an EXPLICIT flag over the model file's rope keys over
// the struct defaults. The empty --rope-scaling and the 0 --rope-scale mean the flag is absent, so the
// model file decides; an explicit value - `none` and `1` included, the opt-outs - wins over the model file.
{
// The model file's rope keys (llama.cpp's names under the arch prefix), when it carries any - the
// artifact today ships none, so this is a no-op defaults channel for future fine-tunes.
std::string gguf_rope_type;
double gguf_rope_base = 0, gguf_rope_factor = 0, gguf_rope_orig_ctx = 0;
if (!o.native_preset.empty()) {
// a pruned variant (GSQ-RCO Coder) ships fewer experts than the canonical 512x10; the model file
// is the authority on its own MoE shape - everything else in the geometry is unchanged
try {
strata::GgufFile model_gguf(o.native_shards.front()); // the metadata shard
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.expert_count")) g.n_expert = (int64_t) v->u;
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.expert_used_count")) K = (int64_t) v->u;
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.rope.freq_base")) gguf_rope_base = v->num();
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.rope.scaling.type")) gguf_rope_type = v->s;
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.rope.scaling.factor")) gguf_rope_factor = v->num();
if (const strata::MetaValue* v = model_gguf.get("qwen4exp.rope.scaling.original_context_length"))
gguf_rope_orig_ctx = v->num();
} catch (const std::exception& e) {
std::fprintf(stderr, "strata generate: reading the model's expert shape from %s: %s\n",
o.native_preset.c_str(), e.what());
return 1;
}
}
using RST = strata::kernels::RopeScalingType;
if (!o.rope_scaling.empty()) {
// the early validation pinned the spelling; `none` here is the CLI opting OUT of the model file's keys
if (o.rope_scaling == "linear") rope_cfg.type = RST::Linear;
else if (o.rope_scaling == "yarn") rope_cfg.type = RST::YaRN;
else rope_cfg.type = RST::None;
} else if (!gguf_rope_type.empty()) {
if (gguf_rope_type == "linear") rope_cfg.type = RST::Linear;
else if (gguf_rope_type == "yarn") rope_cfg.type = RST::YaRN;
else if (gguf_rope_type != "none") {
std::fprintf(stderr, "strata generate: %s carries rope.scaling.type '%s' - none, linear or yarn only\n",
o.native_preset.c_str(), gguf_rope_type.c_str());
return 2;
}
}
if (o.rope_scale > 0) rope_cfg.factor = o.rope_scale; // an explicit factor, 1 included
else if (gguf_rope_factor > 1.0) rope_cfg.factor = gguf_rope_factor;
if (o.rope_freq_base > 0) rope_cfg.freq_base = o.rope_freq_base;
else if (gguf_rope_base > 1.0) rope_cfg.freq_base = gguf_rope_base;
if (o.yarn_orig_ctx > 0) rope_cfg.orig_ctx = o.yarn_orig_ctx;
else if (gguf_rope_orig_ctx >= 1) rope_cfg.orig_ctx = gguf_rope_orig_ctx;
rope_cfg.freq_scale_in = o.rope_freq_scale;
rope_cfg.ext_factor = o.yarn_ext_factor >= 0 ? o.yarn_ext_factor
: (rope_cfg.type == RST::YaRN ? 1.0 : 0.0);
rope_cfg.attn_factor = o.yarn_attn_factor;
rope_cfg.beta_fast = o.yarn_beta_fast;
rope_cfg.beta_slow = o.yarn_beta_slow;
if (rope_cfg.type == RST::None) {
// none is the trained rotation, exactly: the scaling knobs are inert (the table builder and
// `kernel_args` ignore them), and resetting them keeps the logged/queried config honest. Only the
// frequency base survives - it is the rotation itself, not a scaling knob.
const bool knobs = o.rope_scale > 1.0 || o.rope_freq_scale > 0 || o.yarn_ext_factor > 0 ||
o.yarn_attn_factor != 1.0;
const double base = rope_cfg.freq_base;
rope_cfg = strata::kernels::RopeScaling{};
rope_cfg.freq_base = base;
if (knobs)
std::fprintf(stderr, "strata generate: note: no rope scaling is active (none), so --rope-scale, "
"--rope-freq-scale and the --yarn-* knobs have no effect\n");
}
// THE RESOLVED CONFIG IS VALIDATED AS A WHOLE, with the one rule every rotation site also applies
// (rope_scaling.hpp): the CLI ranges above cannot see a model-file value, nor a combination such as a
// --rope-freq-scale that turns the resolved factor non-finite.
if (const char* why = strata::kernels::rope_scaling_invalid(rope_cfg)) {
std::fprintf(stderr, "strata generate: invalid rope scaling configuration: %s (type %s, factor %g, "
"freq_scale %g, base %g, original context %g)\n",
why, rope_cfg.type == RST::YaRN ? "yarn" : rope_cfg.type == RST::Linear ? "linear" : "none",
rope_cfg.factor, rope_cfg.freq_scale(), rope_cfg.freq_base, rope_cfg.orig_ctx);
return 2;
}
strata::kernels::rope_scaling_set(rope_cfg);
if (rope_cfg.type != RST::None) {
const char* tn = rope_cfg.type == RST::YaRN ? "yarn" : "linear";
std::fprintf(stderr,
"strata generate: rope scaling %s, factor %.6g (freq_scale %.6g, base %.6g, mscale %.6f), "
"--max-context %lld against a trained context of %.0f\n",
tn, rope_cfg.factor, rope_cfg.freq_scale(), rope_cfg.freq_base, rope_cfg.mscale(),
(long long) o.max_context, rope_cfg.orig_ctx);
if ((double) o.max_context <= rope_cfg.orig_ctx)
std::fprintf(stderr,
"strata generate: note: the context is within the trained %.0f - no position needs the "
"extension, and the resolved scaling still applies to every angle\n",
rope_cfg.orig_ctx);
// only when there IS a magnitude correction: YaRN's log term (ext_factor != 0) or an explicit
// --yarn-attn-factor; plain linear (mscale 1) has none, and saying otherwise was TODO 22
if (rope_cfg.mscale() != 1.0)
std::fprintf(stderr,
"strata generate: note: the %s magnitude correction scales cos and sin by %.6f "
"at every position\n",
rope_cfg.type == RST::YaRN ? "YaRN" : "--yarn-attn-factor", rope_cfg.mscale());
}
}
strata::core::NativeEmbed native_embed;
if (native_pack) {
if (o.native_preset.empty() || o.spec < 2 || o.keep_canonical ||
(o.prefill_chunk <= 0 && o.tokens.size() > 1)) {
std::fprintf(stderr, "strata generate: %s is a native (IQ) pack: it needs --native SHARD1, --spec T (T >= 2) "
"and --prefill CHUNK\n", o.pack.c_str());
return 2;
}
const strata::core::ModelGeometry g0;
if (!native_embed.load(o.embd_gguf.empty() ? o.native_shards : std::vector<std::string>{o.embd_gguf}, g0.n_embd,
248320, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
strata::core::set_native_embed(&native_embed);
std::fprintf(stderr, "strata generate: native pack: %s experts (largest blob %.2f MB), token embedding "
"%s in mapped host memory (%.0f MiB)\n",
o.pack.c_str(), (double) strata::kernels::cpu::expert_layout().max_blob / 1e6,
strata::ggml_type_name((uint32_t) native_embed.type()), (double) native_embed.bytes() / 1048576.0);
}
// Plan v0.3 P1: tensors served in native form are not also loaded in canonical form (~2.7 GB of VRAM back
// to the expert cache with --native). `--keep-canonical` loads both, as before.
std::set<std::string> skip;
if (!o.keep_canonical) {
if (!o.native_dense_gguf.empty() &&
!strata::core::NativeDense::served_names(o.native_dense_gguf, o.native_ple_key, skip, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
if (!o.native_head_gguf.empty()) skip.insert("output.weight");
// the PLE module validates its canonical key at construction (8 MB); a native pack has none to load
if (!native_pack) skip.erase("blk.1.ple_key.weight");
// #326: a --compat-bf16 pack (OrcaRouter IQ3_XXS) keeps its BF16 key in the arena
if (native_pack && !strata::core::NativeDense::keep_unquantized_ple_key(o.pack, skip, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
if (native_pack) skip.insert("token_embd.weight");
}
uint64_t pool_bytes = 0;
if (!strata::core::WeightTable::pool_bytes(o.pack, pool_bytes, err, skip.empty() ? nullptr : &skip)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
void* arena = nullptr;
if (const cudaError_t ce = cudaMalloc(&arena, pool_bytes); ce != cudaSuccess) {
// #486: the arena is the first large allocation and its size does not depend on the context, so what is
// missing is held by something else: say how much was free
cudaGetLastError();
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
std::fprintf(stderr, "strata generate: cudaMalloc(%llu) for the weight arena failed (%s): %llu MiB of %llu "
"MiB VRAM free on this GPU. The arena is allocated first, before the KV and expert "
"caches: another program (or an engine that is still exiting) holds the rest - "
"nvidia-smi / rocm-smi lists them\n",
(unsigned long long) pool_bytes, cudaGetErrorString(ce), (unsigned long long) (free_b >> 20),
(unsigned long long) (total_b >> 20));
return 1;
}
strata::core::WeightTable wt;
if (!wt.load(o.pack, arena, pool_bytes, err, skip.empty() ? nullptr : &skip)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: %llu MiB of weights loaded from %s (%zu canonical tensors skipped: "
"served natively)\n",
(unsigned long long) (pool_bytes >> 20), o.pack.c_str(), skip.size());
strata::core::NativeDense native_dense;
if (!o.native_dense_gguf.empty()) {
if (!native_dense.load(o.native_dense_gguf, wt, err, o.native_ple_key)) {
std::fprintf(stderr, "strata generate: native dense projections: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: %zu native projection matrices, %.2f MiB of weights\n",
native_dense.tensor_count(), (double) native_dense.weight_bytes() / (1024.0 * 1024.0));
}
strata::kernels::gr_set_fp32_activations(o.gr_fp32_activations);
strata::kernels::gr_set_native_mmvf(o.gr_native_mmvf);
// Plan v0.3 P3: the fused hyper-connection read rides the native (FP32-activation) contract; the per-stage
// and dump measurements need the unfused layout of R, so they keep the old kernels.
strata::core::layer_set_fast_attn(!o.no_fast_attn);
strata::core::layer_set_publish_kernel(!o.no_publish_kernel);
strata::core::layer_set_fused_gdn(!o.no_fused_gdn);
strata::core::layer_set_fast_select(!o.no_fast_select);
strata::core::layer_set_fused_gr(o.gr_native_mmvf && !o.no_fused_gr && !o.gpu_stages && o.dump_layers.empty() &&
o.dump_halves.empty() && !o.stage_timing);
strata::core::layer_set_native_bf16(o.native_bf16);
strata::core::layer_set_native_flash_attn_short(o.native_flash_attn_short);
strata::kernels::ple_set_native_bf16(o.native_bf16_extra);
strata::kernels::shared_expert_set_native_bf16(o.native_bf16_extra);
strata::kernels::native_moe_combine_set_enabled(o.native_moe_combine);
strata::kernels::native_gdn_set_enabled(o.native_gdn);
strata::kernels::native_router_set_enabled(o.native_router);
strata::kernels::native_qsa_set_enabled(o.native_qsa);
strata::kernels::native_qsa_indexer_set_enabled(o.native_qsa_indexer);
strata::kernels::native_rope_set_enabled(o.native_rope);
// The vision path: every rope kernel reads a cell's (t, h, w) from this table (strata/kernels/mrope.hpp). It is
// the identity until an image request, and it is set here, before any CUDA graph captures a rope kernel.
int32_t* d_mrope = nullptr;
std::vector<int32_t> mrope_host;
if (o.vision) {
const int64_t cells = o.max_context + 64;
mrope_host.resize((size_t) cells * 3);
for (int64_t c = 0; c < cells; ++c)
mrope_host[(size_t) c * 3] = mrope_host[(size_t) c * 3 + 1] = mrope_host[(size_t) c * 3 + 2] = (int32_t) c;
if (cudaMalloc(&d_mrope, mrope_host.size() * sizeof(int32_t)) != cudaSuccess ||
cudaMemcpy(d_mrope, mrope_host.data(), mrope_host.size() * sizeof(int32_t), cudaMemcpyHostToDevice) !=
cudaSuccess) {
std::fprintf(stderr, "strata generate: cannot allocate the image position table\n");
return 1;
}
strata::kernels::mrope_table_set(d_mrope);
}
strata::kernels::ple_set_native_postops(o.native_ple_postops);
// before session_init: every graph captured from here on has the vector's kernels where it applies
std::string cvec_summary = "0";
if (!o.cvec_files.empty()) {
std::string ce;
if (!load_control_vectors(o, g, cvec_summary, ce)) {
std::fprintf(stderr, "strata generate: control vector: %s\n", ce.c_str());
return 2;
}
}
if (o.max_context < (int64_t) o.tokens.size() + o.max_new) {
std::fprintf(stderr, "strata generate: --max-context %lld cannot hold %zu prompt + %lld new tokens\n",
(long long) o.max_context, o.tokens.size(), (long long) o.max_new);
return 2;
}
strata::core::SessionState ss;
void* sbuf = nullptr; // allocated after the layer-split search, sized to CUDA0's own layer range (the carve)
// **THE ENGINE RAN ON THE LEGACY DEFAULT STREAM, WHICH ON WDDM IS THE SLOW PATH.** All four session
// calls - `session_capture`, `session_replay`, `session_token` and `session_loop` - were handed `nullptr`,
// i.e. stream 0. `bench/micro/kernel_costs.cu` measures what that costs: EVERY kernel it launches through
// a wrapper comes back at 28-31 us REGARDLESS OF SIZE, `scale_inplace` on 2,048 floats and `silu_inplace`
// on 10,240 floats being indistinguishable, which is a fixed per-launch cost and not execution.
// `bench/micro/graph_node_cost.cu` measures the same kernels on a real stream at 3.63 us ungrapped and
// 0.805 us inside a graph. **That is an ~8x penalty on every launch in the engine.**
cudaStream_t main_stream = nullptr;
if (cudaStreamCreateWithFlags(&main_stream, cudaStreamNonBlocking) != cudaSuccess) {
std::fprintf(stderr, "strata generate: cannot create the main stream\n");
return 1;
}
void* const main_cs = (void*) main_stream;
// ---- **THE HALF-LEVEL DUMP HAS TO BE ARMED BEFORE `session_capture`, AND THE LADDER MUST NOT BE.** The
// half copies are issued from inside `block_layer_pre`/`block_layer_post`, so they are only ever enqueued
// while a graph is being CAPTURED - arming `ss.block.dump` afterwards would produce a file of zeros that
// reads exactly like a wrong answer. The ladder is the opposite: `session_loop` enqueues it per token on
// the replay stream, so it must be armed after capture to stay out of the graph.
const uint64_t half_stride = (uint64_t) 2 * g.n_embd + (uint64_t) 2 * g.hc +
(uint64_t) g.n_head * g.head_dim +
(uint64_t) 5 * g.n_head_kv * g.head_dim + 8;
std::FILE* half_dump = nullptr;
float* half_stage = nullptr;
if (!o.dump_halves.empty()) {
half_dump = std::fopen(o.dump_halves.c_str(), "wb");
if (half_dump == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_halves.c_str());
return 1;
}
const size_t n = (size_t) g.n_layers * (size_t) half_stride;
if (cudaHostAlloc((void**) &half_stage, n * sizeof(float), cudaHostAllocDefault) != cudaSuccess) {
std::fprintf(stderr, "strata generate: cannot pin the half-dump staging buffer\n");
return 1;
}
ss.block.dump = half_stage;
}
strata::core::Doorbell db;
if (strata::core::doorbell_init(g, K, db) == 0) {
std::fprintf(stderr, "strata generate: doorbell_init failed\n");
return 1;
}
ss.db = &db;
// ================================ THE PLE ================================
//
// **ITS ABSENCE IS WHY GATE C1 FAILED** (LEDGER L123): layer 1 carries six `blk.1.ple_*` tensors, the whole
// module was built and parity-tested, and nothing called it. Everything below is construction - the table
// is a mapping of the ORIGINAL second GGUF shard, the six weights are already loaded in the arena, and the
// three buffers are the only allocation.
strata::kernels::PleTable ple_table;
std::vector<float> ple_emb_host((size_t) strata::kernels::NG_N_EMBD);
float* ple_emb_dev = nullptr;
float* ple_scratch = nullptr;
if (!o.ple_gguf.empty()) {
strata::kernels::PleIoOptions pio;
pio.mode = o.ple_io == "mmap" || o.ple_io == "ram" ? strata::kernels::PleIo::Mmap : strata::kernels::PleIo::Direct;
pio.lock = o.ple_io == "ram";
const auto tpl = Clock::now();
pio.max_inflight = (uint32_t) o.ple_inflight;
pio.cache_rows = (uint64_t) o.ple_row_cache;
pio.io_thread = !o.ple_sync_submit;
// Keep the SSD awake while rows are being asked for (PleReader::set_keepalive): some SSDs stall the first
// reads 50-150 ms after ~250 ms without a command. STRATA_SSD_KEEPALIVE = ms without a read before one
// page is read anyway (default 100, 0 = off), STRATA_SSD_KEEPALIVE_WINDOW = seconds after the last row
// request that this goes on (default 60; then the SSD may sleep until the next request).
{
const char* ka = std::getenv("STRATA_SSD_KEEPALIVE");
const char* kw = std::getenv("STRATA_SSD_KEEPALIVE_WINDOW");
pio.keepalive_ms = ka != nullptr && *ka ? std::clamp(std::atof(ka), 0.0, 10000.0) : 100.0;
pio.keepalive_window_s = kw != nullptr && *kw ? std::clamp(std::atof(kw), 1.0, 86400.0) : 60.0;
if (pio.mode != strata::kernels::PleIo::Direct || !pio.io_thread) pio.keepalive_ms = 0;
}
if (!ple_table.open(o.ple_gguf, err, pio)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
if (pio.lock)
std::fprintf(stderr, "strata generate: PLE table %s (--ple-io ram) in %.1f s\n",
ple_table.locked() ? "locked in RAM" : "loaded (not locked)",
std::chrono::duration<double>(Clock::now() - tpl).count());
if (pio.keepalive_ms > 0)
std::fprintf(stderr, "strata generate: the SSD is kept awake while rows are read: one page of the table after "
"%.0f ms without a read, until %.0f s after the last request "
"(STRATA_SSD_KEEPALIVE=0 turns it off)\n", pio.keepalive_ms, pio.keepalive_window_s);
else if (pio.mode == strata::kernels::PleIo::Direct)
std::fprintf(stderr, "strata generate: SSD keep-alive off: the SSD may fall asleep between reads\n");
const strata::core::WeightRef* wk = wt.find("blk.1.ple_key.weight");
const strata::core::WeightRef* wv = wt.find("blk.1.ple_value.weight");
const strata::core::WeightRef* wnk = wt.find("blk.1.ple_norm_key.weight");
const strata::core::WeightRef* wnq = wt.find("blk.1.ple_norm_query.weight");
const strata::core::WeightRef* wnc = wt.find("blk.1.ple_norm_conv.weight");
const strata::core::WeightRef* wc = wt.find("blk.1.ple_conv1d.weight");
if (!wk || !wv || !wnk || !wnq || !wnc || !wc) {
std::fprintf(stderr, "strata generate: the pack has no blk.1.ple_* tensors, so the PLE cannot be "
"wired - and running without it is a DIFFERENT MODEL (LEDGER L123)\n");
return 1;
}
// `ple_key` is S2 and the loader has already widened its scales to f32, so the two planes are located
// by the sizes the `WeightRef` records rather than re-derived - the same rule `plane_ptrs` follows.
if (!wk->quantized()) {
// plan v0.3 P6: the IQ model files' BF16 key (the pack's extra.bin, raw BF16)
ss.ple.w.key_bf16 = (const uint16_t*) wk->data;
} else if (wk->data != nullptr) {
ss.ple.w.key_codes = (const uint8_t*) wk->data;
ss.ple.w.key_scales = (const float*) ((const uint8_t*) wk->data + wk->codes_bytes);
}
if (o.native_ple_key && wk->quantized()) {
if (!wk->native_data || (wk->native_type != 42 && wk->native_type != 18 && wk->native_type != 23 &&
wk->native_type != 8) || !wk->native_q8_1) {
std::fprintf(stderr, "strata generate: native PLE key is absent or incompatible\n");
return 1;
}
ss.ple.w.key_native_data = wk->native_data;
ss.ple.w.key_native_type = wk->native_type;
ss.ple.w.key_native_q8_1 = wk->native_q8_1;
}
ss.ple.w.value_bf16 = (const uint16_t*) wv->data;
ss.ple.w.norm_key = (const float*) wnk->data;
ss.ple.w.norm_query = (const float*) wnq->data;
ss.ple.w.norm_conv = (const float*) wnc->data;
// The conv1d kernel reads F16, so this cast is a claim about the pack's storage. A checkpoint that keeps
// the tensor F32 (Q8_0, UD-Q4_K_XL) would hand the kernel the low halves of the f32 words - not an error,
// a plausible wrong layer-1 routing. tools/iq_pack.py narrows it (index kind 3); a pack that did not is
// refused here (#255, gopinath87607). F16 is index kind 5 or 3 (F16InF32) in a native pack and kind 0
// (verbatim 2-byte F16) in the canonical Q2_0 pack; BF16 (kind 4) has the same size and is not F16.
const bool f16 = wc->kind == strata::core::WeightKind::F16InF32 ||
(wc->kind == strata::core::WeightKind::Verbatim && wc->code_bits == 0);
if (!f16 || wc->bytes != (uint64_t) wc->elements * 2) {
std::fprintf(stderr, "strata generate: blk.1.ple_conv1d.weight is not stored as F16 (pack index kind %d, "
"%llu B for %lld values); the PLE conv1d kernel reads F16 - repack with "
"tools/iq_pack.py\n",
(int) wc->kind, (unsigned long long) wc->bytes, (long long) wc->elements);
return 1;
}
ss.ple.w.conv1d_f16 = (const uint16_t*) wc->data;
ss.ple.consts = strata::kernels::ple_artifact_consts();
if (o.ple_delay_us > 0) ple_table.set_injected_delay_us(o.ple_delay_us);
ss.ple.table = &ple_table;
ss.ple.token = &ss.ple_token;
ss.ple.prev = ss.ple_prev;
// `ss.ple.hist` and the ready() check wait for `session_init`, which carves the history - the session is
// now allocated after the layer-split search (see the carve), and the wiring lands there
ss.ple.emb_host = ple_emb_host.data();
if (cudaMalloc((void**) &ple_emb_dev, (size_t) strata::kernels::NG_N_EMBD * 4) != cudaSuccess ||
cudaMalloc((void**) &ple_scratch, strata::core::ple_run_scratch_bytes()) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the PLE buffers failed\n");
return 1;
}
ss.ple.emb_dev = ple_emb_dev;
ss.ple.scratch = ple_scratch;
} else {
std::fprintf(stderr,
"strata generate: PLE OFF by explicit --no-ple diagnostic request.\n"
" The tokens below are NOT this model's; this is only useful for A/B measurement.\n");
}
float* d_parts = nullptr;
if (cudaMalloc(&d_parts, (size_t) K * g.n_embd * 4) != cudaSuccess ||
cudaMemset(d_parts, 0, (size_t) K * g.n_embd * 4) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the parts buffer failed\n");
return 1;
}
// ---- --split-skip-if-fits: before any later stage loads, does CUDA0 alone hold every profiled pair? What it
// still has to allocate on one GPU is the whole session (the KV of every layer), the drafter and the head
// (kDrafterMib below), the verify windows and the reserve; the prompt path borrows from the cache. If the
// profile's pairs fit in what is left, a split would only add the hand-offs: run on CUDA0 alone.
if (multi_gpu && split_auto && o.split_skip_if_fits) {
std::vector<std::pair<int32_t, int32_t>> prof;
int64_t pslots = 0;
std::string perr;
const bool remote = o.expert_cache_remote[0] > 0 || o.expert_cache_remote[1] > 0 || o.expert_cache_remote[2] > 0;
if (remote || o.expert_profile.empty() ||
!strata::core::read_expert_profile(o.expert_profile, g.n_layers, g.n_expert, prof, pslots, perr)) {
std::fprintf(stderr, "strata generate: --split-skip-if-fits: %s; the split stays\n",
remote ? "remote expert caches are in use" : perr.empty() ? "no expert profile" : perr.c_str());
} else {
const auto& lay = strata::kernels::cpu::expert_layout();
int64_t pairs_bytes = 0;
for (const auto& pr : prof)
pairs_bytes += native_pack ? ((int64_t) lay.blob_bytes(pr.first) + 255) / 256 * 256 : (int64_t) lay.max_blob;
size_t fb = 0, tb = 0;
cudaMemGetInfo(&fb, &tb);
const int64_t session = (int64_t) strata::core::session_bytes(g, o.max_context, K, 0, g.n_layers);
const int64_t held_back = session + (((int64_t) o.vram_reserve_mib + 1000 + 96) << 20); // + drafter/head, windows
const int64_t room = (int64_t) fb - held_back;
cudaDeviceProp dp{};
cudaGetDeviceProperties(&dp, 0);
if (pairs_bytes <= room) {
std::fprintf(stderr, "strata generate: layer split skipped (--split-skip-if-fits): CUDA0 (%s) holds all "
"%zu profiled pairs (%.2f GiB) with the session (%.2f GiB, %lld-token context), "
"the drafter and the reserve: %.2f GiB free, %.2f GiB to spare - one GPU\n",
dp.name, prof.size(), (double) pairs_bytes / 1073741824.0, (double) session / 1073741824.0,
(long long) o.max_context, (double) fb / 1073741824.0,
(double) (room - pairs_bytes) / 1073741824.0);
multi_gpu = false;
split_auto = false;
split_devs.clear();
split_at.clear();
o.layer_split.clear();
} else {
std::fprintf(stderr, "strata generate: --split-skip-if-fits: CUDA0 (%s) would hold only %.2f of the "
"profile's %.2f GiB (%.2f GiB free, %.2f GiB for the session, drafter and "
"reserve): the split stays\n", dp.name,
(double) std::max<int64_t>(room, 0) / 1073741824.0, (double) pairs_bytes / 1073741824.0,
(double) fb / 1073741824.0, (double) held_back / 1073741824.0);
}
}
}
strata::core::Verifier::set_commit_async(!multi_gpu); // see Verifier::set_commit_async
// ---- layer split across GPUs: each later stage's own copy of the dense weights, its session and (the last) the
// head, made on its device before the host arena is mapped (as the drafter below, for the same WDDM reason)
std::vector<std::unique_ptr<GpuStage>> stages;
for (size_t i = 0; multi_gpu && i < split_devs.size(); ++i) {
stages.push_back(std::make_unique<GpuStage>());
GpuStage& st = *stages.back();
st.dev = split_devs[i];
double free_gib = 0;
if (!strata::core::RemoteExperts::preflight(st.dev, free_gib, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
const strata::core::OnDevice on(st.dev);
void* arena_s = nullptr;
if (cudaMalloc(&arena_s, pool_bytes) != cudaSuccess ||
!st.wt.load(o.pack, arena_s, pool_bytes, err, skip.empty() ? nullptr : &skip)) {
cudaGetLastError();
size_t free_b = 0, total_b = 0; // #486: what that card had free
cudaMemGetInfo(&free_b, &total_b);
std::fprintf(stderr, "strata generate: layer split, CUDA%d weights: %s (%llu MiB needed, %llu MiB of %llu "
"MiB free on that card)\n", st.dev,
err.empty() ? "the weight arena does not fit" : err.c_str(),
(unsigned long long) (pool_bytes >> 20), (unsigned long long) (free_b >> 20),
(unsigned long long) (total_b >> 20));
return 1;
}
if (!o.native_dense_gguf.empty() && !st.dense.load(o.native_dense_gguf, st.wt, err, o.native_ple_key)) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d native dense projections: %s\n", st.dev,
err.c_str());
return 1;
}
// THE SESSION AND THE HEAD WAIT FOR THE SPLIT SEARCH. `session_bytes` prices a stage's session by its
// LAYER RANGE (the carve - every stage used to hold all 48 layers' state whatever it ran), so the
// sessions are allocated after the search below has set `st.lb`/`st.le`; the last stage's head follows.
if (cudaStreamCreateWithFlags(&st.stream, cudaStreamNonBlocking) != cudaSuccess ||
cudaStreamCreateWithFlags(&st.adapt_stream, cudaStreamNonBlocking) != cudaSuccess ||
cudaEventCreateWithFlags(&st.adapt_ev, cudaEventDisableTiming) != cudaSuccess) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d: its streams failed\n", st.dev);
return 1;
}
// a control vector (the speed projection): its tables on this device too - the stage's layers apply it here
if (!strata::kernels::cvec_replicate(err)) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d: %s\n", st.dev, err.c_str());
return 1;
}
// --vision: this device's image-position table (the identity until a picture request), read by every rope
// kernel its stage runs - set before any of its graphs is captured
if (o.vision) {
if (cudaMalloc(&st.mrope, mrope_host.size() * sizeof(int32_t)) != cudaSuccess ||
cudaMemcpy(st.mrope, mrope_host.data(), mrope_host.size() * sizeof(int32_t), cudaMemcpyHostToDevice) !=
cudaSuccess) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d: the image position table failed\n", st.dev);
return 1;
}
strata::kernels::mrope_table_set(st.mrope);
}
// its own PCIe share of the missed experts (the same rule as CUDA0's above: its link is probed). A given
// --pcie-frac is every stage's share and skips these probes (pcie_given); there is no per-stage setting yet.
st.pcie_frac = o.pcie_frac;
if (!pcie_given && native_pack) {
std::string bursts;
const double bw = probe_pcie_h2d_gbps(&bursts);
if (bw > 0.0) st.pcie_frac = pcie_frac_for_gbps(bw, 0.55);
std::fprintf(stderr, "strata generate: layer split: CUDA%d PCIe probe %.1f GB/s (best of %s) -> pcie_frac "
"%.2f\n", st.dev, bw, bursts.c_str(), st.pcie_frac);
}
size_t fb = 0, tb = 0;
cudaMemGetInfo(&fb, &tb);
std::fprintf(stderr, "strata generate: layer split: CUDA%d holds its weights; %.2f GiB free (its session "
"follows the split search)\n", st.dev, (double) fb / 1073741824.0);
}
GpuStage* const last_st = stages.empty() ? nullptr : stages.back().get();
std::vector<std::pair<int32_t, int32_t>> profile;
if (!o.expert_profile.empty()) {
int64_t pslots = 0;
if (!strata::core::read_expert_profile(o.expert_profile, g.n_layers, g.n_expert, profile, pslots, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
// An explicit number truncates the ranked list ("what would 2,000 slots give" without rebuilding the
// file). `--expert-cache 0` used to take the count the profile was built for; the profile now ranks
// every pair (issue #46: a card that holds more than the old 8,000 used to stop there), so it means auto.
if (o.expert_cache == 0) o.expert_cache = -1;
// Multi-GPU: STRATA_PEER_HOT=<f> gives the peer card a share f of the HOT pairs, so both cards
// compute routed experts every layer (the primary alone did ~25 of ~30 per layer-window). Of the first
// STRATA_PEER_HOT_AT (default 8700, ~ the primary's slots) ranks, every pair with floor((r+1)f) > floor(rf)
// moves to just after that point: the primary fills past them, the peer (which takes what the primary does
// not hold, in order) gets them first.
if (o.peer_device >= 1) { // default 0.45 (measured: 0.3-0.6 all better than 0; 0 = off, e.g. for the gate)
const char* ph = std::getenv("STRATA_PEER_HOT");
const double f = ph ? std::atof(ph) : 0.45;
const char* pa = std::getenv("STRATA_PEER_HOT_AT");
const size_t at = std::min(profile.size(), (size_t) (pa ? std::atoll(pa) : 8700));
if (f > 0.0 && f < 1.0 && at > 0) {
std::vector<std::pair<int32_t, int32_t>> keep, moved;
for (size_t r = 0; r < at; ++r) {
const bool to_peer = (int64_t) ((double) (r + 1) * f) > (int64_t) ((double) r * f);
(to_peer ? moved : keep).push_back(profile[r]);
}
const size_t n_moved = moved.size();
// the primary's share continues with the ranks after `at` until it is full; then the moved ones
std::vector<std::pair<int32_t, int32_t>> out;
out.reserve(profile.size());
out.insert(out.end(), keep.begin(), keep.end());
const size_t fill = std::min(profile.size(), at + n_moved); // what the primary still takes
out.insert(out.end(), profile.begin() + (long) at, profile.begin() + (long) fill);
out.insert(out.end(), moved.begin(), moved.end());
out.insert(out.end(), profile.begin() + (long) fill, profile.end());
profile.swap(out);
std::fprintf(stderr, "strata generate: STRATA_PEER_HOT %.2f: %zu of the first %zu ranked pairs moved "
"behind rank %zu (the peer's)\n", f, n_moved, at, fill);
}
}
std::fprintf(stderr, "strata generate: profile %s: %zu ranked pairs, built for %lld slots\n",
o.expert_profile.c_str(), profile.size(), (long long) pslots);
}
// #477: the whole ranking as loaded, the prior of --expert-profile-save's order (a layer split keeps only
// CUDA0's pairs in `profile` below). Empty without --expert-profile-save.
std::vector<std::pair<int32_t, int32_t>> profile_loaded;
if (!o.expert_profile_save.empty()) profile_loaded = profile;
// ---- layer split across GPUs: "auto" places the split points by a cost model of one decode window, measured on
// the 5080 + 3090 rig (bench/results/2026-09-29-layer-split):
// - every layer costs its GPU a time inversely proportional to SMs x clock (0.33 ms on an RTX 5080, 0.50 on a
// 3090: the per-layer round trip and kernels, not the bytes - both cards have ~950 GB/s);
// - an expert no cache holds costs ~190 ms per unit of routed mass: the CPU pool in decode and the PCIe stream
// in prompts (fitted: the sweep's best K, 26-28, is where one more layer on the faster card stops paying
// for the ~0.1% of the mass it pushes out of its cache);
// - which experts a cache holds: its layers' profiled pairs, hottest first, until its free VRAM (less the
// reserve, the prompt path's buffers and, on a later GPU, 1 GiB for its windows and the drafter) is used;
// the routed mass of rank r is taken as (r+1)^-1.2 (fits the sweep's hit rates: K=24/26/28 predicted
// 99.53/99.34/99.15%, measured 99.5/99.4/99.0%).
// Up to 3 GPUs every placement is tried; beyond, the layers are shared in proportion to speed.
// STRATA_SPLIT_MISS_MS tunes the miss cost (a slower CPU: higher).
// THE PROMPT PATH'S BUFFERS ARE BORROWED FROM THE CACHE, NOT WITHHELD BESIDE IT. With borrowing the cache
// is sized first and at full size, and the prompt path is laid out in the tail of it (`Prefill::relayout`),
// so it withholds no VRAM of its own and this reserve is zero. Only without borrowing - no profile to fill
// a cache from, or --no-prefill-borrow - do the buffers take a reserve, and then this estimate stands in
// for buffers that cannot be priced exactly yet because the sessions do not exist. `plan_lend` uses the
// exact `Prefill::bytes_needed` as soon as it can.
const bool pf_borrow = !o.no_prefill_borrow && !o.expert_profile.empty();
// (#340: the estimate predates the streamed ring: from 1024-token chunks the prompt path also holds a ring of
// whole expert blobs, which a split without borrowing sizes at 96 (Prefill::set_ring_override below) and books
// here - without it a `--no-prefill-borrow` split filled the cards and the draft head no longer fit)
const int64_t split_ring_mib =
(multi_gpu && !pf_borrow && o.prefill_chunk >= 1024)
? (int64_t) ((96ull * (uint64_t) strata::kernels::cpu::expert_layout().max_blob + (1ull << 20) - 1) >> 20)
: 0;
if (split_ring_mib > 0) strata::prefill::Prefill::set_ring_override(96);
const int64_t split_pf_mib =
(o.prefill_chunk > 0 && !pf_borrow) ? 160 + (o.prefill_chunk * 680) / 1024 + split_ring_mib : 0;
// ---- WHAT A STAGE RESERVES, AND ON WHICH STAGE. The flat 1 GiB this used to withhold from EVERY stage
// after the first was booked "for its windows and the drafter", but the windows measure 75 MiB ("window up
// to 6 tokens, 74.1 MiB of device buffers", on every boot) and the drafter is loaded on ONE stage - the
// last one, which is also the only one that holds the head. On the two identical 8 GB cards that GiB was
// the entire difference between CUDA0's cache and CUDA1's: 814 slots against 188, 2026-09-30. Both of
// those allocations are already made before a stage's cache is sized, so what has to be held back here is
// the windows and - only on the stage that carries them - the drafter and the head.
const int64_t kWindowMib = 96; // the verify windows; 75 MiB measured, rounded up
const int64_t kDrafterMib = 1000; // the MTP drafter (839 MiB) + the head, on the last stage only
// #340: with the own prompt buffers chosen by the split's rule (not asked for with --no-prefill-borrow) the
// boundary is searched as the borrowing configuration would (no reserve): the reserve then only makes the caches
// smaller, which measured cost no decode (K=28 on 9070 XT + R9700: 58.4 tok/s own vs 58.5 borrowing), while a
// search with the reserve moved the boundary to K=32 and decode to 54.8. STRATA_SPLIT_OWN_PLACE=reserve: the
// search sees the reserve.
static const bool place_with_reserve = [] {
const char* v = std::getenv("STRATA_SPLIT_OWN_PLACE");
return v != nullptr && std::string(v) == "reserve";
}();
auto stage_room = [&](int dev, bool later, bool drafter, bool search = false) -> int64_t {
const strata::core::OnDevice on(dev);
size_t fb = 0, tb = 0;
if (const cudaError_t e = cudaMemGetInfo(&fb, &tb); e != cudaSuccess)
std::fprintf(stderr, "strata generate: layer split: CUDA%d free memory: %s\n", dev < 0 ? 0 : dev,
cudaGetErrorString(e));
const int64_t pf = search && split_own_auto && !place_with_reserve ? 0 : split_pf_mib;
const int64_t reserve = ((int64_t) o.vram_reserve_mib + pf + (later ? kWindowMib : 0) +
(drafter ? kDrafterMib : 0)) << 20;
return std::max<int64_t>((int64_t) fb - reserve, 0);
};
if (multi_gpu && split_auto) {
const auto& lay = strata::kernels::cpu::expert_layout();
const int ns = (int) stages.size() + 1;
std::vector<int64_t> cap((size_t) ns), used((size_t) ns);
std::vector<double> layer_ms((size_t) ns);
for (int i = 0; i < ns; ++i) {
const int dev = i == 0 ? 0 : stages[(size_t) i - 1]->dev;
cap[(size_t) i] = stage_room(i == 0 ? -1 : dev, i > 0, i + 1 == ns, true);
int sms = 0, khz = 0;
cudaDeviceGetAttribute(&sms, cudaDevAttrMultiProcessorCount, dev);
if (cudaDeviceGetAttribute(&khz, cudaDevAttrClockRate, dev) != cudaSuccess || khz <= 0) khz = 1800000;
cudaGetLastError();
const double speed = std::max(1.0, (double) sms * (double) khz / 1e6); // SMs x GHz
layer_ms[(size_t) i] = 0.33 * (84.0 * 2.617) / speed;
std::fprintf(stderr, "strata generate: layer split auto: CUDA%d %d SMs at %.2f GHz -> %.2f ms per layer, "
"%.2f GiB free before its session carve\n", dev, sms, khz / 1e6, layer_ms[(size_t) i],
(double) cap[(size_t) i] / 1073741824.0);
}
const double miss_ms = std::getenv("STRATA_SPLIT_MISS_MS") ? std::atof(std::getenv("STRATA_SPLIT_MISS_MS")) : 190.0;
std::vector<double> mass(profile.size());
double total_mass = 0;
for (size_t r = 0; r < profile.size(); ++r) total_mass += (mass[r] = std::pow((double) r + 1.0, -1.2));
auto cost = [&](int64_t l) -> int64_t {
return native_pack ? ((int64_t) lay.blob_bytes(l) + 255) / 256 * 256 : (int64_t) lay.max_blob;
};
// the predicted window time (ms) of a placement, and the routed mass its caches hold
auto predict = [&](const std::vector<int64_t>& at, double& held_mass, int64_t& held) -> double {
// THE CARVE, PRICED: a placement gives stage i the layers [lb, le), and that range's session is a
// real cost on its device - subtracted here so the search knows what it leaves for experts. This
// is why the sessions are allocated after the search: `session_bytes` is pure arithmetic.
std::vector<int64_t> capr((size_t) ns);
for (int i = 0; i < ns; ++i) {
const int64_t lb = i == 0 ? 0 : at[(size_t) i - 1];
const int64_t le = i + 1 < ns ? at[(size_t) i] : g.n_layers;
capr[(size_t) i] = cap[(size_t) i] - (int64_t) strata::core::session_bytes(g, o.max_context, K, lb, le);
}
std::fill(used.begin(), used.end(), 0);
held_mass = 0;
held = 0;
std::vector<bool> full((size_t) ns, false);
for (size_t r = 0; r < profile.size(); ++r) {
const int64_t l = profile[r].first;
int st = 0;
while (st + 1 < ns && l >= at[(size_t) st]) ++st;
if (full[(size_t) st]) continue;
if (used[(size_t) st] + cost(l) > capr[(size_t) st]) { full[(size_t) st] = true; continue; } // as the fill
used[(size_t) st] += cost(l);
held_mass += mass[r];
++held;
}
held_mass /= std::max(total_mass, 1e-9);
double ms = miss_ms * (1.0 - held_mass);
for (int i = 0; i < ns; ++i) {
const int64_t lb = i == 0 ? 0 : at[(size_t) i - 1], le = i + 1 < ns ? at[(size_t) i] : g.n_layers;
ms += (double) (le - lb) * layer_ms[(size_t) i];
}
return ms;
};
std::vector<int64_t> best, at((size_t) ns - 1);
double best_ms = 1e30, best_mass = 0;
int64_t best_held = 0;
auto consider = [&]() {
double hm = 0;
int64_t held = 0;
const double ms = predict(at, hm, held);
if (ms < best_ms) { best = at; best_ms = ms; best_mass = hm; best_held = held; }
};
const int64_t L = g.n_layers;
if (ns == 2) {
for (int64_t k = 2; k < L; ++k) { at[0] = k; consider(); }
} else if (ns == 3) {
for (int64_t k1 = 2; k1 + 1 < L; ++k1)
for (int64_t k2 = k1 + 1; k2 < L; ++k2) { at[0] = k1; at[1] = k2; consider(); }
} else {
double total = 0;
for (const double c : layer_ms) total += 1.0 / c;
double acc = 0;
for (int i = 0; i + 1 < ns; ++i) {
acc += 1.0 / layer_ms[(size_t) i];
at[(size_t) i] = std::clamp<int64_t>((int64_t) std::llround(acc / total * (double) L),
i == 0 ? 2 : at[(size_t) i - 1] + 1, L - (ns - 1 - i));
}
consider();
}
split_at = best;
std::string ks;
for (const int64_t k : split_at) ks += (ks.empty() ? "" : ",") + std::to_string(k);
std::fprintf(stderr, "strata generate: layer split auto: K=%s - predicted %.1f ms per decode window; the caches "
"hold %lld of %zu profiled pairs (~%.1f%% of the routed mass)\n", ks.c_str(), best_ms,
(long long) best_held, profile.size(), 100.0 * best_mass);
}
for (size_t i = 0; i < split_at.size(); ++i)
if (split_at[i] >= g.n_layers) {
std::fprintf(stderr, "strata generate: --layer-split: layer %lld is past the last (%lld)\n",
(long long) split_at[i], (long long) (g.n_layers - 1));
return 2;
}
// the stage that runs a layer (0: CUDA0's)
auto stage_of = [&](int64_t l) -> int {
int st = 0;
while (st < (int) split_at.size() && l >= split_at[(size_t) st]) ++st;
return st;
};
if (multi_gpu) {
std::vector<std::pair<int32_t, int32_t>> mine;
for (const auto& pr : profile) {
const int st = stage_of(pr.first);
(st == 0 ? mine : stages[(size_t) st - 1]->profile).push_back(pr);
}
profile.swap(mine);
for (size_t i = 0; i < stages.size(); ++i) {
stages[i]->lb = split_at[i];
stages[i]->le = i + 1 < stages.size() ? split_at[i + 1] : g.n_layers;
}
}
// ---- CUDA0's session, and the stages' sessions: sized to each device's own layer range (the carve). A
// stage that runs [lb, le) carves only those layers' GDN rows and QSA pools - before the carve every stage
// held all 48 layers' state whatever layers it ran, which is the same disease the chunked-QSA-prefill PR
// fixed in llama.cpp: allocation sized by the whole model instead of the device's own work.
{
const strata::core::OnDevice on0(0);
const int64_t hi0 = multi_gpu ? split_at[0] : -1;
if (cudaMalloc(&sbuf, strata::core::session_bytes(g, o.max_context, K, 0, hi0)) != cudaSuccess) {
std::fprintf(stderr, "strata generate: session state allocation failed\n");
return 1;
}
if (strata::core::session_init(g, o.max_context, K, sbuf, ss, 0, hi0) == 0) {
std::fprintf(stderr, "strata generate: session_init failed\n");
return 1;
}
if (g.n_qsa_layers() > 0 && ss.qsa_states[ss.qsa_primary()].kv_mode == 1)
std::fprintf(stderr, "strata generate: KV streaming: %lld of %lld cells per QSA layer in VRAM, the K/V in "
"%.2f GiB of pinned RAM\n",
(long long) (ss.qsa_states[ss.qsa_primary()].n_slots * 4),
(long long) o.max_context, (double) strata::core::qsa_kv_host_bytes() / 1073741824.0);
// the PLE block above built everything but the history, which session_init has just carved
ss.ple.hist = ss.ple_hist;
// (#167) generate mode starts from an empty sequence, and nothing else zeroes this state before the prompt
// path or the verifier reads it (--serve zeroes it per request when nothing is reused)
strata::core::session_zero(ss, g, nullptr, main_cs);
if (cudaDeviceSynchronize() != cudaSuccess) {
std::fprintf(stderr, "strata generate: zeroing the session state failed\n");
return 1;
}
if (!o.ple_gguf.empty()) {
if (!ss.ple.ready()) {
std::fprintf(stderr, "strata generate: the PLE run is not ready after construction\n");
return 1;
}
std::fprintf(stderr, "strata generate: PLE on, table %llu rows of %s\n",
(unsigned long long) ple_table.rows(), o.ple_gguf.c_str());
}
}
for (size_t i = 0; i < stages.size(); ++i) {
GpuStage& st = *stages[i];
const strata::core::OnDevice on(st.dev);
void* sbuf_s = nullptr;
if (cudaMalloc(&sbuf_s, strata::core::session_bytes(g, o.max_context, K, st.lb, st.le)) != cudaSuccess ||
strata::core::session_init(g, o.max_context, K, sbuf_s, st.ss, st.lb, st.le) == 0) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d: the session state failed\n", st.dev);
return 1;
}
const bool last = i + 1 == stages.size();
const strata::core::WeightRef* wo_s = st.wt.find("output.weight");
if (wo_s == nullptr ||
(last && !o.native_head_gguf.empty() && !st.head.load(o.native_head_shards, g.n_embd, wo_s->ne1, err))) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d head: %s\n", st.dev,
wo_s == nullptr ? "output.weight is missing" : err.c_str());
return 1;
}
size_t fb = 0, tb = 0;
cudaMemGetInfo(&fb, &tb);
std::fprintf(stderr, "strata generate: layer split: CUDA%d holds its weights, session [%lld, %lld)%s; "
"%.2f GiB free\n", st.dev, (long long) st.lb, (long long) st.le,
last ? " and the head" : "", (double) fb / 1073741824.0);
}
// Secure MTP's CUDA0 allocations before the large host arena is registered with both CUDA contexts.
// In particular WDDM can refuse the draft weights after mapping tens of GiB of host pages.
strata::core::MtpDrafter mtp;
if (!o.mtp.empty()) {
if (o.spec < 2) {
std::fprintf(stderr, "strata generate: --mtp is ignored without --spec T (T >= 2)\n");
o.mtp.clear();
}
if (!o.mtp.empty()) mtp.set_prompt_len((int64_t) o.tokens.size());
// the draft layer is the canonical model's MTP head (512 experts) even when the target is pruned,
// so it always sees the canonical geometry; `static` because MtpDrafter keeps a reference
static const strata::core::ModelGeometry draft_geometry{};
// with a layer split across GPUs the drafter reads the last stage's residual: it lives on that device
const strata::core::OnDevice on_mtp(last_st ? last_st->dev : -1);
if (!o.mtp.empty() && !mtp.load(o.mtp, draft_geometry, last_st ? last_st->ss : ss, o.spec, err, o.mtp_window)) { std::fprintf(stderr, "strata generate: %s\n", err.c_str()); return 1; }
}
// Create the additional contexts after MTP has secured CUDA0 memory, but
// before the host arena maps its expert pages into their address spaces.
for (int r = 1; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0 && !early_remote) {
double free_gib = 0;
if (!strata::core::RemoteExperts::preflight(remote_dev[r], free_gib, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: CUDA%d context ready, %.2f GiB free before expert arena registration\n",
remote_dev[r], free_gib);
}
// ---- the CPU expert pool
//
// R2.1: the experts are loaded into a RESIDENT ARENA by default. The mmap path is kept behind
// `--mmap-experts` because it is the A/B arm, not because it is competitive.
//
// The reasoning is the review's C1 and it is now measured on both sides. `FileExpertSource` maps the 34 GB
// file, and mapped file pages are the first thing the OS reclaims; the engine's rate then depends on whether
// the standby list happens to hold `experts.bin`, which is why two consecutive runs of the SAME BINARY with
// the SAME FLAGS measured 71.97 and 34.78 ms/token in the pool (7.54 vs 12.18 tok/s). The arena is
// anonymous memory the engine owns, and the pool runs at 19.41 ms/token - 1.79x better than the warm mmap
// and 3.7x better than the cold one.
//
// IT IS NOT PINNED, and that is reported rather than hidden: `cudaHostRegister` on 31.64 GiB fails with
// "out of memory" (you cannot pin 34 of 63 GB) and the arena falls back to 4 KB anonymous pages. That is
// fine for the CPU pool - which is all that exists today - and NOT fine for Phase 3, whose cache fills and
// CPU/PCIe miss split need the GPU to DMA out of this arena. Read `note()` when that lands.
//
// The earlier "the arena does not fit" conclusion was WRONG and is worth recording: the failure was a stale
// CUDA error left set by the failed `cudaHostRegister` and read later by `gr_read`'s launch check. See the
// note in `pinned.cu`.
strata::core::FileExpertSource src;
{ // the card, and whether this build has code for it (a binary built for other GPUs fails at its first kernel
// otherwise, after the whole expert arena has loaded) - before the arena starts loading
int dev = 0;
cudaDeviceProp p{};
const bool named = cudaGetDevice(&dev) == cudaSuccess && cudaGetDeviceProperties(&p, dev) == cudaSuccess;
if (!named) cudaGetLastError();
const char* name = named && p.name[0] ? p.name : "(an unnamed GPU)";
#if defined(STRATA_USE_HIP)
std::fprintf(stderr, "strata generate: GPU %d: %s (%s)\n", dev, name, named ? p.gcnArchName : "?");
#if defined(_WIN32)
// #468 #461: which HIP runtime was loaded - the bundled one beside the exe, or an AMD driver's System32 copy
if (HMODULE h = GetModuleHandleA("amdhip64_7.dll")) {
char path[MAX_PATH] = {};
if (GetModuleFileNameA(h, path, MAX_PATH) > 0)
std::fprintf(stderr, "strata generate: HIP runtime %s\n", path);
}
#endif
#else
std::fprintf(stderr, "strata generate: GPU %d: %s, compute capability %d.%d%s\n", dev, name,
strata::cc_major_of(p.major), strata::cc_minor_of(p.minor),
strata::emulated_cc() ? " (STRATA_EMULATE_CC: a test mode, the card is emulated)" : "");
{ // #542: a build whose libcudart is older than its headers (a CUDA 13 kit with a dangling libcudart.so that
// CMake resolved to the system's CUDA 12 one) reads cudaDeviceProp shifted - silently, and slowly
int rt = 0;
if (cudaRuntimeGetVersion(&rt) == cudaSuccess && rt / 1000 != CUDART_VERSION / 1000)
std::fprintf(stderr, "strata generate: WARNING: this engine was compiled with CUDA %d.%d headers but "
"loaded a CUDA %d.%d runtime (libcudart): GPU properties can read wrong and some "
"kernels go unused. Rebuild it against one toolkit (cmake -DCUDAToolkit_ROOT=<the "
"toolkit>, with its libcudart.so present) (#542)\n",
CUDART_VERSION / 1000, CUDART_VERSION % 1000 / 10, rt / 1000, rt % 1000 / 10);
}
#endif
const std::string e = strata::core::device_code_error();
if (!e.empty()) {
std::fprintf(stderr, "strata generate: this engine has no code for %s (sm_%d%d): %s - rebuild it for this "
"card (setup does: START-HERE.bat --setup)\n", name, p.major, p.minor, e.c_str());
return 1;
}
}
strata::core::ArenaExpertSource arena_src;
strata::core::ExpertSource* srcp = nullptr;
if (o.mmap_experts) {
// FileExpertSource maps the pack's experts.bin: a canonical pack has it; a native (IQ) pack has it when
// built with `tools/iq_pack.py --experts-bin` (the per-layer blob sizes of its layout, PR #121). The low-RAM
// mode: the experts come from the file through the OS cache instead of a pinned copy in RAM, for a PC whose
// GPU holds most of them but whose RAM cannot hold them all.
// CS-T: a native pack without experts.bin maps the model's GGUF shards instead (native_experts.txt's spans,
// checked against the files first) - no 30-77 GB copy of the experts on the disk
src.set_gguf(o.native_preset);
if (const char* v = std::getenv("STRATA_FETCH_THREADS"); v != nullptr && std::atoi(v) > 0)
src.set_fetch_threads(std::atoi(v));
if (!src.open(o.pack, g.n_layers, g.n_expert, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: experts via mmap (--mmap-experts; %s)\n",
src.gguf_mode() ? "the GGUF shards in place, no experts.bin" : "the A/B arm of R2.1");
// #286: with a RAM budget the hottest experts live in it, and the rest are read from the drive unbuffered
// when the file cache could not keep them beside the budget anyway (a 32 GB PC) - the mapped reads' page
// faults are small requests on the critical path, and their pages take the RAM the budget was sized for
if (o.resident_budget > 0 || std::getenv("STRATA_UNBUFFERED_LOAD") != nullptr) {
std::string why;
const bool ub = src.set_unbuffered(o.resident_budget, why);
std::fprintf(stderr, "strata generate: the file tier reads %s (%s)\n",
ub ? "unbuffered" : "through the file cache", why.c_str());
}
srcp = &src;
} else {
arena_src.set_gguf(o.native_preset); // plan v0.3 P6: a native pack may take its experts from shard 1
// Under WDDM (Windows, WSL2), a multi-GPU run (a layer split, or remote experts) starts with at most 8 GiB of
// mapped host pages: pinning all of it into two contexts leaves WDDM refusing every later allocation -
// measured on the 5080 + 3090 rig: cudaMemGetInfo and the next cudaMalloc fail. Unregistered layers remain
// in the resident arena; their streamed experts go through the pinned staging ring. A Linux driver has no
// such limit, so there the whole arena is pinned (#253). STRATA_ARENA_PIN_GIB overrides both ways.
const int pin_env = strata::core::arena_pin_cap_gib(); // -1 unset, -2 "auto" (#243, Windows sliced pin)
const bool pin_wddm_cap = pin_env < 0 && (o.expert_cache_remote[0] > 0 || multi_gpu) && under_wddm();
const uint64_t pin_limit = pin_env >= 0 ? (uint64_t) pin_env << 30 : pin_wddm_cap ? (8ull << 30) : 0;
if (pin_env >= 0)
std::fprintf(stderr, "strata generate: STRATA_ARENA_PIN_GIB=%d: %s\n", pin_env,
pin_env == 0 ? "the whole expert arena is pinned" : "the expert arena's pinning is capped");
else if (pin_wddm_cap)
std::fprintf(stderr, "strata generate: multi-GPU under WDDM: at most 8 GiB of the expert arena is pinned "
"(STRATA_ARENA_PIN_GIB changes it)\n");
if (!arena_src.open(o.pack, g.n_layers, g.n_expert, /*threads=*/6, err, pin_limit,
o.shared_expert_arena)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: expert arena: %s\n", arena_src.note().c_str());
std::fprintf(stderr, "strata generate: loaded %.2f GiB at %.2f GiB/s\n",
(double) strata::kernels::cpu::expert_layout().total / (1024.0 * 1024 * 1024),
arena_src.load_gib_per_second());
// A rate under ~0.2 GiB/s is not the hardware. Task Scheduler / service contexts throttle this
// read+fill about 24x (measured 0.05 vs 1.42 GiB/s for the same binary, args and cache state; the
// scheduler's defaults - Below normal priority and a least-privilege token - were the only
// difference between the runs). Say so instead of letting the user blame the disk; see
// docs/DETAILS.md, "Running it at startup (Task Scheduler)".
#ifdef _WIN32 // a Windows launch context; elsewhere a load this slow is the disk
if (arena_src.load_gib_per_second() > 0.0 && arena_src.load_gib_per_second() < 0.2) {
std::fprintf(stderr,
"strata generate: hint: ~24x below what this hardware streams from a normal "
"launch. If Strata is started by Task Scheduler or a service, register the task "
"with Priority 4 (Normal) and 'Run with highest privileges' - the scheduler's "
"defaults (Below normal + a least-privilege token) throttle the load. See "
"docs/DETAILS.md ('Running it at startup').\n");
}
#endif
srcp = &arena_src;
}
strata::kernels::cpu::ExpertPool pool(o.pool_workers, /*pin=*/true, /*host_works=*/!o.no_host_worker, o.pool_affinity);
if (pool.is_hybrid() && pool.affinity() != strata::kernels::cpu::PoolAffinity::All) {
const char* aff_str = pool.affinity() == strata::kernels::cpu::PoolAffinity::PCores ? "p-cores" :
pool.affinity() == strata::kernels::cpu::PoolAffinity::All ? "all" : "auto";
std::fprintf(stderr, "strata generate: hybrid CPU detected (%d P-cores / %d threads, %d E-cores), pool workers: %d, affinity: %s\n",
pool.p_cores(), pool.p_threads(), pool.e_cores(), pool.workers(), aff_str);
}
if (o.no_ple_prefetch) strata::kernels::ple_prefetch_enable(false);
// ---- R4's slot storage. Allocated AFTER the weights and the session, so `cudaMemGetInfo` inside `open`
// sees the memory this process actually has left rather than the card's idle figure - and refuses with both
// numbers if the slots do not fit, instead of handing back a cache smaller than it was asked for.
mem_mark("the weights, the session and the drafter");
strata::core::ExpertCache xcache;
// THE HEAD BEFORE THE CACHE. The expert cache takes what is free minus the reserve, so everything allocated
// after it comes out of the reserve. The native head (~0.5 GB with IQ3_S) was loaded after it and ate most of
// the 700 MiB: 128K IQ3_S ended with 30 MiB free, the driver paged, and a request stalled for good at its first
// verify window. Loaded first, the cache is sized around it.
const strata::core::WeightRef* wo = wt.find("output.weight");
if (wo == nullptr) { std::fprintf(stderr, "strata generate: output.weight is missing\n"); return 1; }
const int64_t n_vocab = wo->ne1;
strata::core::NativeHead native_head;
if (!o.native_head_gguf.empty() && !multi_gpu) { // a layer split's head is on its last stage
if (!native_head.load(o.native_head_shards, g.n_embd, n_vocab, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: experimental native Q5_K head, %llu bytes\n",
(unsigned long long) native_head.weight_bytes());
}
std::vector<float> logits((size_t) n_vocab);
float* d_logits = nullptr;
if (cudaMalloc(&d_logits, (size_t) n_vocab * 4) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the logits buffer failed\n");
return 1;
}
const bool auto_cache = o.expert_cache < 0;
bool reserve_adapted = false; // #496: the auto sizing lowered the reserve so a small card's cache fits
if (o.expert_cache < 0) {
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
// Plan v0.3 P5: the batched prompt path's chunk buffers are allocated later, so they are reserved here -
// under WDDM an over-subscribed allocation does not fail, it pages to system memory and crawls.
// (with borrowing - the default with a profile - the prompt path lends cache slots instead; `pf_borrow` is
// the predicate a local `borrow` was here, hoisted above so both cache-size branches read the same one)
const int64_t prefill_mib = (o.prefill_chunk > 0 && !pf_borrow) ? 160 + (o.prefill_chunk * 680) / 1024 : 0;
// the draft layer's head and logits are allocated when it binds, after this: 0.1.27's CJK subset made them
// ~110-180 MiB larger, and out of the reserve they left 16 GB cards below the stall line (#199)
const int64_t mtp_bind = (!o.mtp.empty() && native_head.loaded())
? (int64_t) mtp.bind_bytes(native_head.row_bytes(), n_vocab) : 0;
const int64_t reserve = (((int64_t) o.vram_reserve_mib + prefill_mib) << 20) + mtp_bind;
const int64_t blob = (int64_t) strata::kernels::cpu::expert_layout().max_blob;
int64_t slots = ((int64_t) free_b - reserve) / blob;
if (!profile.empty()) slots = std::min<int64_t>(slots, (int64_t) profile.size());
o.expert_cache = (int) std::max<int64_t>(slots, 0);
std::fprintf(stderr, "strata generate: expert cache auto: %.2f GiB free, %d MiB reserved (+%lld MiB for the "
"draft head) -> %d slots\n",
(double) free_b / 1073741824.0, o.vram_reserve_mib, (long long) (mtp_bind >> 20), o.expert_cache);
// #496: the verify window cannot start without a cache (#174), and a cache too small to lend the prompt path
// a 256-token chunk's buffers (plus the 128 slots a loan leaves; one slot without --prefill) makes it
// allocate its own on top - more than the reserve. When the default reserve leaves less than that (a 6 GB
// card), the reserve shrinks to what leaves exactly that cache, down to kSmallReserveMib: what is allocated
// after the cache - the prompt path's own part, the verify buffers, the draft head - comes out of the reserve,
// and below ~550 MiB a card ends with less than the 256 MiB the serve check calls LOW (IQ3_XXS, 32K, a 300 MiB
// reserve: 5 MiB left), so the cache gets no more than it needs, and the serve check says so plainly when it
// ends LOW (`reserve_adapted`). A reserve given on the command line is kept. No slot at all: the start
// stops, saying what is short and what makes room. A card the default reserve leaves that much is sized as
// before.
constexpr int kSmallReserveMib = 300;
const int64_t min_slots = (o.prefill_chunk > 0 && pf_borrow)
? ((int64_t) strata::prefill::Prefill::bytes_needed(g, ss, 256) + blob - 1) / blob + 128 : 1;
if (o.expert_cache < min_slots && !o.vram_reserve_given && o.vram_reserve_mib > kSmallReserveMib) {
// the largest reserve (in MiB) that still leaves min_slots
const int64_t fit_mib = ((int64_t) free_b - mtp_bind - min_slots * blob) / (1 << 20) - prefill_mib;
if (fit_mib >= kSmallReserveMib) {
const int r = (int) std::min<int64_t>(fit_mib, o.vram_reserve_mib);
int64_t s2 = ((int64_t) free_b - ((((int64_t) r + prefill_mib) << 20) + mtp_bind)) / blob;
if (!profile.empty()) s2 = std::min<int64_t>(s2, (int64_t) profile.size());
std::fprintf(stderr, "strata generate: expert cache auto: the %d MiB reserve leaves too few slots on "
"this card (a working cache needs %lld): a %d MiB reserve instead -> %lld slots\n",
o.vram_reserve_mib, (long long) min_slots, r, (long long) s2);
o.vram_reserve_mib = r;
o.expert_cache = (int) s2;
reserve_adapted = true;
}
}
if (o.expert_cache == 0) {
// what is short, and what makes room: the numbers a small card picks from
const int64_t at_reserve = o.vram_reserve_given ? o.vram_reserve_mib
: std::min(o.vram_reserve_mib, kSmallReserveMib);
const int64_t need_b = (((int64_t) at_reserve + prefill_mib) << 20) + mtp_bind + min_slots * blob;
const int64_t short_mib = std::max<int64_t>(1, (need_b - (int64_t) free_b + (1 << 20) - 1) >> 20);
const int64_t session_mib =
(int64_t) (strata::core::session_bytes(g, o.max_context, K, 0, g.n_layers) >> 20);
const std::string reserve_tip =
o.vram_reserve_given && o.vram_reserve_mib > kSmallReserveMib
? ", a smaller --vram-reserve-mib (" + std::to_string(o.vram_reserve_mib) + " now; " +
std::to_string(kSmallReserveMib) + " is enough on a small card)"
: std::string();
std::fprintf(stderr, "strata generate: no VRAM is left for the expert cache: it needs at least %lld slots "
"(%lld MiB), about %lld MiB more than this card has free. To make room: a smaller "
"--max-context (the session, mostly its KV cache, takes %lld MiB at %lld tokens), "
"--kv q4_0, the English draft subset (setup --draft-vocab en; the draft head takes "
"%lld MiB now)%s, images on the CPU, or close other programs that use the GPU\n",
(long long) min_slots, (long long) ((min_slots * blob) >> 20), (long long) short_mib,
(long long) session_mib, (long long) o.max_context, (long long) (mtp_bind >> 20),
reserve_tip.c_str());
}
} else if (multi_gpu && o.expert_cache > 0) {
// an explicit cache size leaves room for the prompt path's buffers and the reserve, or the first prompt
// fails with "device buffers ... do not fit" (with borrowing - the default with a profile - the path lends
// slots instead and `prefill_mib` is 0, so only the reserve is checked)
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
const int64_t prefill_mib = (o.prefill_chunk > 0 && !pf_borrow) ? 160 + (o.prefill_chunk * 680) / 1024 : 0;
const int64_t reserve = ((int64_t) o.vram_reserve_mib + prefill_mib) << 20;
const int64_t fit = std::max<int64_t>(((int64_t) free_b - reserve) / (int64_t) strata::kernels::cpu::expert_layout().max_blob, 0);
if (o.expert_cache > fit) {
// a WARNING that names the knob: the user asked for this size, and gets fewer slots
std::fprintf(stderr, "strata generate: WARNING: layer split: --expert-cache %d leaves no room for the "
"prompt path's buffers (%lld MiB) and the %d MiB reserve on CUDA0: %lld slots instead "
"(a smaller --vram-reserve-mib leaves more of them)\n", o.expert_cache,
(long long) prefill_mib, o.vram_reserve_mib, (long long) fit);
o.expert_cache = (int) fit;
}
}
// plan v0.3 P6: a native pack's blobs differ per layer, so with a profile its slots are sized per pair: the
// same VRAM holds ~30% more IQ3_XXS experts than slots of the largest blob would
// #369: not with --expert-cache-per-layer - its per-layer slot ranges ignore the profile rank a sized slot was cut
// for, so a layer's larger blob could land in a smaller slot: that mode keeps slots of the largest blob
std::vector<int64_t> sized_slots;
if (native_pack && o.expert_cache > 0 && !profile.empty() && !o.expert_cache_per_layer) {
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
const auto& lay = strata::kernels::cpu::expert_layout();
const uint64_t budget = (uint64_t) o.expert_cache * lay.max_blob; // what the uniform sizing granted
uint64_t used = 0;
size_t free_room = free_b > ((size_t) o.vram_reserve_mib << 20) ? free_b - ((size_t) o.vram_reserve_mib << 20) : 0;
const uint64_t cap = std::min<uint64_t>(budget, (uint64_t) free_room);
for (const auto& pr : profile) {
const uint64_t b = (lay.blob_bytes(pr.first) + 255) / 256 * 256;
if (used + b > cap) break;
used += b;
sized_slots.push_back((int64_t) lay.blob_bytes(pr.first));
}
o.expert_cache = (int) sized_slots.size();
}
if (o.expert_cache > 0) {
// keep the first `keep_bytes` of the cache (the profile's hottest experts first); false when nothing is left
auto shrink_to = [&](int64_t keep_bytes) -> bool {
if (keep_bytes <= 0) { o.expert_cache = 0; sized_slots.clear(); return false; }
if (!sized_slots.empty()) {
int64_t used = 0;
size_t keep = 0;
while (keep < sized_slots.size() && used + (sized_slots[keep] + 255) / 256 * 256 <= keep_bytes)
used += (sized_slots[keep++] + 255) / 256 * 256;
sized_slots.resize(keep);
o.expert_cache = (int) keep;
} else {
o.expert_cache = (int) (keep_bytes / (int64_t) strata::kernels::cpu::expert_layout().max_blob);
}
if (o.expert_cache <= 0) { o.expert_cache = 0; sized_slots.clear(); return false; }
return true;
};
auto cache_bytes = [&]() -> int64_t {
if (sized_slots.empty()) return (int64_t) o.expert_cache * (int64_t) strata::kernels::cpu::expert_layout().max_blob;
int64_t b = 0;
for (const int64_t s : sized_slots) b += (s + 255) / 256 * 256;
return b;
};
// With `--expert-cache auto` the reserve must still be free once the slots are WRITTEN: under WDDM an
// allocation is not resident until it is touched, and the free figure read before it can be ~1 GB too
// high. A cache sized from it filled the card to 0 MiB, the driver then paged, and a request that needed a
// page back while the verify graph spun on a host flag never finished. So the slots are zeroed and the
// free figure read again; while it is short of the reserve the cache is reopened smaller.
// STRATA_TEST_CACHE_FAIL=N: the first N opens fail as an out-of-commit cudaMalloc does (tests the retry)
int fake_fails = std::getenv("STRATA_TEST_CACHE_FAIL") ? std::atoi(std::getenv("STRATA_TEST_CACHE_FAIL")) : 0;
int failed = 0;
int zero_reads = 0;
for (int attempt = 0;; ++attempt) {
bool ok = false;
if (fake_fails > 0) {
--fake_fails;
err = "ExpertCache: cudaMalloc failed: out of memory (STRATA_TEST_CACHE_FAIL)";
} else {
ok = sized_slots.empty()
? xcache.open(o.expert_cache, g.n_layers, g.n_expert, (int64_t) strata::kernels::cpu::expert_layout().max_blob, err)
: xcache.open_sized(sized_slots, g.n_layers, g.n_expert, err);
}
if (!ok) {
// Issue #60: on Windows a device allocation is also charged to the system commit (RAM + page file),
// so with a small page file the cache's one big cudaMalloc fails while the VRAM is free. An auto
// cache then tries three quarters of the size, a few times, instead of stopping the engine.
char commit[96] = "";
#if defined(_WIN32)
MEMORYSTATUSEX ms{};
ms.dwLength = sizeof ms;
if (GlobalMemoryStatusEx(&ms))
std::snprintf(commit, sizeof commit, " (Windows has %.1f GiB of commit left: RAM + page file)",
(double) ms.ullAvailPageFile / 1073741824.0);
#endif
if (auto_cache && failed < 8 && shrink_to(cache_bytes() / 4 * 3)) {
++failed;
std::fprintf(stderr, "strata generate: %s%s; trying a smaller expert cache: %d slots\n", err.c_str(),
commit, o.expert_cache);
continue;
}
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
#if defined(_WIN32)
std::fprintf(stderr, "strata generate: on Windows the graphics card's memory also needs room in the page "
"file: set it to \"System managed\" (System > About > Advanced system settings > "
"Performance > Advanced > Virtual memory), or lower --expert-cache\n");
#endif
return 1;
}
if (!auto_cache || attempt - failed >= 6) break;
cudaMemset(xcache.device_slot(0), 0, (size_t) xcache.bytes());
cudaDeviceSynchronize();
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
const int64_t want = (int64_t) o.vram_reserve_mib << 20;
if ((int64_t) free_b >= want - (64ll << 20)) break;
// short by (want - free); a figure of 0 only says "at least": the first two such reads give back 1 GiB
// each (under WDDM the free figure read before the allocation runs ~0.7 GiB high), later ones a quarter
int64_t give = want - (int64_t) free_b + (64ll << 20);
if (free_b < ((size_t) 16 << 20))
give = std::max<int64_t>(give, ++zero_reads <= 2 ? 1ll << 30 : xcache.bytes() / 4);
const int64_t keep_bytes = xcache.bytes() - give;
std::fprintf(stderr, "strata generate: only %lld MiB free once the slots are written (reserve %d MiB); "
"shrinking the expert cache\n", (long long) (free_b >> 20), o.vram_reserve_mib);
xcache.close();
if (!shrink_to(keep_bytes)) break;
}
if (failed > 0 && o.expert_cache > 0)
std::fprintf(stderr, "strata generate: expert cache: %d slots (%.2f GiB) after %d smaller tries - a bigger "
"page file lets it use more of the free VRAM\n",
o.expert_cache, (double) xcache.bytes() / 1073741824.0, failed);
}
if (o.expert_cache > 0) {
std::fprintf(stderr, "strata generate: expert cache %lld slots, %.2f GiB of VRAM; policy is\n",
(long long) xcache.slots(), xcache.gib());
mem_mark("opening the expert cache");
xcache.set_per_layer_admission(o.expert_cache_per_layer);
// Round 328 warned here that the GPU hit path was wrong (tokens diverged from a cache-off run from
// token 0). That fault was fixed long since (native_expert_parity, expert_parity, the grouped kernels'
// tests), and the warning outlived it (issue #23). What remains is rounding: a GPU expert and the CPU's
// compute the same quantized expert with different float order, so a near-tie can flip. Measured teacher-
// forced on 2,557 tokens (bench/results/2026-09-27-cache-parity): 95-98% same top-1, and perplexity equal
// (on - off = -0.005 +- 0.005 nats). Neither output is more correct than the other.
std::fprintf(stderr,
"strata generate: the GPU computes the experts in the cache; it rounds differently from the CPU,\n"
" so a reply can differ slightly from a run without the cache (same quality:\n"
" bench/results/2026-09-27-cache-parity).\n");
if (o.expert_cache_per_layer) {
int64_t lo = 0, hi = 0;
xcache.layer_slot_range(0, lo, hi);
std::fprintf(stderr, " R4.2g PER-LAYER: each layer owns %lld slots (%lld..%lld).\n",
(long long) (hi - lo), (long long) lo, (long long) (hi - 1));
} else if (profile.empty()) {
std::fprintf(stderr, " compulsory-miss (fills with whatever the run routes first).\n");
} else {
std::fprintf(stderr, " PROFILE, ranked by routing frequency, no eviction.\n");
}
}
// ---- R4.2e: fill the tier from the profile. This is the only place the plan is applied, and it runs
// ONCE: with `slots` pairs and `slots` slots the cache is full when this returns, so the decode-time
// admission finds no room and every non-profiled expert stays a CPU miss. That is what makes the profile
// the policy rather than a hint.
int64_t prefilled = 0;
if (!profile.empty() && srcp != nullptr) {
// #369 (dag08): per layer, a full layer skips only its own pairs - each layer takes its hottest experts until
// its range is full (one full layer used to end the whole fill, leaving most layers empty)
const bool per_layer = xcache.per_layer_admission();
const int64_t want = per_layer ? (int64_t) profile.size()
: std::min<int64_t>((int64_t) profile.size(), xcache.slots());
// #286: an unbuffered file tier reads the pairs in batches of 64, the next batch while this one is copied
std::future<void> ahead;
auto read_batch = [&](int64_t at) { src.prefetch_pairs(profile.data() + at, std::min<int64_t>(64, want - at)); };
for (int64_t i = 0; i < want; ++i) {
if (!per_layer && srcp == &src && src.unbuffered() && i % 64 == 0) {
if (ahead.valid()) ahead.get();
else read_batch(i);
if (i + 64 < want) ahead = std::async(std::launch::async, read_batch, i + 64);
}
const int32_t slot = xcache.admit(profile[(size_t) i].first, profile[(size_t) i].second);
if (slot == strata::core::kNotResident) {
if (per_layer) continue;
break;
}
const uint8_t* b = srcp->blob(profile[(size_t) i].first, profile[(size_t) i].second);
if (b == nullptr || !xcache.fill_slot_blocking(slot, b, err,
(int64_t) strata::kernels::cpu::expert_layout().blob_bytes(profile[(size_t) i].first))) {
std::fprintf(stderr, "strata generate: the profile fill failed at pair %lld: %s\n",
(long long) i, err.c_str());
return 1;
}
++prefilled;
}
// **AND ONE SLOT IS READ BACK AND COMPARED.** A residency table that is right about indices and wrong
// about bytes produces a plausible token, which is this project's most expensive failure mode; the
// cache's own `verify_slot` is the check and it costs one 1.38 MB D2H at startup.
if (prefilled > 0 && !xcache.verify_slot(xcache.slot_of(profile[0].first, profile[0].second),
srcp->blob(profile[0].first, profile[0].second), err,
(int64_t) strata::kernels::cpu::expert_layout().blob_bytes(profile[0].first))) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
mem_mark("the profile fill");
std::fprintf(stderr, "strata generate: pre-filled %lld of %lld slots from the profile; slot 0 verified\n",
(long long) prefilled, (long long) (per_layer ? xcache.slots() : want));
}
for (auto& stp : stages) {
GpuStage& st = *stp;
const auto& lay = strata::kernels::cpu::expert_layout();
// the drafter and the head are already allocated by now (they load above, before this), so what is left
// to hold back is the windows - and `free_b` has already lost the drafter.
const int64_t room = stage_room(st.dev, true, false);
const strata::core::OnDevice on(st.dev);
std::vector<int64_t> sized;
int64_t used = 0;
for (const auto& pr : st.profile) {
const int64_t b = native_pack ? ((int64_t) lay.blob_bytes(pr.first) + 255) / 256 * 256 : (int64_t) lay.max_blob;
if (used + b > room) break;
used += b;
sized.push_back((int64_t) lay.blob_bytes(pr.first));
}
if (sized.empty() ||
!(native_pack ? st.cache.open_sized(sized, g.n_layers, g.n_expert, err)
: st.cache.open((int64_t) sized.size(), g.n_layers, g.n_expert, (int64_t) lay.max_blob, err))) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d expert cache: %s\n", st.dev,
sized.empty() ? "no room" : err.c_str());
return 1;
}
int64_t filled = 0;
for (const auto& pr : st.profile) {
if (filled >= st.cache.slots()) break;
const int32_t slot = st.cache.admit(pr.first, pr.second);
if (slot == strata::core::kNotResident) break;
const uint8_t* b = srcp->blob(pr.first, pr.second);
if (b == nullptr || !st.cache.fill_slot_blocking(slot, b, err, (int64_t) lay.blob_bytes(pr.first))) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d profile fill failed at pair %lld: %s\n",
st.dev, (long long) filled, err.c_str());
return 1;
}
++filled;
}
if (filled == 0 || !st.cache.verify_slot(st.cache.slot_of(st.profile[0].first, st.profile[0].second),
srcp->blob(st.profile[0].first, st.profile[0].second), err,
(int64_t) lay.blob_bytes(st.profile[0].first))) {
std::fprintf(stderr, "strata generate: layer split, CUDA%d expert cache: %s\n", st.dev,
filled == 0 ? "nothing filled" : err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: layer split: CUDA%d runs layers %lld-%lld, expert cache %lld slots "
"(%.2f GiB), %lld of its %zu profiled pairs; slot 0 verified\n",
st.dev, (long long) st.lb, (long long) (st.le - 1), (long long) st.cache.slots(), st.cache.gib(),
(long long) filled, st.profile.size());
}
if (multi_gpu)
std::fprintf(stderr, "strata generate: layer split: CUDA0 runs layers 0-%lld\n", (long long) (split_at[0] - 1));
std::array<strata::core::RemoteExperts, 3> remote_experts;
const bool multi_remote = o.expert_cache_remote[1] > 0 || o.expert_cache_remote[2] > 0;
if (o.expert_cache_remote[0] > 0) {
if (o.expert_cache <= 0 || profile.empty() || o.no_pool) {
std::fprintf(stderr, "strata generate: remote experts need --expert-profile, "
"a CUDA0 expert cache and the expert pool\n");
return 2;
}
std::vector<std::pair<int32_t, int32_t>> ranked = profile;
if (!stages.empty()) { // a layer split: CUDA0's share of the profile, then the later stages' pairs no cache holds
for (auto& st : stages)
for (const auto& pr : st->profile)
if (st->cache.slot_of(pr.first, pr.second) < 0) ranked.push_back(pr);
}
if (multi_remote) {
// The shipped frequency profile names only 8000 of 24576 experts. Once exhausted,
// fill remaining VRAM from unranked pairs in expert-then-layer order: this spreads
// the tail across all layers instead of concentrating it on layer zero.
std::vector<uint8_t> seen((size_t) g.n_layers * (size_t) g.n_expert, 0);
for (const auto& pair : ranked)
if (pair.first >= 0 && pair.first < g.n_layers && pair.second >= 0 && pair.second < g.n_expert)
seen[(size_t) pair.first * (size_t) g.n_expert + (size_t) pair.second] = 1;
for (int64_t e = 0; e < g.n_expert; ++e)
for (int64_t l = 0; l < g.n_layers; ++l)
if (!seen[(size_t) l * (size_t) g.n_expert + (size_t) e])
ranked.emplace_back((int32_t) l, (int32_t) e);
std::fprintf(stderr, "strata generate: remote ranking: %zu profiled pairs, "
"%zu other pairs to fill CUDA1..3\n", profile.size(), ranked.size() - profile.size());
}
std::array<std::vector<std::pair<int32_t, int32_t>>, 3> by_device;
if (multi_remote) {
// Either stripe experts for parallel GPU work, or give each layer one
// secondary GPU to reduce switches and transfers over shared USB4.
std::vector<uint8_t> assigned((size_t) g.n_layers * (size_t) g.n_expert, 0);
const int devices = 1 + (o.expert_cache_remote[1] > 0) + (o.expert_cache_remote[2] > 0);
int next = 0;
for (const auto& pair : ranked) {
if (pair.first < 0 || pair.first >= g.n_layers || pair.second < 0 || pair.second >= g.n_expert ||
xcache.slot_of(pair.first, pair.second) >= 0) continue;
const size_t index = (size_t) pair.first * (size_t) g.n_expert + (size_t) pair.second;
if (assigned[index]) continue;
int target = -1;
if (o.expert_cache_remote_placement == "layer") {
target = pair.first % devices;
// Other layers' owners may still have room: keep scanning ranks.
if (by_device[(size_t) target].size() >=
(size_t) o.expert_cache_remote[(size_t) target]) continue;
} else {
for (int i = 0; i < devices; ++i) {
const int r = (next + i) % devices;
if (by_device[(size_t) r].size() < (size_t) o.expert_cache_remote[(size_t) r]) {
target = r;
break;
}
}
if (target < 0) break;
}
assigned[index] = 1;
by_device[(size_t) target].push_back(pair);
next = (target + 1) % devices;
}
std::fprintf(stderr, "strata generate: remote ranks %s across %d CUDA devices\n",
o.expert_cache_remote_placement == "layer" ? "grouped by layer" : "striped", devices);
} else {
by_device[0] = std::move(ranked);
}
std::vector<uint8_t> claimed((size_t) g.n_layers * (size_t) g.n_expert, 0);
for (auto& st : stages) // a layer split: what a stage's cache holds is no helper's
for (const auto& pr : st->profile)
if (st->cache.slot_of(pr.first, pr.second) >= 0)
claimed[(size_t) pr.first * (size_t) g.n_expert + (size_t) pr.second] = 1;
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0) {
if (!remote_experts[(size_t) r].open(remote_dev[r], o.expert_cache_remote[(size_t) r],
g.n_layers, g.n_expert, by_device[(size_t) r], xcache, *srcp, claimed, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: CUDA%d: %lld additional experts, %.2f GiB; "
"results return through pinned host rows\n", remote_dev[r],
(long long) remote_experts[(size_t) r].resident(), remote_experts[(size_t) r].gib());
}
}
// ---- Multi-GPU: the second GPU's expert tier, filled with the ranked pairs the primary does not hold
strata::core::PeerExperts peer;
if (o.peer_device >= 1) {
if (profile.empty() || srcp == nullptr || o.expert_cache <= 0) {
std::fprintf(stderr, "strata generate: --peer-device needs --expert-profile and the expert cache\n");
return 1;
}
const auto tp0 = Clock::now();
if (!peer.open(o.peer_device, profile, xcache, *srcp, g.n_layers, g.n_expert, o.peer_reserve_mib, o.peer_slots,
err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: peer GPU %d: %lld experts, %.2f GiB (filled in %.1f s); with the primary's "
"%lld that is %lld of %lld on the GPUs\n", o.peer_device, (long long) peer.resident(), peer.gib(),
std::chrono::duration<double>(Clock::now() - tp0).count(), (long long) xcache.slots(),
(long long) (peer.resident() + xcache.slots()), (long long) (g.n_layers * g.n_expert));
}
Drive drive;
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0)
drive.d.remote[drive.d.remote_count++] = &remote_experts[(size_t) r];
drive.d.peer = peer.valid() ? &peer : nullptr;
drive.d.hit_cpu_order = o.expert_cache_cpu_order;
drive.d.split_rows = !o.no_split_rows;
drive.d.pool = &pool;
drive.d.src = srcp;
drive.d.n_expert = g.n_expert;
drive.d.jobs.resize((size_t) K);
// CS-T: routing-aware prefetch of the file tier (the GGUF in place): the next layer's router on this layer's MoE
// input predicts its experts and their pages are warmed meanwhile. It only warms pages; STRATA_LOOKAHEAD=0 is
// the A/B arm, STRATA_LOOKAHEAD_K the experts per token (default 10).
strata::core::RouterLookahead lookahead;
if (srcp == &src && src.warms() && [] { const char* v = std::getenv("STRATA_LOOKAHEAD"); return v == nullptr || std::atoi(v) != 0; }()) {
std::vector<std::vector<uint16_t>> routers((size_t) g.n_layers);
bool ok = true;
for (int64_t l = 0; l < g.n_layers && ok; ++l) {
const strata::core::WeightRef* w = wt.find("blk." + std::to_string(l) + ".ffn_gate_inp.weight");
ok = w != nullptr && w->kind == strata::core::WeightKind::Bf16InF32 &&
w->bytes == (uint64_t) (g.n_expert * g.n_embd) * 2;
if (!ok) break;
routers[(size_t) l].resize((size_t) (g.n_expert * g.n_embd));
ok = cudaMemcpy(routers[(size_t) l].data(), w->data, (size_t) w->bytes, cudaMemcpyDeviceToHost) == cudaSuccess;
}
const char* kv = std::getenv("STRATA_LOOKAHEAD_K");
if (ok && lookahead.start(std::move(routers), g.n_embd, g.n_expert, kv ? std::atoi(kv) : 10, &src, err)) {
drive.d.lookahead = &lookahead;
std::fprintf(stderr, "strata generate: routing-aware prefetch of the file tier on (the next layer's router)\n");
} else {
(void) cudaGetLastError();
std::fprintf(stderr, "strata generate: routing-aware prefetch off (%s)\n",
ok ? err.c_str() : "the routers are not BF16 in the arena");
err.clear();
}
}
// ---- R4.2c: THE HIT PATH. Every one of these is required for `hits_ready()`, which is all-or-nothing on
// purpose: a half-configured hit path would compute some experts twice and others not at all, and a token
// built on that is wrong rather than refused.
void* hit_scratch = nullptr;
int32_t* d_hit_slot = nullptr;
int32_t* d_hit_dst = nullptr;
uint8_t* d_hit_q8 = nullptr;
float* d_hit_q8_scale = nullptr; ///< R4.2h: the fp32 activation scales the CPU path also uses
float* d_hit_out = nullptr;
if (o.expert_cache > 0 && !o.no_pool) {
const uint64_t sb = strata::kernels::moe_hit_grouped_scratch_bytes(K, g.n_embd, strata::kernels::cpu::FF);
if (cudaMalloc(&hit_scratch, (size_t) sb) != cudaSuccess ||
cudaMalloc((void**) &d_hit_slot, (size_t) K * sizeof(int32_t)) != cudaSuccess ||
cudaMalloc((void**) &d_hit_dst, (size_t) K * sizeof(int32_t)) != cudaSuccess ||
cudaMalloc((void**) &d_hit_q8, (size_t) (g.n_embd / 32) * 34) != cudaSuccess ||
// R4.2h: the fp32 activation scales. Without this the GPU's hits use the block's fp16 `d`
// while the CPU's misses use `ActQ::scale`, which is fp32 - a 4.761e-04 relative disagreement on
// every chunk, and the reason enabling the cache changed the tokens.
cudaMalloc((void**) &d_hit_q8_scale, (size_t) (g.n_embd / 32) * sizeof(float)) != cudaSuccess ||
cudaMalloc((void**) &d_hit_out, (size_t) K * g.n_embd * 4) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the R4 hit path could not allocate its device buffers\n");
return 1;
}
drive.d.cache = &xcache;
drive.d.cache_stream = main_cs;
drive.d.cache_base = (const uint8_t*) xcache.device_slot(0);
drive.d.cache_blob = (int64_t) strata::kernels::cpu::expert_layout().max_blob;
drive.d.cache_slot_off = xcache.slot_offsets();
drive.d.hit_scratch = hit_scratch;
drive.d.parts_out = d_parts;
drive.d.hit_out = d_hit_out;
drive.d.parts_elems = K * g.n_embd;
drive.d.mixed = ss.block.mixed;
drive.d.x_q8_0_hit = d_hit_q8;
drive.d.x_q8_0_hit_scale = d_hit_q8_scale;
drive.d.d_slot = d_hit_slot;
drive.d.d_dst = d_hit_dst;
drive.d.h_slot.resize((size_t) K);
cudaEvent_t hit_done = nullptr;
if (cudaEventCreate(&hit_done) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the hit path could not create its probe event\n");
return 1;
}
drive.d.hit_done = (void*) hit_done;
drive.d.hit_poke = !o.no_hit_poke;
drive.d.h_dst.resize((size_t) K);
mem_mark("the R4 hit path");
std::fprintf(stderr, "strata generate: R4 hit path ON - resident experts are computed on the GPU\n");
}
// ---- P0.S8: the routing trace. Only meaningful with the pool running, because the ids arrive through
// the doorbell that the pool consumes - so `--no-pool` is refused rather than silently producing an empty
// file that would read as "the router selected nothing".
// ---- PER-STAGE TIMING. `--no-capture` only: an event recorded inside a stream capture is silently
// dropped, so a captured graph cannot carry these events and the numbers would be zeros that read as
// "every stage is free". Refusing is the fix.
if (o.stage_timing) {
if (!o.no_capture) {
std::fprintf(stderr, "strata generate: --stage-timing records CUDA events inside the layer path, "
"and an event record inside a stream capture is silently dropped. Pass "
"--no-capture as well.\n");
return 2;
}
if (!strata::core::stage_timing_enable()) {
std::fprintf(stderr, "strata generate: stage_timing_enable failed\n");
return 1;
}
strata::core::stage_timing_name(0, "gr_read (attn)");
strata::core::stage_timing_name(1, "attention block");
strata::core::stage_timing_name(2, "gr_write (attn)");
strata::core::stage_timing_name(3, "gr_read (ffn)");
strata::core::stage_timing_name(4, "moe_route");
strata::core::stage_timing_name(5, "moe_finish");
strata::core::stage_timing_name(6, "gr_write (ffn)");
// The GDN block's internals. It is 36 of the 48 layers, 0.96 ms each, and its entire weight traffic
// is ~26 MB - so ~0.11 ms at the measured read rate. ~13 tiny latency-bound launches live in it and
// a single "attention block" number cannot say which one costs anything.
strata::core::stage_timing_name(8, " gdn: quantize x");
strata::core::stage_timing_name(9, " gdn: qkv gemv");
strata::core::stage_timing_name(10, " gdn: conv+silu");
strata::core::stage_timing_name(11, " gdn: l2 norms");
strata::core::stage_timing_name(12, " gdn: alpha/beta/gate");
strata::core::stage_timing_name(13, " gdn: gdn_step");
strata::core::stage_timing_name(14, " gdn: z + out_norm");
strata::core::stage_timing_name(15, " gdn: out gemv");
}
std::FILE* routing = nullptr;
if (!o.dump_routing.empty()) {
if (o.no_pool) {
std::fprintf(stderr, "strata generate: --dump-routing needs the expert pool; the routed ids reach "
"the host through the doorbell the pool reads. Drop --no-pool.\n");
return 2;
}
routing = std::fopen(o.dump_routing.c_str(), "wb");
if (routing == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_routing.c_str());
return 1;
}
drive.routing = routing;
}
strata::core::PoolFn pool_fn = o.no_pool ? nullptr : &drive_pool;
// The hit hook rides the same switch as the pool: with no pool there is no `parts` staging to
// write into, and a hit path with nowhere to write is a wrong token rather than an error.
strata::core::HitFn hit_fn =
(o.no_pool || o.expert_cache <= 0) ? nullptr : &strata::core::expert_hit_run;
void* pool_user = o.no_pool ? nullptr : (void*) &drive;
std::fprintf(stderr, "strata generate: %d expert-pool workers%s%s\n", pool.workers(),
pool.host_works() ? " + the host thread" : "",
o.no_pool ? " (UNUSED: --no-pool)" : "");
// **THE MISALIGNMENT WARNING THAT STOOD HERE IS GONE, BECAUSE THE MISALIGNMENT IS FIXED.**
//
// It said the tokens were not the model's, and it was true: `session_loop` handed layer `l`'s expert
// outputs to layer `l+1`, which multiplied them by layer `l+1`'s router weights (LEDGER L100). The loop
// now runs a captured PAIR per layer - `pre[l]` ending with the router and the doorbell, then the CPU
// pool, then `post[l]` which combines those experts with THAT layer's weights - so layer `l`'s experts meet
// layer `l`'s routing. The generated ids changed the moment it landed, which is what a correctness fix
// looks like from the outside.
//
// The cost is real and is recorded rather than hidden: the window for the CPU pool is now whatever GPU
// work follows the ring inside `pre[l]`, which is the shared expert and nothing else - 0.038 ms against
// 0.514 ms of CPU work per layer. A per-layer CPU expert pool cannot be hidden behind a strictly serial
// residual chain; the CPU term is answered by Phase 3's VRAM expert cache, not by this pipeline.
// ---- the graphs
strata::core::SessionGraphs gr;
if (!o.no_capture && !native_pack) { // plan v0.3 P6: a native pack runs verify windows only
// a layer split's CUDA0 session owns only [0, split_at[0]), so its graphs cover that range; the
// whole-model replay paths (`session_loop`, the plain generate loop) refuse rather than read another
// stage's state - a split runs its layers on the stages' verifiers (serve) or prefill stage chain
if (!strata::core::session_capture(wt, g, ss, d_parts, gr, err, /*split=*/o.gpu_stages, 0,
multi_gpu ? split_at[0] : -1)) {
std::fprintf(stderr, "strata generate: session_capture: %s\n", err.c_str());
return 1;
}
}
// **`--no-capture` AND THE EXPERTS ARE MUTUALLY EXCLUSIVE, AND SILENTLY SO.**
//
// The CPU expert pool is wired into `session_loop` - the host loop around the captured graphs - and
// `session_token` has no pool hook at all. So `--no-capture` did not merely change HOW the layers were
// launched: it ran the whole model with `parts` left at whatever the buffer held, which is ZERO, and the
// only symptom was `expert blobs 0` in a stats line nobody had to read. A run that silently omits the
// routed experts is not a slow measurement of this model, it is a measurement of a different model.
//
// Refusing is the fix. `--no-pool` is the explicit way to say "I want the GPU-only floor".
if (o.no_capture && !o.no_pool) {
std::fprintf(stderr,
"strata generate: --no-capture runs `session_token`, which has NO CPU expert pool hook, so "
"the routed experts would silently contribute nothing. Pass --no-pool as well if the "
"GPU-only floor is what you want.\n");
return 2;
}
// The ladder is written by `session_loop`, and `session_token` does not touch the staging buffer at all - so
// accepting the flag there would produce a file of uninitialised memory, which reads as a wrong answer rather
// than as a mistake. `--no-capture` without `--no-pool` is already refused above, so this catches the pair.
if (o.no_capture && !o.dump_layers.empty()) {
std::fprintf(stderr,
"strata generate: --dump-layers is written by `session_loop`; `--no-capture` runs "
"`session_token` instead, which never fills the staging buffer. Drop one of the two.\n");
return 2;
}
if (o.no_capture && !o.dump_halves.empty()) {
std::fprintf(stderr,
"strata generate: --dump-halves is CAPTURED into the layer graphs, so it needs the "
"captured path; `--no-capture` never records it. Drop one of the two.\n");
return 2;
}
mem_mark("the expert cache and the graphs");
std::fprintf(stderr, "strata generate: session is up (engine %s)\n", STRATA_VERSION);
auto run_head = [&](void* stream) -> bool {
if (!native_head.loaded())
return strata::core::lm_head(wt, g, ss.block, d_logits, stream, err);
return strata::core::lm_head_mix(wt, g, ss.block, stream, err) &&
native_head.run(ss.block.mixed, d_logits, stream, err);
};
float* d_emb = nullptr;
if (cudaMalloc(&d_emb, (size_t) g.n_embd * 4) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the embedding buffer failed\n");
return 1;
}
// **`sample_tokens` TAKES DEVICE POINTERS.** It is a kernel launch; `logits` and `out` are both read and
// written on the device. Passing `logits.data()` - the host vector - faults inside the kernel and the
// error surfaces at the NEXT synchronising call, which here was the next token's `embed_row`, reporting an
// illegal access on a weight plane. Nothing in the parameter names said device.
int* d_next = nullptr;
if (cudaMalloc(&d_next, sizeof(int)) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the sampler output buffer failed\n");
return 1;
}
// **`R` IS BOTH THE INPUT AND THE OUTPUT, SO THE NEW TOKEN'S EMBEDDING HAS TO REPLACE THE OLD RESIDUAL.**
// At `pos == 0` that is `session_zero`, which is the reference's own initial condition - the embedding
// broadcast to all `hc` streams. After that `session_zero` would also wipe the recurrence, so the
// broadcast is done directly. Getting this wrong is invisible for exactly one token.
void* token_stream = o.stream_token ? main_cs : nullptr;
auto put_input = [&](int64_t tok, int64_t pos) -> bool {
if (!strata::core::embed_row(wt, g, tok, d_emb, token_stream, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return false;
}
if (pos == 0) {
strata::core::session_zero(ss, g, d_emb, token_stream);
} else {
for (int64_t c = 0; c < g.hc; ++c)
if (cudaMemcpyAsync(ss.R + (size_t) c * g.n_embd, d_emb, (size_t) g.n_embd * 4,
cudaMemcpyDeviceToDevice, (cudaStream_t) token_stream) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the residual broadcast failed\n");
return false;
}
}
return o.stream_token || cudaDeviceSynchronize() == cudaSuccess;
};
strata::kernels::SamplerParams sp;
sp.greedy = o.greedy;
sp.seed = o.seed;
sp.top_k = o.top_k;
sp.top_p = o.top_p;
sp.temperature = o.temperature;
// what this run actually samples with (the speculative loop below gets the same parameters); serve samples
// per request instead
if (!o.serve) {
if (sp.greedy || sp.temperature <= 0.0f)
std::fprintf(stderr, "strata generate: sampling greedy\n");
else
std::fprintf(stderr, "strata generate: sampling temperature=%g top_k=%d top_p=%g seed=%llu\n",
(double) sp.temperature, sp.top_k > 0 && sp.top_k < 64 ? sp.top_k : 64, (double) sp.top_p,
(unsigned long long) sp.seed);
}
std::FILE* dump = nullptr;
// The logits header is written WITH THE FIRST ROW, not at open: a native pack never reaches the
// per-token dump site, and a header promising rows that were never written is worse than no file.
int32_t hdr[2] = {0, 0};
bool hdr_written = false;
const int64_t dump_positions = (int64_t) o.tokens.size() - 1 + o.max_new;
if (!o.dump_logits.empty()) {
if (dump_positions > INT32_MAX || n_vocab > INT32_MAX) {
std::fprintf(stderr, "strata generate: logits dump dimensions exceed int32\n");
return 2;
}
dump = std::fopen(o.dump_logits.c_str(), "wb");
if (dump == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_logits.c_str());
return 1;
}
// **THE COUNT IS `n_prompt - 1 + max_new`, NOT `n_prompt + max_new`.** The loop writes one row per
// position from 0, and it stops once `produced` holds `max_new` tokens - and `produced` only starts
// receiving at position `n_prompt - 1`. So a 5-token prompt with `--max-new 6` writes 10 rows, and the
// header used to claim 11. A header that describes a different file from the one written is the same
// class of defect as a self-check that verifies the wrong invariant: anything reading the count instead
// of the size gets a wrong answer that looks authoritative. `tools/logits_identical.py` caught it by
// parsing the header and refusing the file.
const int32_t n_rows = (int32_t) strata::program::logits_selection::row_count(dump_positions, o.logits_stride);
hdr[0] = n_vocab; hdr[1] = n_rows;
}
// ---- THE C1 ORACLE: ONE RESIDUAL SNAPSHOT PER LAYER PER POSITION, so the engine can be bisected against
// `llama-debug`'s `l_last-<il>` node instead of against a single end-to-end perplexity. The buffer is
// PINNED because `session_loop` enqueues a device-to-host copy into it after every layer and the transfer
// would otherwise be staged through a pageable bounce buffer on the critical path.
std::FILE* layer_dump = nullptr;
float* layer_stage = nullptr;
const size_t layer_floats = (size_t) (g.n_layers + 1) * (size_t) g.hc * (size_t) g.n_embd;
if (!o.dump_layers.empty()) {
layer_dump = std::fopen(o.dump_layers.c_str(), "wb");
if (layer_dump == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_layers.c_str());
return 1;
}
if (cudaHostAlloc((void**) &layer_stage, layer_floats * sizeof(float), cudaHostAllocDefault) !=
cudaSuccess) {
std::fprintf(stderr, "strata generate: cannot pin the layer-dump staging buffer\n");
return 1;
}
}
// ---- prefill is the DECODE PATH ONE TOKEN AT A TIME, which `phase-2-correct-engine.md:12-13` says is
// fine here: "process the prompt through the decode-style graphs in small batches; a 19K-token prompt will
// take minutes". A real batched prefill is P2.S6's other half and is not this.
//
// **THE LOOP IS `feed -> 48 layers -> head -> sample -> feed`, AND THE FIRST GENERATED TOKEN COMES FROM THE
// LAST *PROMPT* POSITION.** The first version sampled only on the decode positions, so `produced` was
// still empty when the first generated position asked for `produced.back()` - an out-of-bounds read on an
// empty vector. Teacher forcing below is what makes the distinction unnecessary to special-case: for every
// position before the last prompt one, the next input is the PROMPT's next token, and after that it is the
// sampled one.
std::vector<int64_t> produced;
double total_ms = 0;
double prefill_ms = 0; // positions 0 .. n_prompt-2: prompt tokens that only condition
const Clock::time_point t_start = Clock::now();
double ttft_ms = 0;
const int64_t n_prompt = (int64_t) o.tokens.size();
int64_t tok = o.tokens[0];
// ---- THE PURE-GPU MEASUREMENT. `session_replay` launches all 48 `pre` graphs back to back on one stream
// with NO host work between them - no doorbell poll, no pool, no parts copy - so what it times is the GPU
// executing the layer sequence and nothing else. It had been declared, defined and never called since the
// day it was written.
//
// **THIS IS THE MEASUREMENT THAT SAYS WHETHER THE ENGINE IS HOST-BOUND OR GPU-BOUND**, and the stage table
// cannot answer it: those events measure the interval between two marks on a stream, which includes every
// gap where the GPU sat idle waiting for the host to enqueue the next kernel. In `--no-capture` those gaps
// are the host's launch latency and they are proportional to the KERNEL COUNT rather than to any work, so
// the no-capture stage shares are shares of kernel count - which is why the attention block, with the most
// kernels, looks like 55% of the token there.
// ---- R0.9: THE PER-STAGE TABLE ON THE CAPTURED GRAPH.
if (o.gpu_stages) {
strata::core::doorbell_reset(db);
double mix = 0, ffn = 0, post = 0;
if (!strata::core::session_replay_stages(g, 0, 0, ss, gr, main_cs, mix, ffn, post, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
const int reps = 20;
double t_mix = 0, t_ffn = 0, t_post = 0;
for (int r = 0; r < reps; ++r) {
if (!strata::core::session_replay_stages(g, 0, 0, ss, gr, main_cs, mix, ffn, post, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
t_mix += mix;
t_ffn += ffn;
t_post += post;
}
const double tot = t_mix + t_ffn + t_post;
std::printf("\nper-stage GPU time on the CAPTURED graph, one token over %lld layers\n",
(long long) g.n_layers);
std::printf(" %-22s %9.3f ms/token %8.3f ms/layer %6.1f%%\n", "mixer (gr_read+attn+gr_write)",
t_mix / reps, t_mix / reps / (double) g.n_layers, 100.0 * t_mix / tot);
std::printf(" %-22s %9.3f ms/token %8.3f ms/layer %6.1f%%\n", "ffn front + router",
t_ffn / reps, t_ffn / reps / (double) g.n_layers, 100.0 * t_ffn / tot);
std::printf(" %-22s %9.3f ms/token %8.3f ms/layer %6.1f%%\n", "post (moe_finish+gr_write)",
t_post / reps, t_post / reps / (double) g.n_layers, 100.0 * t_post / tot);
std::printf(" %-22s %9.3f ms/token\n", "sum of the three", tot / reps);
// ---- AND THE MIXER BY LAYER KIND, because 36 of the 48 are GDN and 12 are QSA and a total cannot
// separate them. Round 309's uncaptured table put GDN at 10.88 ms for 36 layers against QSA's 4.56 for
// 12, which would make the recurrence the largest single R3 target - and that table had `moe_finish`
// wrong by 4x, so the ratio is re-derived here from the captured graph rather than inherited.
{
// **ACCUMULATED OVER `reps`, NOT MEASURED ONCE AND THEN DIVIDED.** The first version called the
// per-layer replay a single time and printed `gdn / reps`, which reported GDN at 0.480 ms/token
// against a mixer total of 14.039 - a factor of exactly `reps`, and the tell was that
// 0.480 + 0.220 = 0.700 = 14.039 / 20.
std::vector<double> acc((size_t) g.n_layers, 0.0), per;
double f2 = 0, p2 = 0;
for (int r = 0; r < reps; ++r) {
if (!strata::core::session_replay_stages_per_layer(g, 0, 0, ss, gr, main_cs, per, f2, p2, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
for (int64_t l = 0; l < g.n_layers; ++l) acc[(size_t) l] += per[(size_t) l];
}
double gdn = 0, qsa = 0, worst = 0;
int64_t ng = 0, nq = 0, worst_l = 0;
for (int64_t l = 0; l < g.n_layers; ++l) {
const double v = acc[(size_t) l] / reps;
if (strata::core::is_qsa_layer(g, l)) { qsa += v; ++nq; }
else { gdn += v; ++ng; }
if (v > worst) { worst = v; worst_l = l; }
}
std::printf("\n the mixer by layer kind, averaged over %d runs\n", reps);
std::printf(" %-22s %9.3f ms/token %8.3f ms/layer (%lld layers)\n", "GDN layers",
gdn, ng ? gdn / (double) ng : 0.0, (long long) ng);
std::printf(" %-22s %9.3f ms/token %8.3f ms/layer (%lld layers)\n", "QSA layers",
qsa, nq ? qsa / (double) nq : 0.0, (long long) nq);
std::printf(" %-22s layer %lld at %.3f ms\n", "worst mixer layer", (long long) worst_l, worst);
std::printf(" %-22s %9.3f ms/token (must equal the mixer above)\n", "GDN + QSA", gdn + qsa);
}
// ================================ R0.11: THE FIVE STAGES SEPARATELY ================================
//
// Prefixes 1..5 are replayed per layer with the residual restored between them, and consecutive
// differences are the per-stage times. **THE CHECK IS THAT THE FIVE SUM TO THE THREE-GRAPH TOTAL** -
// the same independent-restatement test that caught round 320's divide-by-reps bug, and it is the only
// reason to believe a table built out of differences.
{
std::vector<double> acc5(5, 0.0), per, s5;
for (int r = 0; r < reps; ++r) {
if (!strata::core::session_replay_stage_prefixes(g, 0, 0, ss, gr, main_cs, s5, per, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
for (int k = 0; k < 5; ++k) acc5[(size_t) k] += s5[(size_t) k];
}
static const char* sn[5] = {"0 gr_read (attn)", "1 attention", "2 gr_write (attn)",
"3 gr_read (ffn)", "4 moe_route (router)"};
const char* kind[5] = {"GR", "ATTN", "GR", "GR", "ROUTER"};
double tot5 = 0, gr_ms = 0;
std::printf("\n the five stages separately, by differencing prefixes\n");
std::printf(" %-24s %-8s %10s %10s %8s\n", "stage", "kind", "ms/token", "ms/layer", "share");
for (int k = 0; k < 5; ++k) {
const double v = acc5[(size_t) k] / reps;
tot5 += v;
if (k != 1 && k != 4) gr_ms += v;
std::printf(" %-24s %-8s %10.3f %10.4f %7.1f%%\n", sn[k], kind[k], v,
v / (double) g.n_layers, 0.0);
}
for (int k = 0; k < 5; ++k) {
const double v = acc5[(size_t) k] / reps;
(void) v;
}
std::printf(" %-24s %-8s %10.3f\n", "sum of the five", "", tot5);
std::printf(" %-24s %-8s %10.3f <- R3.3's target is <= 3 ms for all four passes\n",
"GR passes (0,2,3)", "GR", gr_ms);
std::printf("\n the three-graph total above was %.3f ms/token; the five must account for it.\n",
tot / reps);
// ================================ R3.5c: THE SAME TABLE, A DIFFERENT WAY ================================
//
// The differencing table above mixes five graphs per layer, so stage 4's interval carries the launch
// of the FULL five-stage graph while stage 3's carries a four-stage one. This sweep launches ONE
// graph type per layer and nothing else, so that bias cannot exist. **If the two disagree, the
// difference IS the bias and this one is right** - and stage 4 is the router, so it is exactly the
// number that must not be wrong.
{
double sweep[6] = {0, 0, 0, 0, 0, 0};
for (int k = 1; k <= 5; ++k) {
double acc = 0, one = 0;
for (int r = 0; r < reps; ++r) {
if (!strata::core::session_replay_stage_sweep(g, 0, 0, ss, gr, main_cs, k, one, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
acc += one;
}
sweep[k] = acc / reps;
}
std::printf("\n the same five stages, by SWEEPING each prefix back to back (no graph switching)\n");
std::printf(" %-24s %10s %10s %12s\n", "stage", "prefix sweep", "difference", "bias");
static const char* sn2[5] = {"0 gr_read (attn)", "1 attention", "2 gr_write (attn)",
"3 gr_read (ffn)", "4 moe_route (router)"};
for (int k = 0; k < 5; ++k) {
const double sw = sweep[k + 1] - sweep[k];
const double df = acc5[(size_t) k] / reps;
std::printf(" %-24s %10.3f %10.3f %11.1f%%\n", sn2[k], sw, df,
sw != 0.0 ? 100.0 * (df / sw - 1.0) : 0.0);
}
std::printf(" %-24s %10.3f (full pre, 48 layers)\n", "prefix 5 total", sweep[5]);
}
}
std::printf("\n compare `--gpu-only-full`, which replays the same work as TWO graphs per layer. The\n");
std::printf(" three sum slightly above it because each launch carries the driver's gap.\n");
strata::core::session_graphs_free(gr);
strata::core::doorbell_free(db);
cudaFree(d_next);
return 0;
}
if (o.graph_only) {
strata::core::doorbell_reset(db);
// one warm pass so the first launch does not pay for page mapping
if (!strata::core::session_replay(g, 0, 0, ss, gr, main_cs, err)) {
std::fprintf(stderr, "strata generate: session_replay warm: %s\n", err.c_str());
return 1;
}
if (cudaDeviceSynchronize() != cudaSuccess) {
std::fprintf(stderr, "strata generate: session_replay warm faulted\n");
return 1;
}
const int reps = 20;
const Clock::time_point t0 = Clock::now();
for (int r = 0; r < reps; ++r) {
if (!strata::core::session_replay(g, 0, 0, ss, gr, main_cs, err)) {
std::fprintf(stderr, "strata generate: session_replay: %s\n", err.c_str());
return 1;
}
}
if (cudaDeviceSynchronize() != cudaSuccess) {
std::fprintf(stderr, "strata generate: session_replay faulted: %s\n", cudaGetErrorString(cudaGetLastError()));
return 1;
}
const double ms = std::chrono::duration<double, std::milli>(Clock::now() - t0).count() / (double) reps;
std::printf("pre graphs only %8.2f ms per token over %lld layers -> %.2f tok/s of GPU work\n", ms,
(long long) g.n_layers, ms > 0 ? 1000.0 / ms : 0.0);
std::printf(" %8.3f ms per layer\n", ms / (double) g.n_layers);
return 0;
}
// ---- R0.3: THE TRUE PER-TOKEN GPU FLOOR.
//
// `--graph-only` above launches ONLY `gr.execs[l]`, the `pre` graphs. It omits the 48 `post` graphs - the
// shared expert, the combine and the second `gr_write` - and the LM head. Everything this project published
// as "39.8 ms pure GPU" came from that loop while being described as the whole GPU, which also made the
// "host's share = 49.4 - 39.8 = 9.6 ms" figure wrong by however much the missing work costs. See
// Memory/ERRORS.md A4/A5.
//
// This flag is what "pure GPU" has to mean, and it replaces that number everywhere. No pool runs, so
// `parts` keeps whatever the buffer holds and the timing is GPU work alone.
if (o.gpu_only_full) {
strata::core::doorbell_reset(db);
const int reps = 20;
double ms_layers = 0, ms_head = 0;
for (int r = -1; r < reps; ++r) { // r == -1 is the warm pass, not counted
const Clock::time_point t0 = Clock::now();
if (!strata::core::session_replay_full(g, 0, 0, ss, gr, main_cs, err)) {
std::fprintf(stderr, "strata generate: session_replay_full: %s\n", err.c_str());
return 1;
}
// The sync is INSIDE the interval on purpose: it is the wait for the GPU, so t1 - t0 is GPU time.
if (cudaStreamSynchronize((cudaStream_t) main_cs) != cudaSuccess) {
std::fprintf(stderr, "strata generate: gpu-only-full layers faulted: %s\n",
cudaGetErrorString(cudaGetLastError()));
return 1;
}
const Clock::time_point t1 = Clock::now();
if (!run_head(main_cs)) {
std::fprintf(stderr, "strata generate: gpu-only-full lm_head: %s\n", err.c_str());
return 1;
}
if (cudaStreamSynchronize((cudaStream_t) main_cs) != cudaSuccess) {
std::fprintf(stderr, "strata generate: gpu-only-full head faulted: %s\n",
cudaGetErrorString(cudaGetLastError()));
return 1;
}
const Clock::time_point t2 = Clock::now();
if (r < 0) continue;
ms_layers += std::chrono::duration<double, std::milli>(t1 - t0).count();
ms_head += std::chrono::duration<double, std::milli>(t2 - t1).count();
}
ms_layers /= (double) reps;
ms_head /= (double) reps;
const double ms = ms_layers + ms_head;
std::printf("GPU floor pre+post+head %7.2f ms per token -> %.2f tok/s of GPU work\n", ms,
ms > 0 ? 1000.0 / ms : 0.0);
std::printf(" %8.3f ms layers (%lld x pre+post)\n", ms_layers, (long long) g.n_layers);
std::printf(" %8.3f ms per layer\n", ms_layers / (double) g.n_layers);
std::printf(" %8.3f ms LM head\n", ms_head);
return 0;
}
// ---- **THE TOKEN PATH ALLOCATES NOTHING (P2.T10, review finding H3).**
//
// `session_loop` used to allocate its pinned staging buffer, its probe event and its host pin ON EVERY
// TOKEN, and `cudaFreeHost` at the end of each call implicitly synchronises the device - so every token
// finished with a device-wide sync nobody asked for. The scratch is created once here and reused; it also
// owns the host pin for the whole session rather than taking and releasing it per token.
strata::core::SessionLoopScratch loop_scratch;
struct ScratchFree {
strata::core::SessionLoopScratch* p;
~ScratchFree() { if (p != nullptr) p->free(); }
} scratch_free{&loop_scratch};
// Initialised unconditionally, including under --no-pool: the loop validates the scratch it is handed, so
// passing a default-constructed one is an error rather than a fallback. (It was, and the guard caught it -
// which is the point of the guard.) One allocation at setup either way.
if (!loop_scratch.init((size_t) K * g.n_embd * 4, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
// Plan v0.3 P3: the whole token as ONE graph whenever nothing needs a host step between the ring and post[l]
// (the VRAM expert tier and the per-layer dumps do). `--no-token-graph` keeps two graphs per layer.
strata::core::TokenGraph tgraph;
struct TokenGraphFree {
strata::core::TokenGraph* p;
~TokenGraphFree() { strata::core::token_graph_free(*p); }
} tgraph_free{&tgraph};
// Plan v0.3 P4: with a PROFILE-filled cache the residency is static, so the hit decision moves onto the
// device and the token graph keeps it. (A cache filled on demand still needs the per-layer host path.)
std::vector<int32_t> host_res;
int32_t* d_res = nullptr;
int32_t* d_hit_count = nullptr;
strata::core::TokenHits thits;
const bool graph_hits = hit_fn != nullptr && !profile.empty() && !o.no_pool;
if (graph_hits && !o.no_capture && !o.no_token_graph && layer_dump == nullptr && half_dump == nullptr) {
host_res.assign((size_t) (g.n_layers * g.n_expert), strata::core::kNotResident);
int64_t resident = 0;
for (int64_t l = 0; l < g.n_layers; ++l)
for (int64_t e = 0; e < g.n_expert; ++e) {
const int st = multi_gpu ? stage_of(l) : 0;
const int32_t slot = st > 0 ? stages[(size_t) st - 1]->cache.slot_of(l, e) : xcache.slot_of(l, e);
host_res[(size_t) (l * g.n_expert + e)] = slot;
if (slot != strata::core::kNotResident) ++resident;
}
if (cudaMalloc((void**) &d_res, host_res.size() * sizeof(int32_t)) != cudaSuccess ||
cudaMalloc((void**) &d_hit_count, sizeof(int32_t)) != cudaSuccess ||
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice) != cudaSuccess) {
std::fprintf(stderr, "strata generate: the device residency table could not be staged\n");
return 1;
}
thits.d_res = d_res;
thits.n_expert = g.n_expert;
for (auto& st : stages) { // layer split across GPUs: the same table on every device
const strata::core::OnDevice on(st->dev);
if (cudaMalloc((void**) &st->d_res, host_res.size() * sizeof(int32_t)) != cudaSuccess ||
cudaMemcpy(st->d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice) !=
cudaSuccess) {
std::fprintf(stderr, "strata generate: layer split: CUDA%d residency table failed\n", st->dev);
return 1;
}
}
thits.cache_base = drive.d.cache_base;
thits.blob = drive.d.cache_blob;
thits.d_slot = drive.d.d_slot;
thits.d_dst = drive.d.d_dst;
thits.d_count = d_hit_count;
thits.x_q8 = drive.d.x_q8_0_hit;
thits.x_scale = drive.d.x_q8_0_hit_scale;
thits.scratch = drive.d.hit_scratch;
thits.hit_out = drive.d.hit_out;
drive.d.host_res = host_res.data();
std::fprintf(stderr, "strata generate: token graph hit path: %lld resident experts, decided on the device\n",
(long long) resident);
}
if (!o.no_capture && !o.no_token_graph && layer_dump == nullptr && half_dump == nullptr &&
(hit_fn == nullptr || thits.on()) && !native_pack && !multi_gpu) { // a split's token graph cannot span stages
if (!strata::core::session_capture_token(wt, g, ss, d_parts, loop_scratch.y_miss, loop_scratch.parts_bytes,
tgraph, err, thits.on() ? &thits : nullptr)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: token graph captured (48 layers, one launch per token)\n");
}
// ================================ WHERE THE HOST TERM GOES, PER TOKEN ================================
//
// **`--gpu-only-full` MEASURES THE 48 LAYER GRAPHS AND THE LM HEAD AND NOTHING ELSE.** It never enters
// this loop, so it does not run `ple_stage_token`, `embed_row`, the whole-vocabulary logits readback, the
// NaN scan or the sampler - and `--no-pool --stats` against that floor was being read as "the per-layer
// round trip costs 12.4 ms" when an unknown part of it is per TOKEN, not per layer. That is the same error
// the review catalogued as A4/A5, one level down: a difference between two measurements attributed to a
// mechanism that neither of them isolates.
//
// Six accumulators, because the six have different fixes. Reported in `--stats` as ms/token. **The
// boundary after the layer loop is the one that matters**: without it the head's interval swallows all 48
// layers and the report reads as "head = 50 ms", which is not a thing that can happen to 1.4 ms of GPU
// work. That is not hypothetical - it is what the first version of this printed.
double ms_ple = 0, ms_embed = 0, ms_layers = 0, ms_head = 0, ms_readback = 0, ms_sample = 0;
int64_t phase_tokens = 0;
// ================================ plan v0.3 P8: THE PERSISTENT ENGINE (--serve) ================================
//
// The weights, the expert arena and the VRAM tier load once; then requests arrive on stdin, one per line,
//
// GEN <max_new> <id,id,...>
//
// and each generated token is written to stdout as `T <id>` as soon as its verify window is done, followed by
//
// DONE <generated> <prompt_tokens> <prompt_ms> <decode_ms> <stop|length|cancel> <drafts accepted>
// <drafts offered> <prompt tokens reused> ... <prompt tokens read> (see the DONE line below; #471)
//
// Before that, `RESUME <n>` (n prompt tokens are not read again), `PP <position> <prompt_tokens> <ms> <tok/s>`
// after every prompt chunk, and `REUSED <n>` once the prompt is read. (`ERR <message>` instead when a request
// cannot run; `STOP` ends the running request at its next step; `QUIT` ends the process.) A request continues
// from the live session or the longest conversation checkpoint its prompt starts with (see ConvCheckpoint),
// otherwise from an empty sequence (`session_zero`); the rest of the prompt goes through the batched prompt path
// and its last token through the first verify window - the path all three model files share. Decoding is greedy.
// The expert-cache slots (from the end of the cache) that hold the prompt path's buffers for a chunk, and the
// bytes from the first of them to the end.
// the share of expert bytes the arena could pin (sizes the prompt path's streamed ring and its lend cap)
if (srcp != nullptr && o.prefill_chunk > 0) {
uint64_t pinned = 0, total = 0;
const auto& lay = strata::kernels::cpu::expert_layout();
for (int64_t l = 0; l < g.n_layers; ++l)
for (int64_t e = 0; e < g.n_expert; ++e) {
const uint64_t b = lay.blob_bytes(l);
total += b;
if (srcp->pinned(l, e)) pinned += b;
}
strata::prefill::Prefill::set_pinned_share(total ? (double) pinned / (double) total : 1.0);
}
auto lend_slots = [&](int64_t c) -> int64_t {
const uint64_t need = strata::prefill::Prefill::bytes_needed(g, ss, c);
const int64_t blob = (int64_t) strata::kernels::cpu::expert_layout().max_blob;
int64_t k = (int64_t) ((need + (uint64_t) blob - 1) / (uint64_t) blob);
if (xcache.slot_offsets() != nullptr) { // sized slots: take slots from the end until they hold `need`
k = 0;
while (k < xcache.slots() &&
(uint64_t) (xcache.bytes() - (int64_t) xcache.slot_offsets()[xcache.slots() - k]) < need) ++k;
}
return k;
};
// `lend_bytes` went with the single-cache serve loan: a participant's loan is priced by `part_bytes` from its
// OWN cache, and the only other user of the old helper was the serve path's own relayout.
auto request_chunk = [](int64_t tokens, int64_t max_chunk) -> int64_t {
if (tokens <= 0 || max_chunk <= 0) return 0;
const int64_t rounded = tokens > std::numeric_limits<int64_t>::max() - 255
? tokens
: ((tokens + 255) / 256) * 256;
return std::min(max_chunk, rounded);
};
// The prompt path's chunk and the slots it borrows for its buffers: the requested chunk halved until it fits,
// or with --prefill auto the largest of kAutoChunks whose buffers take at most kAutoLendPct % of the slots (a
// lent slot's expert is streamed during the prompt and refilled after it; measured on a 12 GB card, 32K Q2_0
// prompt: 4096 791 tok/s, 6144 878, 8192 973 with 69% of the slots lent). A request lends only what its own
// prompt needs (Prefill::relayout), so a big chunk costs short prompts nothing. 0 = none fits.
// at 8192-token chunks nearly every expert streams anyway, so a lent slot costs little: 90% when the
// copies are DMA from pinned RAM (Q2_0 8192 + a 384-slot ring: 1283 tok/s), 85% when host copies are the
// limit (lending more only streams more through them). STRATA_PREFILL_LEND_PCT overrides (tuning).
// Hoisted out of plan_lend: the serve path's per-stage loans obey the same cap, one participant at a time.
const int64_t kAutoLendPct = [] {
const char* v = std::getenv("STRATA_PREFILL_LEND_PCT");
return v ? (int64_t) std::atoi(v)
: (int64_t) (strata::prefill::Prefill::pinned_share() >= 0.9 ? 90 : 85);
}();
auto plan_lend = [&](int64_t& chunk) -> int64_t {
// 32768 and 16384 (#282, opt-in: --prefill auto:32768): a 32K prompt with IQ2_XS (RTX 5090, 64K context)
// read at 5,624 tok/s in 8192-token chunks and 6,465 in one 32768 chunk (40K -> 18K experts streamed; an
// NVFP4 pack at 262K: 3,535 -> 5,201)
static constexpr int64_t kAutoChunks[] = {32768, 16384, 8192, 6144, 4096, 3072, 2048, 1024, 512, 256};
auto slots_for = lend_slots;
if (o.prefill_auto) {
for (const int64_t c : kAutoChunks) {
// above 8192: only when asked for, and only when a prompt of the context can use it
if (c > 8192 && (c > o.prefill_auto_max || c > o.max_context)) continue;
const int64_t k = slots_for(c);
if (k + 128 <= xcache.slots() && k * 100 <= kAutoLendPct * xcache.slots()) { chunk = c; return k; }
}
return 0;
}
for (int64_t c = chunk; c >= 256; c /= 2) {
const int64_t k = slots_for(c);
if (k + 128 <= xcache.slots()) { chunk = c; return k; }
}
return 0;
};
// ---- the resident RAM mode (--resident-experts / --resident-cpu-experts): the experts the GPU cache does not
// hold are copied from experts.bin into RAM once, so no decode or prompt step reads the file (the plain mmap
// mode reads them through the OS file cache, which a small-RAM PC keeps giving back to the SSD). Built here,
// after the prompt path's lend plan is known: the slots it may lend (the cache's last ones) have their experts
// streamed during a prompt and copied back after it, so those are kept in RAM too as far as RAM allows. The
// bytes are the file's bytes and the placement is the same, so the answers are the plain mmap mode's; the
// share-of-pinned figure above (which sizes the prompt path) is left as the mmap mode's for the same reason.
if (o.resident_cpu_experts) {
int64_t lend_from = -1;
if (o.prefill_chunk > 0 && !o.no_prefill_borrow && d_res != nullptr && xcache.slots() > 0) {
int64_t chunk = o.prefill_chunk;
const int64_t k = plan_lend(chunk);
if (k > 0) lend_from = xcache.slots() - k;
}
bool resident_ok = src.pin_cache_complement(xcache, err, o.resident_pin, {}, lend_from, o.resident_headroom,
o.resident_budget, &profile);
std::string whole_err;
if (!resident_ok && o.resident_soft) {
// #467: the whole complement does not fit - keep what does, the hottest by the profile, through the #403
// budget path (sized by the RAM alone) instead of none: the misses outside it read the same file bytes
// the mmap fallback reads, so the answers are unchanged. Nothing pinned: the old fallback below.
whole_err = err;
resident_ok = src.pin_cache_complement(xcache, err, o.resident_pin, {}, -1, o.resident_headroom,
strata::core::FileExpertSource::kResidentWhatFits, &profile);
if (resident_ok)
std::fprintf(stderr, "strata generate: WARNING: the whole resident RAM mode does not fit (%s); %.2f "
"GiB of the experts the GPU does not hold, the hottest by the expert profile, are "
"kept in RAM and the rest are read from the model folder through the OS file "
"cache\n",
whole_err.c_str(), (double) src.resident_bytes() / 1073741824.0);
else
err = whole_err + "; " + err;
}
if (resident_ok) {
if (o.adapt_every > 0 && o.adapt_swaps > 0 &&
!src.reserve_exchanges(std::min<int64_t>(o.adapt_swaps, 96), err)) {
std::fprintf(stderr, "strata generate: CPU expert residency: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata generate: resident RAM mode: %.2f GiB of experts in RAM (%s), %lld in the GPU "
"cache; adaptive swaps %s\n",
(double) src.resident_bytes() / 1073741824.0,
src.complement_pinned() ? "page-locked" : src.locked_bytes() > 0 ? "locked" : "pageable",
(long long) xcache.resident(),
o.adapt_every > 0 && o.adapt_swaps > 0 ? "exchange them with the GPU cache (no file reads)"
: "off");
} else if (o.resident_soft) {
std::fprintf(stderr, "strata generate: WARNING: the resident RAM mode does not fit (%s); the experts the "
"GPU does not hold are read from the model folder through the OS file cache "
"(--mmap-experts), which is slower when the RAM cannot keep them\n", err.c_str());
} else if (o.resident_budget > 0) {
// #403: a RAM budget that cannot be kept is not a reason to stop - the experts it would have held are
// read from the files like the ones outside it (pin_cache_complement leaves nothing half-built)
std::fprintf(stderr, "strata generate: WARNING: the RAM budget (--resident-budget-gib) cannot be kept (%s); "
"every expert the GPU does not hold is read from the model files through the OS file "
"cache (--mmap-experts), which is slower\n", err.c_str());
} else {
std::fprintf(stderr, "strata generate: CPU expert residency: %s\n", err.c_str());
return 1;
}
}
if (o.serve) {
if (o.spec < 2 || o.mtp.empty() || o.prefill_chunk <= 0 ||
(graph_hits && (thits.d_res == nullptr || host_res.empty()))) {
std::fprintf(stderr, "strata serve: needs --spec T, --mtp DIR and --prefill CHUNK (and a fillable "
"--expert-cache; the graphed hit path additionally needs --expert-profile P)\n");
return 2;
}
strata::prefill::Prefill sp;
void* borrow = nullptr;
uint64_t borrow_bytes = 0;
int32_t lend_first = -1; // the first slot the prompt path may borrow (its largest chunk)
// ---- WHO BORROWS, AND FROM WHOSE CACHE. One entry per prompt path: CUDA0's (layers [0, split_at[0]),
// which is the whole model without a split) borrowing the tail of CUDA0's cache, then one per stage
// borrowing the tail of ITS OWN cache. A loan is sized by the exact `Prefill::bytes_needed` for the
// chunk, is laid out by `Prefill::relayout`, and is refilled before any window reads - so outside the
// prompt the whole cache is expert cache. THIS IS THE POINT OF THE STRUCT: the loan used to exist only
// for CUDA0, and `no_prefill_borrow` made every stage instead withhold a chunk-sized reserve from its
// cache for the entire session, which is what cost the 4-way its context (see the note at the top of the
// layer-split block). A stage may only lend the rows for ITS OWN layers: `host_res` is one table whose
// slot values are indices into whichever cache owns the layer, so a loan that marked rows by slot number
// alone would hand CUDA0 a slot belonging to another stage's cache.
struct PfPart {
strata::core::ExpertCache* cache = nullptr;
const strata::core::SessionState* ses = nullptr;
strata::prefill::Prefill* sp = nullptr;
int dev = -1; // -1: leave the device alone (CUDA0)
int64_t lb = 0, le = 0; // the layers whose rows this cache holds - the only rows it may lend
int32_t first = -1; // the first slot it may lend, for the chunk that was chosen
int32_t first_now = -1; // where its buffers are laid out now
int64_t lent_chunk = 0;
std::vector<std::pair<int32_t, int32_t>> lent;
};
auto part_slots = [&](const PfPart& p, int64_t c) -> int64_t {
const uint64_t need = strata::prefill::Prefill::bytes_needed(g, *p.ses, c);
strata::core::ExpertCache& xc = *p.cache;
if (xc.slot_offsets() != nullptr) { // sized slots: from the end until they hold `need`
int64_t k = 0;
while (k < xc.slots() &&
(uint64_t) (xc.bytes() - (int64_t) xc.slot_offsets()[xc.slots() - k]) < need) ++k;
return k;
}
const int64_t blob = (int64_t) strata::kernels::cpu::expert_layout().max_blob;
return (int64_t) ((need + (uint64_t) blob - 1) / (uint64_t) blob);
};
auto part_bytes = [&](const PfPart& p, int32_t first) -> uint64_t {
strata::core::ExpertCache& xc = *p.cache;
return xc.slot_offsets() ? (uint64_t) (xc.bytes() - (int64_t) xc.slot_offsets()[first])
: (uint64_t) (xc.slots() - first) *
(uint64_t) strata::kernels::cpu::expert_layout().max_blob;
};
// #340: a layer split whose caches already hold most experts streams few of them through the prompt path, so
// the 384-slot ring (sized for a card that streams nearly every expert of a chunk) only makes every stage's
// loan bigger: 96 slots (0.1.30's ring here) when >= 75% of the (layer, expert) pairs are resident. One GPU
// keeps the pinned-share rule. STRATA_SPLIT_RING=N: N slots on a split; 0: the pinned-share rule.
if (multi_gpu && !host_res.empty()) {
int64_t res_n = 0;
for (const int32_t r : host_res) res_n += r >= 0;
const double res_share = (double) res_n / (double) host_res.size();
const char* v = std::getenv("STRATA_SPLIT_RING");
const int ring = v ? std::atoi(v) : (res_share >= 0.75 ? 96 : 0);
if (ring > 0) {
strata::prefill::Prefill::set_ring_override(ring);
std::fprintf(stderr, "strata serve: layer split: %.0f%% of the experts resident, the prompt path's "
"streamed ring %d slots\n", 100.0 * res_share, ring);
}
}
std::vector<PfPart> pf_parts;
// a cache too small to lend the prompt path its buffers would make it allocate them on top - on a card
// whose cache already filled its reserve, that is the over-subscription the auto sizing avoids - so the
// chunk is the largest one EVERY participant can lend (a smaller chunk only reads slower)
if (pf_borrow && d_res != nullptr) {
pf_parts.push_back({&xcache, &ss, &sp, -1, 0, multi_gpu ? split_at[0] : g.n_layers, -1, -1, 0, {}});
for (auto& st : stages)
pf_parts.push_back({&st->cache, &st->ss, &st->sp, st->dev, st->lb, st->le, -1, -1, 0, {}});
// The two tests plan_lend makes for CUDA0 alone, one participant at a time: a loan must leave the
// 128-slot floor. The percentage cap is an AUTO-chunk rule and only the auto scan applies it - an
// explicit --prefill is the operator's number, and a loan of it only has to fit. With one participant
// (no split) this reduces to plan_lend exactly, so the single-GPU loan is unchanged from main.
auto fits_one = [&](const PfPart& p, int64_t c, bool cap) -> bool {
const int64_t k = part_slots(p, c);
if (k <= 0 || k + 128 > p.cache->slots()) return false;
return !(cap && k * 100 > kAutoLendPct * p.cache->slots());
};
// `only`: CUDA0's cache alone (#448: what one GPU would choose, for the log below); null: every one
auto fits = [&](int64_t c, bool cap, const PfPart* only = nullptr) -> bool {
if (only != nullptr) return fits_one(*only, c, cap);
for (const PfPart& p : pf_parts)
if (!fits_one(p, c, cap)) return false;
return true;
};
static constexpr int64_t kAutoChunks[] = {32768, 16384, 8192, 6144, 4096, 3072, 2048, 1024, 512, 256};
auto pick = [&](const PfPart* only) -> int64_t {
if (o.prefill_auto) {
for (const int64_t c : kAutoChunks) {
if (c > 8192 && (c > o.prefill_auto_max || c > o.max_context)) continue; // #282, as plan_lend
if (fits(c, true, only)) return c;
}
} else {
for (int64_t c = o.prefill_chunk; c >= 256; c /= 2)
if (fits(c, false, only)) return c;
}
return 0;
};
const int64_t chunk = pick(nullptr);
// #448: a small card in a layer split caps every stage's chunk (an RTX 3080's 512-slot cache held a
// 32 GB card's split to 512 tokens: prompts 6.2x slower, decode the same). Named when it bites, so the
// regression is one log line: each stage that cannot fund the chunk CUDA0 alone would read in.
if (pf_parts.size() > 1) {
const int64_t alone = pick(&pf_parts[0]);
if (alone > chunk) {
for (size_t i = 1; i < pf_parts.size(); ++i) {
const PfPart& p = pf_parts[i];
if (fits_one(p, alone, o.prefill_auto)) continue;
const int dev = p.dev < 0 ? 0 : p.dev;
cudaDeviceProp prop{};
if (cudaGetDeviceProperties(&prop, dev) != cudaSuccess) {
(void) cudaGetLastError();
prop.name[0] = 0;
}
const std::string pct =
o.prefill_auto ? ", and lend at most " + std::to_string(kAutoLendPct) + "%" : "";
std::fprintf(stderr, "strata serve: WARNING: prompt chunk %lld tokens, not %lld: CUDA%d (%s) "
"has %lld expert-cache slots, and a %lld-token chunk borrows %lld of them "
"(it must keep 128%s) - prompts read slower than on CUDA0 alone (#448)\n",
(long long) chunk, (long long) alone, dev, prop.name,
(long long) p.cache->slots(), (long long) alone,
(long long) part_slots(p, alone), pct.c_str());
// the helper tiers start at CUDA1 without a split (and are enabled in order)
std::fprintf(stderr, "strata serve: a card this small can serve as a helper expert cache "
"instead of a split stage: without --layer-split, with %s "
"(docs/SECOND_GPU.md)\n",
dev == 1 ? "--expert-cache-device1 N" : "--expert-cache-device1..3 N, in order");
}
}
}
if (chunk > 0) {
if (o.prefill_auto)
std::fprintf(stderr, "strata serve: prompt chunk auto: %lld tokens\n", (long long) chunk);
else if (chunk != o.prefill_chunk)
std::fprintf(stderr, "strata serve: prompt chunk %lld -> %lld tokens so its buffers fit in "
"every expert cache\n", (long long) o.prefill_chunk, (long long) chunk);
o.prefill_chunk = chunk;
for (PfPart& p : pf_parts) {
p.first = (int32_t) (p.cache->slots() - part_slots(p, chunk));
p.first_now = p.first;
}
// #340: a split stage whose card still has room for the chunk's buffers (its cache already holds
// every expert of its layers, so auto stopped short of its VRAM) keeps them as its own instead of
// borrowing cache slots: nothing of it is lent, streamed during the prompt or refilled after it.
// Room = the buffers + 1.5 GiB (the verify windows, the draft head, hipBLAS/cuBLAS workspaces made
// after this). Layer splits only - one GPU keeps its loan exactly as before.
// STRATA_SPLIT_OWN_BUFFERS=0: every stage borrows (0.1.30/0.1.31).
static const bool own_ok = [] {
const char* v = std::getenv("STRATA_SPLIT_OWN_BUFFERS");
return v == nullptr || v[0] != '0';
}();
if (pf_parts.size() > 1 && own_ok) {
for (PfPart& p : pf_parts) {
const strata::core::OnDevice on(p.dev);
size_t fb = 0, tb = 0;
if (cudaMemGetInfo(&fb, &tb) != cudaSuccess) { (void) cudaGetLastError(); continue; }
const uint64_t need = strata::prefill::Prefill::bytes_needed(g, *p.ses, chunk);
if ((uint64_t) fb >= need + (3ull << 29)) {
std::fprintf(stderr, "strata serve: CUDA%d keeps its own prompt buffers (%.2f GiB of "
"%.2f GiB free): no loan\n", p.dev < 0 ? 0 : p.dev,
(double) need / 1073741824.0, (double) fb / 1073741824.0);
p.first = -1;
p.first_now = -1;
}
}
}
lend_first = pf_parts[0].first;
borrow = lend_first >= 0 ? xcache.device_slot(lend_first) : nullptr;
borrow_bytes = lend_first >= 0 ? part_bytes(pf_parts[0], lend_first) : 0;
} else if (o.prefill_auto) {
o.prefill_chunk = 1024; // nothing lendable: small buffers of its own
} else if (pf_parts.size() > 1) {
// An explicit chunk no stage can lend in full. main falls back to the prompt path's own buffers
// here and so do we, rather than refusing to start - but say what every stage has, because a split
// stage that has to allocate these on top of a cache that already filled its VRAM will not fit,
// and `init` would otherwise report only that the buffers do not fit.
std::fprintf(stderr, "strata serve: no stage can lend the prompt path its %lld-token buffers, so "
"each stage allocates its own:\n", (long long) o.prefill_chunk);
for (const PfPart& p : pf_parts)
std::fprintf(stderr, "strata serve: CUDA%d has %lld slots, and a %lld-token chunk needs "
"the last %lld of them\n", p.dev < 0 ? 0 : p.dev,
(long long) p.cache->slots(), (long long) o.prefill_chunk,
(long long) part_slots(p, o.prefill_chunk));
}
} else if (o.prefill_auto && d_res == nullptr) {
o.prefill_chunk = 1024; // #85: no expert cache at all (a full 8 GB card): small buffers of its own
}
bool any_loan = borrow != nullptr;
for (size_t i = 1; i < pf_parts.size(); ++i) any_loan = any_loan || pf_parts[i].first >= 0;
if (any_loan) {
if (borrow != nullptr)
std::fprintf(stderr, "strata serve: the prompt path borrows %lld CUDA0 cache slots (%.2f GiB)\n",
(long long) (xcache.slots() - lend_first), (double) borrow_bytes / 1073741824.0);
for (size_t i = 1; i < pf_parts.size(); ++i) // one loan per stage, from that stage's own cache
if (pf_parts[i].first >= 0) std::fprintf(stderr, "strata serve: CUDA%d prompt path borrows %lld of its %lld slots (%.2f GiB)\n",
pf_parts[i].dev, (long long) (pf_parts[i].cache->slots() - pf_parts[i].first),
(long long) pf_parts[i].cache->slots(),
(double) part_bytes(pf_parts[i], pf_parts[i].first) / 1073741824.0);
} else {
std::fprintf(stderr, "strata serve: the prompt path allocates its own buffers (too few cache slots to borrow)\n");
}
// layer split across GPUs: a prompt path per stage, each handing its chunk's rows to the next.
// THE CHUNK STEPS DOWN INSTEAD OF EXITING. A split's stage caches are sized after the arena is registered, and
// the prompt path's own buffers (no loan) are not priced into them: with the whole arena pinned (#253) a
// `--prefill auto` split could stop at start with "device buffers for a chunk of 2048 tokens do not fit". A
// chunk that does not fit is tried again one size smaller, down to 512 tokens (a smaller chunk only reads slower).
auto init_prompt_paths = [&]() -> int { // 0: ready; 1: failed (err set); 2: failed with "do not fit"
for (size_t i = 0; i < stages.size(); ++i) {
GpuStage& st = *stages[i];
st.sp.set_stage(st.lb, i + 1 < stages.size() ? st.le : -1, i + 1 < stages.size() ? &stages[i + 1]->sp : nullptr);
const strata::core::OnDevice on(st.dev);
void* sb = nullptr; // this stage's own loan, out of its own cache
uint64_t sbb = 0;
// `first < 0`: no loan was taken (nothing was lendable), so this stage allocates its own buffers
if (i + 1 < pf_parts.size() && pf_parts[i + 1].first >= 0) {
sb = st.cache.device_slot(pf_parts[i + 1].first);
sbb = part_bytes(pf_parts[i + 1], pf_parts[i + 1].first);
}
if (!st.sp.init(st.wt, g, st.ss, srcp, &st.cache, host_res.data(), o.prefill_chunk, (void*) st.stream,
err, sb, sbb)) {
err = "layer split, CUDA" + std::to_string(st.dev) + " prompt path: " + err;
return err.find("do not fit") != std::string::npos ? 2 : 1;
}
}
if (multi_gpu) sp.set_stage(0, split_at[0], &stages[0]->sp);
if (!sp.init(wt, g, ss, srcp, &xcache, host_res.data(), o.prefill_chunk, main_cs, err, borrow, borrow_bytes))
return err.find("do not fit") != std::string::npos ? 2 : 1;
return 0;
};
static constexpr int64_t kStepChunks[] = {6144, 4096, 3072, 2048, 1536, 1024, 512};
{
// First by arithmetic: a prompt path without a loan allocates its buffers, so the chunk must leave
// headroom on that device (a chunk that fits to the last MiB left hipBLAS nothing: its GEMMs then
// failed to launch on gfx1201 and the prompt hung). `bytes_needed` is the same count `init` makes.
const int64_t kHeadroom = 512ll << 20;
auto own_fits = [&](int64_t c, int& dev_out, int64_t& need_out, int64_t& free_out) -> bool {
for (size_t i = 0; i <= stages.size(); ++i) {
const bool loan = i == 0 ? borrow != nullptr : (i < pf_parts.size() && pf_parts[i].first >= 0);
if (loan) continue;
const int dev = i == 0 ? -1 : stages[i - 1]->dev;
const strata::core::OnDevice on(dev);
size_t fb = 0, tb = 0;
cudaMemGetInfo(&fb, &tb);
const int64_t need = (int64_t) strata::prefill::Prefill::bytes_needed(g, i == 0 ? ss : stages[i - 1]->ss, c);
if (need + kHeadroom > (int64_t) fb) {
dev_out = dev < 0 ? 0 : dev; need_out = need; free_out = (int64_t) fb;
return false;
}
}
return true;
};
int dev = 0;
int64_t need = 0, fb = 0;
if (!own_fits(o.prefill_chunk, dev, need, fb)) {
int64_t c = 0;
for (const int64_t s : kStepChunks) {
int d2 = 0;
int64_t n2 = 0, f2 = 0;
if (s < o.prefill_chunk && own_fits(s, d2, n2, f2)) { c = s; break; }
}
if (c > 0) {
std::fprintf(stderr, "strata serve: a %lld-token chunk's prompt buffers need %lld MiB on CUDA%d, "
"%lld MiB free: %lld-token chunks\n", (long long) o.prefill_chunk,
(long long) (need >> 20), dev, (long long) (fb >> 20), (long long) c);
o.prefill_chunk = c;
}
}
}
for (;;) {
const int r = init_prompt_paths();
if (r == 0) break;
int64_t next = 0;
for (const int64_t c : kStepChunks)
if (c < o.prefill_chunk) { next = c; break; }
if (r == 2 && next > 0) {
std::fprintf(stderr, "strata serve: %s: trying a %lld-token chunk\n", err.c_str(), (long long) next);
// start the prompt paths over (a failed init may hold buffers), and consume the failed allocation's
// error: it is sticky, and the next launch check would report "out of memory" for a kernel
for (auto& stp : stages) {
const strata::core::OnDevice on(stp->dev);
stp->sp.reset();
(void) cudaGetLastError();
}
sp.reset();
(void) cudaGetLastError();
o.prefill_chunk = next;
if (any_loan) { // smaller loans for the smaller chunk
for (PfPart& p : pf_parts)
if (p.first >= 0) {
p.first = (int32_t) (p.cache->slots() - part_slots(p, next));
p.first_now = p.first;
}
lend_first = pf_parts[0].first;
borrow = lend_first >= 0 ? xcache.device_slot(lend_first) : nullptr;
borrow_bytes = lend_first >= 0 ? part_bytes(pf_parts[0], lend_first) : 0;
}
err.clear();
continue;
}
std::fprintf(stderr, "strata serve: %s\n", err.c_str());
if (err.find("fit") != std::string::npos) // #85: say what frees VRAM
std::fprintf(stderr, "strata serve: the GPU has too little free VRAM for the prompt path: turn images "
"off (setup: --vision no), close other programs using the GPU, use a shorter "
"context, or read prompts in smaller chunks (--prefill 512)\n");
return 1;
}
if (peer.valid() && o.peer_prefill_rows != 0) {
const int64_t rows = o.peer_prefill_rows > 0 ? o.peer_prefill_rows : o.prefill_chunk * K / 2;
if (!sp.set_peer(&peer, rows, err)) {
std::fprintf(stderr, "strata serve: %s - the prompt path stays on the primary GPU\n", err.c_str());
err.clear();
}
}
mem_mark("the head and the prompt path");
// #340: STRATA_SPLIT_SMALL_OWN=S (tokens): on a layer split, every stage that borrows keeps the slots for an
// S-token chunk's buffers for the whole session (0.1.29's own buffers, carved from the tail of its cache):
// a request of at most S prompt tokens then lends, streams and refills nothing, a longer one lends (and
// refills) only the slots above them. Their experts stay non-resident (the CPU / PCIe path computes them).
// STRATA_SPLIT_SMALL_MAX=M: a request of at most M prompt tokens reads in S-token chunks on those stages
// (several chunks, still nothing lent) instead of borrowing for a bigger one.
int64_t split_small = 0, split_small_max = 0;
if (multi_gpu && !pf_parts.empty() && !host_res.empty()) {
const char* v = std::getenv("STRATA_SPLIT_SMALL_OWN");
const int64_t S = v ? request_chunk(std::atoll(v), o.prefill_chunk) : 0;
const char* vm = std::getenv("STRATA_SPLIT_SMALL_MAX");
split_small = S;
split_small_max = S > 0 ? std::max<int64_t>(S, vm ? std::atoll(vm) : S) : 0;
int64_t evicted = 0;
for (PfPart& p : pf_parts) {
if (S <= 0 || p.first < 0) continue;
const int32_t first_s = std::max<int32_t>(p.first, (int32_t) (p.cache->slots() - part_slots(p, S)));
int64_t n = 0;
for (int64_t l = p.lb; l < p.le; ++l)
for (int64_t ex = 0; ex < g.n_expert; ++ex) {
int32_t& r = host_res[(size_t) (l * g.n_expert + ex)];
if (r >= first_s) { r = strata::core::kNotResident; ++n; }
}
evicted += n;
std::fprintf(stderr, "strata serve: CUDA%d keeps %lld slots for %lld-token prompts (%lld experts "
"no longer resident)\n", p.dev < 0 ? 0 : p.dev,
(long long) (p.cache->slots() - first_s), (long long) S, (long long) n);
}
if (evicted > 0) {
if (d_res != nullptr)
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
for (auto& st : stages) {
const strata::core::OnDevice on(st->dev);
cudaMemcpy(st->d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
}
}
}
// the penalty-history buffer: one row per verify-window row (`penalty_rows`), each the last
// `penalty_last_n` tokens that row's pick follows, -1 padded in front. Allocated once at the cap for
// the widest window; a request without penalties gets a null buffer and takes the byte-for-byte
// neutral path (no upload, no buffer handed to the sampler).
constexpr int kPenaltyWindowCap = 4096;
constexpr size_t kHistSlots = (size_t) kPenaltyWindowCap * (size_t) strata::kernels::kVerifyMaxT;
int32_t* d_hist = nullptr;
std::vector<int32_t> hist_stage(kHistSlots, -1);
const int hist_dev = last_st ? last_st->dev : -1; // with the head: the last stage's device
if (const strata::core::OnDevice on_h(hist_dev); cudaMalloc(&d_hist, kHistSlots * sizeof(int32_t)) != cudaSuccess) {
std::fprintf(stderr, "strata serve: the penalty-history allocation failed\n");
return 1;
}
strata::core::Verifier ver;
strata::core::VerifyHits vh;
vh.d_res = thits.d_res;
vh.cache_base = thits.cache_base;
vh.blob = thits.blob;
vh.slot_off = xcache.slot_offsets(); // E-6: the device plan's pointers
vh.n_slots = xcache.slots();
// Layer split: `ver` runs layers [0, K1) and hands its residual to the next stage's verifier, and so on; the
// last runs the head. The hand-offs are mapped pinned memory, portable: a stage on another GPU reads it.
// (--split-device 0: the second stage on this GPU, sharing its weights, session and cache - the A/B.)
strata::core::Verifier ver_same;
SplitDrive split_drive;
auto stage_ver = [&](int st) -> strata::core::Verifier& {
return st == 0 ? ver : split_same ? ver_same : stages[(size_t) st - 1]->ver;
};
auto pcie_num_of = [](double f) { return std::max(0, std::min(256, (int) (f * 256.0 + 0.5))); };
const int n_stages = split_devs.empty() ? 1 : (int) split_at.size() + 1;
if (n_stages > 1) {
const size_t hb = (size_t) strata::kernels::kVerifyMaxT *
(size_t) strata::core::Verifier::handoff_floats(g) * sizeof(float);
std::vector<float*> hand((size_t) n_stages - 1, nullptr);
for (float*& h : hand) {
float* hh = nullptr;
if (cudaHostAlloc((void**) &hh, hb, cudaHostAllocMapped | cudaHostAllocPortable) != cudaSuccess ||
cudaHostGetDevicePointer((void**) &h, hh, 0) != cudaSuccess) {
std::fprintf(stderr, "strata serve: the layer-split hand-off allocation failed\n");
return 1;
}
std::memset(hh, 0, hb);
}
split_drive.base = &drive;
split_drive.n = n_stages;
for (int st = 0; st < n_stages; ++st) {
stage_ver(st).set_stage(st == 0 ? 0 : split_at[(size_t) st - 1], st + 1 < n_stages ? split_at[(size_t) st] : -1,
st == 0 ? nullptr : hand[(size_t) st - 1], st + 1 < n_stages ? hand[(size_t) st] : nullptr);
split_drive.end[st] = st + 1 < n_stages ? split_at[(size_t) st] : g.n_layers;
split_drive.cache_base[st] = drive.d.cache_base;
split_drive.cache_slot_off[st] = drive.d.cache_slot_off;
split_drive.pcie_num[st] = pcie_num_of(o.pcie_frac);
}
for (int st = 1; st < n_stages; ++st) {
bool ok_s = false;
if (split_same) {
ok_s = ver_same.init(wt, g, ss, vh, native_head.loaded() ? &native_head : nullptr, o.spec, err);
} else {
GpuStage& gs = *stages[(size_t) st - 1];
const strata::core::OnDevice on(gs.dev);
strata::core::VerifyHits vs;
vs.d_res = gs.d_res;
vs.cache_base = gs.cache.device_slot(0);
vs.blob = thits.blob;
vs.slot_off = gs.cache.slot_offsets();
vs.n_slots = gs.cache.slots();
ok_s = gs.ver.init(gs.wt, g, gs.ss, vs, gs.head.loaded() ? &gs.head : nullptr, o.spec, err);
split_drive.cache_base[st] = gs.cache.device_slot(0);
split_drive.cache_slot_off[st] = gs.cache.slot_offsets();
split_drive.pcie_num[st] = pcie_num_of(gs.pcie_frac);
}
if (!ok_s) {
std::fprintf(stderr, "strata serve: layer split, stage %d: %s\n", st + 1, err.c_str());
return 1;
}
}
for (int st = 0; st + 1 < n_stages; ++st) stage_ver(st).set_next(&stage_ver(st + 1), &split_drive);
std::string plan_s = "0-" + std::to_string(split_at[0] - 1) + " (CUDA0)";
for (int st = 1; st < n_stages; ++st)
plan_s += ", " + std::to_string(split_at[(size_t) st - 1]) + "-" + std::to_string(split_drive.end[st] - 1) +
" (CUDA" + std::to_string(split_same ? 0 : stages[(size_t) st - 1]->dev) + ")";
std::fprintf(stderr, "strata serve: layer split: layers %s, one hand-off per window\n", plan_s.c_str());
}
if (!ver.init(wt, g, ss, vh, native_head.loaded() ? &native_head : nullptr, o.spec, err) ||
!mtp.bind(last_st ? last_st->wt : wt, last_st ? &last_st->head : &native_head, ver.final_R_all(), err)) {
std::fprintf(stderr, "strata serve: %s\n", err.c_str());
return 1;
}
for (int st = 0; st < n_stages && n_stages > 1; ++st) {
split_drive.plan[st] = stage_ver(st).plan_sink();
if (st > 0) {
stage_ver(st).set_split(o.spec_split);
stage_ver(st).set_pcie_mode(o.pcie_mode == "dma" ? 0 : o.pcie_mode == "direct" ? 1 : 2);
}
}
// the pool the verify windows call: with a layer split, the wrapper that routes each layer to its stage
const strata::core::PoolMultiFn win_pool_fn = n_stages > 1 ? &drive_pool_split : &drive_pool_multi;
void* const win_pool_user = n_stages > 1 ? (void*) &split_drive : (void*) &drive;
mem_mark("the verifier and the drafter's binding");
ver.set_split(o.spec_split);
// auto: the copy kernel for every pack. DMA (the native packs' default until 0.1.13) has the host call
// cudaMemcpyAsync + cudaLaunchHostFunc inside a verify window while the GPU spins on the flag they raise;
// issue #31's thread dumps show the host stuck in that cudaMemcpyAsync on a driver lock for good. The copy
// kernel needs no host CUDA call there, and costs ~1-3% decode on IQ3_S (45.3 -> 44.8 tok/s, 8 requests).
ver.set_pcie_mode(o.pcie_mode == "dma" ? 0 : o.pcie_mode == "direct" ? 1 : 2);
std::vector<int64_t> cur;
// ---- the conversation cache (see ConvCheckpoint). `live` is what the session holds right now: the tokens
// it has consumed, so a request that starts with exactly them continues without any copy. `checks` are the
// saved points; every one of them is a prefix of `live` (the loop drops the rest), so they form a chain -
// the radix cache's tree collapsed onto the one branch of history whose cells the session holds. The
// chain's root is the deepest point every request so far shared (the end of the system prompt, in
// practice); the retention policy pins it and rotates the rest LRU (conv_cache.hpp), so a NEW chat that
// shares that prefix mounts through it instead of reading it again.
std::vector<int32_t> live;
std::vector<ImgKey> live_imgs, req_imgs;
bool live_ok = false;
std::vector<ConvCheckpoint> checks;
uint64_t check_clock = 0; // the checkpoints' LRU clock; creation and every use advance it
bool cvec_cached = true; // the control vector's state the live session and the checkpoints were read with
strata::core::ConversationCache conversations(
o.prompt_cache > 0 ? (size_t) o.conversation_cache_mib * 1024 * 1024 : 0,
(size_t) o.conversation_cache_slots);
// Save only on a switch/rewind, not on each continuing request. No graph
// addresses change: all parked images live in ordinary host vectors.
auto park_current = [&](size_t held) -> bool {
if (!conversations.enabled() || !live_ok || live.empty()) return true;
const strata::core::ConversationView view{live, live_imgs, checks, cvec_cached};
auto reuse = conversations.take_reuse();
size_t estimate = 0;
if (!strata::core::conversation_snapshot_bytes(view, ss, g, mtp.kv_state(), estimate, err)) {
std::fprintf(stderr, "strata serve: conversation cache: skip parking (%s)\n", err.c_str());
err.clear(); // A recoverable miss must not poison the batched draft prefill's error channel.
return true;
}
const size_t fresh_estimate = estimate;
if (!reuse.kv.empty() && !strata::core::conversation_snapshot_capture_bytes(
reuse, view, ss, g, mtp.kv_state(), estimate, err)) {
reuse = {};
estimate = fresh_estimate;
err.clear();
}
// A park carrying its own retained K/V replaces memory the cache
// already held, so capacity is make_room's call - it runs next either
// way, and put()'s accounting still bounds the budget. The with-reuse
// estimate must stay uncapped: it counts the retained buffers'
// capacity and directories, and put() charges that same true size -
// a capped figure would under-evict and overfill the budget.
// #342: before make_room evicts oldest-first, the copies of this conversation a turn back go (they hold
// nothing the outgoing chain does not, apart from the tail this conversation rewrote)
if (const size_t dropped = conversations.drop_superseded(live, live_imgs, checks, cvec_cached))
std::fprintf(stderr, "strata serve: conversation cache: dropped %zu superseded cop%s of this "
"conversation; parked=%zu\n", dropped, dropped == 1 ? "y" : "ies", conversations.size());
if (!conversations.make_room(estimate, held)) {
std::fprintf(stderr, "strata serve: conversation cache: skip parking (snapshot %zu MiB exceeds available budget)\n",
estimate >> 20);
return true;
}
const auto t0 = Clock::now();
try {
const uint64_t floor = (uint64_t) o.conversation_cache_min_free_mib * 1024 * 1024;
const size_t additional = estimate - reuse.bytes();
if (!strata::core::conversation_memory_admit(strata::core::conversation_available_memory(),
additional, floor)) {
std::fprintf(stderr, "strata serve: conversation cache: skip parking (physical RAM admission; need %zu MiB plus %lld MiB floor, or telemetry unavailable)\n",
additional >> 20, (long long) o.conversation_cache_min_free_mib);
return true;
}
strata::core::SavedConversation image;
size_t reused_bytes = 0;
if (!strata::core::conversation_snapshot_save(image, view, ss, g, mtp.kv_state(), err,
std::move(reuse), &reused_bytes)) return false;
if (!strata::core::conversation_memory_admit(strata::core::conversation_available_memory(), 0, floor)) {
std::fprintf(stderr, "strata serve: conversation cache: skip parking (physical RAM floor after capture, or telemetry unavailable)\n");
return true;
}
const size_t snapshot_bytes = image.bytes();
const bool stored = conversations.put(std::move(image), held);
std::fprintf(stderr, "strata serve: conversation cache: %s %zu tokens in %.1f ms; parked=%zu bytes=%zu evictions=%zu snapshot_bytes=%zu reused_kv_bytes=%zu\n",
stored ? "parked" : "skipped", live.size(),
std::chrono::duration<double, std::milli>(Clock::now() - t0).count(),
conversations.size(), conversations.bytes(), conversations.evictions(), snapshot_bytes, reused_bytes);
} catch (const std::bad_alloc&) {
// The active state has not been touched. Continue with normal
// prompt processing rather than killing a serving process.
std::fprintf(stderr, "strata serve: conversation cache: allocation failed; skip parking\n");
}
return true;
};
int64_t pp_total = 0, pp_from = 0, pp_next_check = 0;
// #471: the position the prompt pass has read up to (a chunk's or a window's end): what a request cancelled
// mid-read reports as read, instead of the whole prompt
int64_t pp_reached = 0;
Clock::time_point pp_t0 = Clock::now();
auto imgs_below = [&](const std::vector<ImgKey>& all, int64_t L) {
std::vector<ImgKey> v;
for (const ImgKey& k : all) if (k.start < L) v.push_back(k);
return v;
};
// a checkpoint of the state after `cur[0, L)`; false only when the copy itself failed
// A layer split's mid-prompt checkpoints: when the last stage reports a chunk, the earlier ones already read
// the next, so each stage saves its own part when IT reaches a checkpoint position (the same rule as below:
// every `prompt_cache_every` tokens from where the request resumed), and the last stage puts them together.
std::mutex part_mu;
std::map<int64_t, std::vector<ConvCheckpoint>> part_at; // position -> one part per stage
std::vector<int64_t> part_next(stages.size() + 1, INT64_MAX);
// a checkpoint of the state after `cur[0, L)`; false only when the copy itself failed. `parts`: the stages'
// states saved at L (a split's mid-prompt checkpoint); without, they are read now (everything is at L)
auto checkpoint_at = [&](int64_t L, std::vector<ConvCheckpoint>* parts = nullptr) -> bool {
if (o.prompt_cache <= 0 || L < 1) return true;
for (ConvCheckpoint& c : checks)
if ((int64_t) c.ids.size() == L) { c.used = ++check_clock; return true; }
ConvCheckpoint c;
c.ids.assign(cur.begin(), cur.begin() + L);
c.imgs = imgs_below(req_imgs, L);
if (parts != nullptr) {
if (parts->size() != stages.size() + 1) return false;
c.gdn = std::move((*parts)[0].gdn);
c.ple = std::move((*parts)[0].ple);
c.tails = std::move((*parts)[0].tails);
c.dead = std::move((*parts)[0].dead);
c.block_pos = std::move((*parts)[0].block_pos);
for (size_t i = 1; i < parts->size(); ++i) c.stage_parts.push_back(std::move((*parts)[i]));
} else {
if (cudaDeviceSynchronize() != cudaSuccess || !checkpoint_save(c, ss, g)) return false;
for (auto& st : stages) { // a layer split's later stages: their sessions' part
const strata::core::OnDevice on(st->dev);
ConvCheckpoint part;
part.ids = c.ids;
if (cudaDeviceSynchronize() != cudaSuccess || !checkpoint_save(part, st->ss, g)) return false;
c.stage_parts.push_back(std::move(part));
}
}
c.used = ++check_clock;
checks.push_back(std::move(c));
while ((int) checks.size() > o.prompt_cache) {
std::vector<uint64_t> stamps;
stamps.reserve(checks.size());
for (const ConvCheckpoint& k : checks) stamps.push_back(k.used);
const size_t victim = strata::program::conv_cache::eviction_victim(stamps.data(), stamps.size(),
o.prompt_cache);
checks.erase(checks.begin() + (std::ptrdiff_t) victim);
}
return true;
};
sp.on_chunk = [&](const float* R_rows, int64_t T, int64_t p0, std::string& e) -> bool {
std::vector<int32_t> nxt((size_t) T);
for (int64_t t = 0; t < T; ++t) nxt[(size_t) t] = (int32_t) cur[(size_t) (p0 + t + 1)];
// E-9: batched through the prompt path when it can (one GPU: a layer split's drafter is on the last stage)
const bool batched = !multi_gpu && sp.draft_kv(mtp, R_rows, nxt.data(), T, p0, e);
if (!e.empty() || (!batched && !mtp.prefill(R_rows, nxt.data(), T, p0, e))) return false;
if (std::getenv("STRATA_SNAPSHOT_VERIFY") != nullptr)
std::fprintf(stderr, "strata serve: DRAFT_PREFILL path=%s mode=%d cells=%lld\n",
batched ? "batched" : "token", mtp.kv_state().kv_mode, (long long) T);
// progress for the server window: PP <position reached> <prompt tokens> <ms> <fresh tokens/s>
const int64_t done = p0 + T;
pp_reached = done;
const double ms = std::chrono::duration<double, std::milli>(Clock::now() - pp_t0).count();
std::printf("PP %lld %lld %.0f %.1f\n", (long long) done, (long long) pp_total, ms,
ms > 0.0 ? 1000.0 * (double) (done - pp_from) / ms : 0.0);
strata::core::progress_at("reading the prompt (batched), done up to token", done);
strata::core::progress_beat();
std::fflush(stdout);
if (o.prompt_cache_every > 0 && done >= pp_next_check) {
bool saved = false;
if (multi_gpu) { // the stages' parts, saved when each of them read this chunk
std::vector<ConvCheckpoint> parts;
{
std::lock_guard<std::mutex> lk(part_mu);
auto it = part_at.find(done);
if (it != part_at.end()) parts = std::move(it->second);
part_at.erase(part_at.begin(), part_at.upper_bound(done));
}
bool complete = parts.size() == stages.size() + 1;
for (const ConvCheckpoint& k : parts) complete = complete && !k.gdn.empty();
saved = !complete || checkpoint_at(done, &parts); // an incomplete set: no checkpoint here
} else {
saved = checkpoint_at(done);
}
if (!saved) { e = "saving a conversation checkpoint failed"; return false; }
pp_next_check = done + o.prompt_cache_every;
}
return true;
};
if (multi_gpu) { // the batched prompt is reported by its last stage (the drafter's rows are there)
stages.back()->sp.on_chunk = std::move(sp.on_chunk);
sp.on_chunk = nullptr;
for (size_t i = 0; i <= stages.size(); ++i) {
strata::prefill::Prefill& stage_sp = i == 0 ? sp : stages[i - 1]->sp;
strata::core::SessionState& stage_ss = i == 0 ? ss : stages[i - 1]->ss;
stage_sp.on_stage_chunk = [&, i](int64_t done, std::string& e) -> bool {
if (o.prompt_cache <= 0 || o.prompt_cache_every <= 0 || done < part_next[i]) return true;
part_next[i] = done + o.prompt_cache_every;
ConvCheckpoint part; // this stage's state at `done` (its stream is synchronized)
part.ids.assign(cur.begin(), cur.begin() + done);
if (!checkpoint_save(part, stage_ss, g)) { e = "saving a checkpoint part failed"; return false; }
std::lock_guard<std::mutex> lk(part_mu);
auto& v = part_at[done];
v.resize(stages.size() + 1);
v[i] = std::move(part);
return true;
};
}
}
drive.d.plan = ver.plan_sink();
drive.d.pcie_num = std::max(0, std::min(256, (int) (o.pcie_frac * 256.0 + 0.5)));
if (o.adapt_every > 0 && o.adapt_swaps > 0) drive.d.usage.assign((size_t) (g.n_layers * g.n_expert), 0.0f);
// #477 --expert-profile-save: what the adaptive tier learned, kept across restarts (opt-in; off: `heat` stays
// empty and nothing below runs). It needs the adaptive tier's counts and the residency table.
std::vector<double> heat;
if (!o.expert_profile_save.empty()) {
if (drive.d.usage.empty() || host_res.empty())
std::fprintf(stderr, "strata serve: --expert-profile-save needs the adaptive tier (--adapt-every and "
"--adapt-swaps above 0) and --expert-profile: nothing will be saved\n");
else
heat.assign(drive.d.usage.size(), 0.0);
}
Clock::time_point profile_saved_at = Clock::now();
cudaStream_t adapt_stream = nullptr;
if (cudaStreamCreateWithFlags(&adapt_stream, cudaStreamNonBlocking) != cudaSuccess) {
std::fprintf(stderr, "strata serve: cannot create the refill stream\n");
return 1;
}
// plan v0.3 P6: swaps in flight - (residency index, slot) admitted when adapt_ev has completed
std::vector<std::pair<int32_t, int32_t>> pending;
cudaEvent_t adapt_ev = nullptr;
cudaEventCreateWithFlags(&adapt_ev, cudaEventDisableTiming);
// a layer split's later stages keep a copy of the residency table on their devices, and swap on their own
auto res_upload = [&]() {
if (d_res != nullptr)
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
for (auto& st : stages) {
const strata::core::OnDevice on(st->dev);
cudaMemcpy(st->d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
}
};
auto apply_pending = [&](bool wait) {
if (peer.valid()) peer.apply_pending(wait);
if (pending.empty()) return;
if (wait) cudaEventSynchronize(adapt_ev);
else if (cudaEventQuery(adapt_ev) != cudaSuccess) return;
for (auto& st : stages)
if (st->adapt_live) {
if (wait) cudaEventSynchronize(st->adapt_ev);
else if (cudaEventQuery(st->adapt_ev) != cudaSuccess) return;
}
for (auto& st : stages) st->adapt_live = false;
src.commit_exchanges(); // the resident RAM mode: the evicted experts take their places in RAM
for (const auto& [i, slot] : pending) host_res[(size_t) i] = slot;
pending.clear();
res_upload();
};
// the VRAM tier follows the conversation (the same rule as the speculative loop below)
auto adapt = [&]() -> bool {
if (!pending.empty()) return true; // the previous swaps are still in flight
struct Swap { float gain; int32_t layer, in, out; };
std::vector<Swap> swaps;
std::vector<std::pair<float, int32_t>> cand, vict;
for (int64_t l = 0; l < g.n_layers; ++l) {
cand.clear();
vict.clear();
const float* u = drive.d.usage.data() + l * g.n_expert;
const int32_t* r = host_res.data() + l * g.n_expert;
for (int32_t e = 0; e < (int32_t) g.n_expert; ++e) {
if (r[e] < 0) { if (u[e] >= 2.0f && !(peer.valid() && peer.has(l, e))) cand.emplace_back(u[e], e); }
else vict.emplace_back(u[e], e);
}
if (cand.empty() || vict.empty()) continue;
std::sort(cand.begin(), cand.end(), [](auto& a, auto& b) { return a.first > b.first; });
const size_t nc = std::min(cand.size(), vict.size());
std::partial_sort(vict.begin(), vict.begin() + (ptrdiff_t) nc, vict.end(),
[](auto& a, auto& b) { return a.first < b.first; });
for (size_t i = 0; i < nc; ++i) {
if (cand[i].first < vict[i].first + 1.5f) break;
swaps.push_back({cand[i].first - vict[i].first, (int32_t) l, cand[i].second, vict[i].second});
}
}
std::sort(swaps.begin(), swaps.end(), [](const Swap& a, const Swap& b) { return a.gain > b.gain; });
if ((int) swaps.size() > o.adapt_swaps) swaps.resize((size_t) o.adapt_swaps);
if (!resident_stage_swaps(src, xcache, host_res, g.n_expert, swaps, adapt_stream)) return false;
bool main_live = false;
for (const Swap& s : swaps) {
const size_t in = (size_t) s.layer * g.n_expert + s.in, out = (size_t) s.layer * g.n_expert + s.out;
const int32_t slot = host_res[out];
const uint8_t* b = srcp->blob(s.layer, s.in);
const int stn = multi_gpu ? stage_of(s.layer) : 0; // the swap stays in the layer's own cache
GpuStage* gs = stn > 0 ? stages[(size_t) stn - 1].get() : nullptr;
const strata::core::OnDevice on(gs ? gs->dev : -1);
if (slot < 0 || b == nullptr ||
cudaMemcpyAsync(gs ? gs->cache.device_slot(slot) : xcache.device_slot(slot), b,
(size_t) strata::kernels::cpu::expert_layout().blob_bytes(s.layer),
cudaMemcpyHostToDevice, gs ? gs->adapt_stream : adapt_stream) != cudaSuccess)
return false;
if (gs) gs->adapt_live = true;
else main_live = true;
host_res[out] = strata::core::kNotResident; // evicted now: the CPU computes it meanwhile
pending.emplace_back((int32_t) in, slot); // resident once the copy has landed
}
if (!swaps.empty()) cudaEventRecord(adapt_ev, adapt_stream);
(void) main_live;
for (auto& st : stages)
if (st->adapt_live) {
const strata::core::OnDevice on(st->dev);
cudaEventRecord(st->adapt_ev, st->adapt_stream);
}
// #477: the routing counted since the start (each count adds up to 1 / (1 - --adapt-decay) over its
// decays: the sum is proportional to the routing itself) - only with --expert-profile-save, else `heat`
// is empty
for (size_t i = 0; i < heat.size(); ++i) heat[i] += (double) drive.d.usage[i];
if (peer.valid()) { // multi-GPU: the peer takes the next most-routed CPU misses
std::vector<int32_t> r0 = host_res;
for (const auto& [i, slot] : pending) r0[(size_t) i] = slot; // swapped into the primary already
std::string perr;
if (!peer.adapt(drive.d.usage.data(), r0.data(),
o.peer_adapt_swaps >= 0 ? o.peer_adapt_swaps : o.adapt_swaps, perr)) {
std::fprintf(stderr, "strata serve: %s\n", perr.c_str());
return false;
}
}
for (float& v : drive.d.usage) v *= o.adapt_decay;
return true;
};
// #477: write the learned profile (between requests and at QUIT: a prompt's lent slots are back by then).
// A swap still in flight counts as done - its expert is resident once the copy lands. `why`: for the log.
auto save_profile = [&](const char* why) {
if (heat.empty()) return;
std::vector<uint8_t> resident(host_res.size(), 0);
for (size_t i = 0; i < host_res.size(); ++i) resident[i] = host_res[i] >= 0;
for (const auto& p : pending) resident[(size_t) p.first] = 1;
std::string e;
const auto ranked = strata::core::rank_learned_profile(g.n_layers, g.n_expert, resident, heat,
profile_loaded);
if (strata::core::write_expert_profile(o.expert_profile_save, g.n_layers, g.n_expert, ranked, e))
std::fprintf(stderr, "strata serve: expert profile saved to %s (%s)\n", o.expert_profile_save.c_str(),
why);
else
std::fprintf(stderr, "strata serve: the expert profile was not saved: %s\n", e.c_str());
profile_saved_at = Clock::now();
};
// stdin is read on its own thread, so a STOP line reaches a request that is still running (the client went
// away, or pressed Esc): the flag is checked between prompt chunks and between verify windows.
std::atomic<bool> stop_req{false};
std::mutex in_mu;
std::condition_variable in_cv;
std::deque<std::string> in_lines;
bool in_eof = false;
std::thread([&] {
// read(2) on the descriptor, not std::cin: glibc's exit() flushes every stdio stream and waits for
// stdin's lock, which getline holds while it waits for input - an engine ending on an error (every
// std::exit) would hang in exit() on Linux, and the server would wait for it forever
std::string l, buf;
char chunk[4096];
auto getline_fd = [&](std::string& out) -> bool {
for (;;) {
const size_t nlpos = buf.find('\n');
if (nlpos != std::string::npos) {
out.assign(buf, 0, nlpos);
buf.erase(0, nlpos + 1);
return true;
}
#if defined(_WIN32)
const int n = _read(0, chunk, (unsigned) sizeof chunk);
#else
const ssize_t n = ::read(0, chunk, sizeof chunk);
if (n < 0 && errno == EINTR) continue;
#endif
if (n <= 0) {
if (buf.empty()) return false;
out.swap(buf);
buf.clear();
return true;
}
buf.append(chunk, (size_t) n);
}
};
while (getline_fd(l)) {
if (!l.empty() && l.back() == '\r') l.pop_back();
if (l == "STOP") { stop_req.store(true); continue; }
std::lock_guard<std::mutex> lk(in_mu);
in_lines.push_back(l);
in_cv.notify_one();
}
std::lock_guard<std::mutex> lk(in_mu);
in_eof = true;
in_cv.notify_one();
}).detach();
auto next_line = [&](std::string& out) -> bool {
std::unique_lock<std::mutex> lk(in_mu);
in_cv.wait(lk, [&] { return !in_lines.empty() || in_eof; });
if (in_lines.empty()) return false;
out = std::move(in_lines.front());
in_lines.pop_front();
return true;
};
sp.should_stop = [&] { return stop_req.load(); };
// STRATA_TRACE=1: one stderr line per step of a request (the log shows where a request stops)
const bool trace = std::getenv("STRATA_TRACE") != nullptr;
auto tr = [&](const char* what, long long a = -1, long long b = -1) {
if (!trace) return;
std::fprintf(stderr, "strata trace: %s %lld %lld\n", what, a, b);
std::fflush(stderr);
};
{
// what is left once everything is allocated: under WDDM a GPU filled to the brim does not fail, it pages -
// and a page-in while the verify graph spins on a host flag stalls the request for good
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
// below ~256 MiB a later allocation (a first-used window's buffers, the desktop, another program) can make
// the driver page GPU memory, and a verify graph spinning on a host flag then never finishes
const int64_t free_mib = (int64_t) (free_b >> 20);
if (free_mib >= 256) {
std::fprintf(stderr, "strata serve: %lld MiB of VRAM free with everything loaded\n", (long long) free_mib);
} else if (reserve_adapted) {
// #496: the reserve was already lowered to make the cache fit - a bigger one would leave it no room
std::fprintf(stderr, "strata serve: WARNING: %lld MiB of VRAM free with everything loaded - this card "
"only just fits the model (the VRAM reserve was lowered to %d MiB so the expert "
"cache fits): requests may stall. Close other programs that use the GPU, or lower "
"--max-context\n", (long long) free_mib, o.vram_reserve_mib);
} else {
std::fprintf(stderr, "strata serve: %lld MiB of VRAM free with everything loaded - LOW: requests may stall;"
" add --vram-reserve-mib %lld to the config's args (or lower --max-context)\n",
(long long) free_mib, (long long) (o.vram_reserve_mib + 512 - free_mib));
}
}
// what the server's Monitor tab shows (servers before 0.1.8 skip unknown lines until READY)
{
size_t free_b = 0, total_b = 0;
cudaMemGetInfo(&free_b, &total_b);
// "Experts in VRAM" is every tier, not one card's. This used to report `xcache` alone, so a layer split
// showed CUDA0's cache as if it were the whole GPU's: a 4-GPU 256K run read 3327 experts when the four
// cards held 13320, and the Monitor tab was wrong by 4x for every multi-GPU config. The tiers are
// disjoint by construction (remote_experts.cpp skips any pair a stage already claimed), so they add.
const int64_t slots_primary = (int64_t) xcache.slots();
const int64_t mib_primary = (int64_t) (xcache.bytes() >> 20);
int64_t slots_all = slots_primary, mib_all = mib_primary;
for (const auto& st : stages) {
slots_all += (int64_t) st->cache.slots();
mib_all += (int64_t) (st->cache.bytes() >> 20);
}
for (int r = 0; r < 3; ++r)
if (o.expert_cache_remote[(size_t) r] > 0) {
slots_all += remote_experts[(size_t) r].resident();
mib_all += (int64_t) (remote_experts[(size_t) r].gib() * 1024.0);
}
std::printf("INFO context=%lld kv=%s kv_resident=%lld expert_slots=%lld expert_cache_mib=%lld "
"expert_slots_primary=%lld expert_cache_primary_mib=%lld spec=%d "
"mtp_max=%d lookup=%d vram_free_mib=%lld cvec=%s arena_mib=%lld pool_workers=%d pcie_frac=%.2f "
"spec_min_p=%.2f conversation_cache_mib=%lld conversation_cache_slots=%d "
"conversation_cache_min_free_mib=%lld engine=" STRATA_VERSION "\n",
(long long) o.max_context, o.kv.c_str(),
(long long) (g.n_qsa_layers() > 0 && ss.qsa_states[ss.qsa_primary()].kv_mode == 1
? ss.qsa_states[ss.qsa_primary()].n_slots * 4 : 0),
(long long) slots_all, (long long) mib_all,
(long long) slots_primary, (long long) mib_primary,
o.spec, o.mtp_max_t,
o.suffix_draft, (long long) (free_b >> 20), cvec_summary.c_str(),
(long long) ((o.mmap_experts ? src.resident_bytes() : strata::kernels::cpu::expert_layout().total) >> 20),
pool.workers(), o.pcie_frac,
o.spec_min_p, (long long) o.conversation_cache_mib, o.conversation_cache_slots,
(long long) o.conversation_cache_min_free_mib);
}
// issue #29: a request whose heartbeat (tokens, prompt chunks, verify windows) stops for this long is stuck on
// a flag nobody will raise - end the engine with where it was, so the server starts it again instead of the
// GPU spinning forever. STRATA_WATCHDOG_S=0 turns it off. Issue #31: before it does, it reports what every
// part was doing (stall_report), so one occurrence says where the wait is.
{
const char* ws = std::getenv("STRATA_WATCHDOG_S");
const int limit = ws ? std::atoi(ws) : 60; // one step (a prompt layer, a verify window) takes seconds
if (limit > 0)
std::thread([limit] {
strata::core::Progress& p = strata::core::progress();
uint64_t last = p.beats.load(), ticks_at = p.ticks.load();
auto since = std::chrono::steady_clock::now();
for (;;) {
std::this_thread::sleep_for(std::chrono::seconds(1));
const auto now = std::chrono::steady_clock::now();
const uint64_t b = p.beats.load();
if (!p.busy.load() || b != last) { last = b; ticks_at = p.ticks.load(); since = now; continue; }
if (now - since < std::chrono::seconds(limit)) continue;
std::fprintf(stderr, "strata serve: no progress for %d s during a request (%s) - stopping "
"the engine so the server starts it again (issue #29)\n",
limit, stage_text().c_str());
stall_report(stderr, p.ticks.load() - ticks_at);
strata::core::release_gpu_waits(stderr); // #267: no spin kernel outlives the process
std::fflush(stderr);
std::abort();
}
}).detach();
}
std::printf("READY %lld stop\n", (long long) o.max_context); // "stop": this engine honours STOP
std::fflush(stdout);
std::string line;
int64_t rounds = 0;
const int S = o.spec;
const int S_mtp = o.mtp_max_t > 0 ? std::min(o.mtp_max_t, S) : S; // the MTP's windows; suffixes go up to S
if (S_mtp < S) mtp.set_max_drafts(S_mtp - 1);
strata::spec::SuffixDrafter sfx(std::max(1, o.suffix_draft), 64, (size_t) o.max_context + 4096);
strata::spec::DraftPolicy policy(S); // MTP or lookup window, learned over the whole process
// The vision path (--vision): GENI <max_new> <embeddings file> <id,id,...> carries images. The file is one
// or more strata-vision records (int32 'SVE1', n, nx, ny, n_embd, then n x n_embd floats) in prompt order;
// each image's rows go to its run of <|image_pad|> tokens, whose M-RoPE positions are mtmd's: t = p,
// h = p + y, w = p + x, and the text after the image continues at p + max(nx, ny).
constexpr int64_t kImagePad = 248056; // qwen4exp.ple.image_token_id: the PLE hash reads it for image cells
bool mrope_identity = true;
std::vector<float> img_rows;
std::vector<const float*> row_ptr;
while (next_line(line)) {
// #477: every --expert-profile-save-every minutes, before the next request (at QUIT: after the loop)
if (!heat.empty() && line != "QUIT" && o.expert_profile_save_min > 0 &&
Clock::now() - profile_saved_at >= std::chrono::duration<double>(o.expert_profile_save_min * 60.0))
save_profile("periodic");
if (line == "QUIT") break;
// the watchdog watches a request from here until this iteration ends, whichever way it ends
struct BusyScope {
BusyScope() { strata::core::progress().busy.store(true); strata::core::progress_at("request"); }
~BusyScope() { strata::core::progress().busy.store(false); strata::core::progress_at("idle"); }
} busy_scope;
stop_req.store(false); // a STOP that arrived between requests is stale
err.clear();
const bool geni = line.rfind("GENI ", 0) == 0;
if (!geni && line.rfind("GEN ", 0) != 0) {
std::printf("ERR expected: GEN <max_new> <id,id,...> or GENI <max_new> <file> <id,id,...>\n");
continue;
}
char* endp = nullptr;
const long long max_new = std::strtoll(line.c_str() + (geni ? 5 : 4), &endp, 10);
// optional sampling keys between max_new and the ids: temperature=F, top_p=F, top_k=N, min_p=F,
// penalty_last_n=N, penalty_repeat=F, penalty_freq=F, penalty_present=F, seed=N (text requests
// only). Absent keys keep today's behavior: greedy, no penalties.
float req_temperature = 0.0f, req_top_p = 1.0f;
int req_top_k = 20; // the sampler's own default; the sampled path REQUIRES top_k in 1..64
unsigned long long req_seed = 0;
float req_min_p = 0.0f, req_penalty_repeat = 1.0f, req_penalty_freq = 0.0f, req_penalty_present = 0.0f;
int req_penalty_last_n = 0;
int req_cvec = 1; // cvec=0|1: a loaded control vector for this request (on when absent)
// tuning keys (setup's calibration measures settings without restarting the engine): the PCIe share of
// the missed experts and the draft-probability floor, for this request only
double req_pcie_frac = o.pcie_frac, req_spec_min_p = o.spec_min_p;
if (endp != nullptr) { // GENI takes the same keys (#75: image requests were always greedy); its
// embedding file path is the first token without an =
for (;;) {
while (*endp == ' ') ++endp;
const char* start = endp;
while (*endp != '\0' && *endp != ' ') ++endp;
if (endp == start) break;
const std::string tok(start, (size_t) (endp - start));
const size_t eq = tok.find('=');
if (eq == std::string::npos) { endp = const_cast<char*>(start); break; }
const std::string key = tok.substr(0, eq);
const float fv = std::strtof(tok.c_str() + eq + 1, nullptr);
if (key == "cvec") req_cvec = std::atoi(tok.c_str() + eq + 1);
else if (key == "temperature") req_temperature = fv;
else if (key == "top_p") req_top_p = fv;
else if (key == "top_k") req_top_k = std::atoi(tok.c_str() + eq + 1);
else if (key == "min_p") req_min_p = fv;
else if (key == "penalty_last_n") req_penalty_last_n = std::atoi(tok.c_str() + eq + 1);
else if (key == "penalty_repeat") req_penalty_repeat = fv;
else if (key == "penalty_freq") req_penalty_freq = fv;
else if (key == "penalty_present") req_penalty_present = fv;
else if (key == "seed") req_seed = std::strtoull(tok.c_str() + eq + 1, nullptr, 10);
else if (key == "pcie_frac") req_pcie_frac = std::clamp((double) fv, 0.0, 1.0);
else if (key == "spec_min_p") req_spec_min_p = std::clamp((double) fv, 0.0, 1.0);
// unknown keys are skipped: the ids start at the first token without '='
}
}
std::string emb_path;
if (geni && endp != nullptr) {
while (*endp == ' ') ++endp;
char* gap = std::strchr(endp, ' ');
if (gap != nullptr) { emb_path.assign(endp, (size_t) (gap - endp)); endp = gap; }
}
std::vector<int64_t> ids;
std::string pe;
if (max_new < 1 || endp == nullptr || (geni && emb_path.empty()) || !parse_i64_list(endp, ids, pe)) {
std::printf("ERR bad request: %s\n", pe.empty() ? "max_new" : pe.c_str());
continue;
}
const int64_t n = (int64_t) ids.size();
req_imgs.clear();
if (geni && !o.vision) { std::printf("ERR this engine was started without --vision\n"); continue; }
if (geni || !mrope_identity) {
// positions for every cell this request can reach; the identity again for a text request
std::string ve;
row_ptr.assign((size_t) n, nullptr);
const int64_t cells = (int64_t) mrope_host.size() / 3;
auto put = [&](int64_t c, int64_t t, int64_t h, int64_t w) {
mrope_host[(size_t) c * 3] = (int32_t) t;
mrope_host[(size_t) c * 3 + 1] = (int32_t) h;
mrope_host[(size_t) c * 3 + 2] = (int32_t) w;
};
if (!geni) {
for (int64_t c = 0; c < cells; ++c) put(c, c, c, c);
} else {
struct Img { int64_t n, nx, ny; size_t off; };
std::vector<Img> imgs;
img_rows.clear();
std::FILE* f = std::fopen(emb_path.c_str(), "rb");
if (!f) ve = "cannot open " + emb_path;
while (f && ve.empty()) {
int32_t hdr[5];
const size_t got = std::fread(hdr, sizeof(int32_t), 5, f);
if (got == 0) break;
if (got != 5 || hdr[0] != 0x31455653 || hdr[1] < 1 || hdr[2] < 1 || hdr[3] < 1 ||
(int64_t) hdr[2] * hdr[3] != hdr[1] || hdr[4] != (int32_t) g.n_embd) {
ve = "bad embeddings file (expected strata-vision records of width " +
std::to_string((long long) g.n_embd) + ")";
break;
}
const size_t off = img_rows.size(), cnt = (size_t) hdr[1] * (size_t) hdr[4];
img_rows.resize(off + cnt);
if (std::fread(img_rows.data() + off, sizeof(float), cnt, f) != cnt) { ve = "short embeddings file"; break; }
imgs.push_back({hdr[1], hdr[2], hdr[3], off});
}
if (f) std::fclose(f);
int64_t p = 0, i = 0;
size_t k = 0;
while (ve.empty() && i < n) {
if (ids[(size_t) i] != kImagePad) { put(i, p, p, p); ++p; ++i; continue; }
if (k >= imgs.size()) { ve = "the prompt has more images than the embeddings file"; break; }
const Img& im = imgs[k++];
{ // what the conversation cache compares: a picture is its grid and its embeddings
const int64_t grid[3] = {im.n, im.nx, im.ny};
uint64_t h = fnv1a(grid, sizeof grid);
h = fnv1a(img_rows.data() + im.off, (size_t) im.n * (size_t) g.n_embd * sizeof(float), h);
req_imgs.push_back({i, h});
}
for (int64_t j = 0; j < im.n && ve.empty(); ++j)
if (i + j >= n || ids[(size_t) (i + j)] != kImagePad)
ve = "image " + std::to_string(k) + " has " + std::to_string((long long) im.n) +
" rows but fewer <|image_pad|> tokens";
for (int64_t j = 0; j < im.n && ve.empty(); ++j) {
const int64_t y = j / im.nx, x = j % im.nx;
put(i + j, p, p + y, p + x);
row_ptr[(size_t) (i + j)] = img_rows.data() + im.off + (size_t) j * (size_t) g.n_embd;
}
i += im.n;
p += std::max(im.nx, im.ny);
}
if (ve.empty() && k != imgs.size()) ve = "the embeddings file has more images than the prompt";
if (ve.empty() && n > 0 && ids[(size_t) (n - 1)] == kImagePad) ve = "the prompt cannot end in an image";
for (int64_t c = n; ve.empty() && c < cells; ++c) put(c, p + (c - n), p + (c - n), p + (c - n));
}
tr("positions built", (long long) img_rows.size());
cudaDeviceSynchronize();
tr("device idle");
// CUDA0's table and, with a layer split, every later stage's (each device reads its own)
auto upload_mrope = [&]() -> bool {
bool ok = cudaMemcpy(d_mrope, mrope_host.data(), mrope_host.size() * sizeof(int32_t),
cudaMemcpyHostToDevice) == cudaSuccess;
for (auto& st : stages) {
const strata::core::OnDevice on(st->dev);
cudaDeviceSynchronize();
ok = ok && cudaMemcpy(st->mrope, mrope_host.data(), mrope_host.size() * sizeof(int32_t),
cudaMemcpyHostToDevice) == cudaSuccess;
}
return ok;
};
if (ve.empty() && !upload_mrope()) ve = "the image position upload failed";
if (!ve.empty()) {
// leave the table as the identity so the next text request is untouched
for (int64_t c = 0; c < cells; ++c) put(c, c, c, c);
upload_mrope();
mrope_identity = true;
std::printf("ERR %s\n", ve.c_str());
std::fflush(stdout);
continue;
}
mrope_identity = !geni;
}
sp.embd_rows = geni ? row_ptr.data() : nullptr;
if (n + max_new + 8 > o.max_context) {
std::printf("ERR prompt (%lld tokens) + max_new (%lld) exceeds the context (%lld)\n", (long long) n,
(long long) max_new, (long long) o.max_context);
continue;
}
bool bad = false;
for (int64_t t : ids) bad = bad || t < 0 || t >= n_vocab;
if (bad) { std::printf("ERR a token id is outside the vocabulary\n"); continue; }
std::array<int64_t, 3> remote_before{};
std::array<int64_t, 3> launches_before{};
std::array<uint64_t, 3> compact_before{}, full_before{};
// ms_begin/ms_wait are cumulative since boot; the log line used to print them next to per-request deltas,
// so the host time read as if it belonged to this request. Take deltas here like every other column.
std::array<double, 3> begin_before{}, wait_before{};
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0)
{
remote_before[(size_t) r] = remote_experts[(size_t) r].computed();
launches_before[(size_t) r] = remote_experts[(size_t) r].launched_layers();
compact_before[(size_t) r] = remote_experts[(size_t) r].returned_bytes();
full_before[(size_t) r] = remote_experts[(size_t) r].full_row_bytes();
begin_before[(size_t) r] = remote_experts[(size_t) r].ms_begin();
wait_before[(size_t) r] = remote_experts[(size_t) r].ms_wait();
}
cur = ids;
const Clock::time_point r0 = Clock::now();
// ---- where this request starts reading: the live session, or a checkpoint, whose tokens AND pictures are
// exactly the start of this prompt - at most n - 1 of them, the last token is always the first window
auto starts_with = [&](const std::vector<int32_t>& pre, const std::vector<ImgKey>& pre_imgs) -> bool {
const int64_t L = (int64_t) pre.size();
if (L < 1 || L > n - 1) return false;
for (int64_t i = 0; i < L; ++i)
if ((int32_t) ids[(size_t) i] != pre[(size_t) i]) return false;
return imgs_below(req_imgs, L) == pre_imgs;
};
const bool want_cvec = strata::kernels::cvec().loaded() ? req_cvec != 0 : true;
// the last request's final commit may still be running on the verifier's stream (set_commit_async):
// everything below reads, restores or zeroes the session from other streams and the host (the end of the
// last request waited already; this covers a request that ended on an error path)
if (!ver.wait_commit(err)) {
std::printf("ERR %s\n", err.c_str());
return 1;
}
int64_t resume = 0;
bool from_live = false;
if (o.prompt_cache > 0 && want_cvec == cvec_cached) {
if (live_ok && starts_with(live, live_imgs)) { resume = (int64_t) live.size(); from_live = true; }
for (const ConvCheckpoint& c : checks)
if ((int64_t) c.ids.size() > resume && starts_with(c.ids, c.imgs)) {
resume = (int64_t) c.ids.size();
from_live = false;
}
}
const auto parked = conversations.best(ids, req_imgs, want_cvec);
std::optional<strata::core::SavedConversation> incoming;
if (parked.tokens > resume) incoming.emplace(conversations.take(parked.index));
// Reject the entire image before parking/overwriting the outgoing
// state. Invalid entries can safely fall back to its existing prefix.
if (incoming && !strata::core::conversation_snapshot_validate(*incoming, ss, g, mtp.kv_state(), err)) {
std::fprintf(stderr, "strata serve: conversation cache: discard invalid snapshot (%s)\n", err.c_str());
incoming.reset();
err.clear();
}
// Preserve the outgoing branch before any checkpoint rewind, reset,
// or incoming restore overwrites the positional state it requires.
if ((!from_live || incoming) && !park_current(incoming ? incoming->bytes() : 0)) {
std::printf("ERR %s\n", err.c_str());
return 1;
}
if (incoming) {
const auto t0 = Clock::now();
if (strata::core::conversation_snapshot_restore(*incoming, ss, g, mtp.kv_state(), err) !=
strata::core::ConversationRestore::restored) {
// Already prevalidated above: a failure here is fatal, never
// permission to decode from a partially restored session.
std::printf("ERR restoring parked conversation: %s\n", err.c_str());
return 1;
}
if (std::getenv("STRATA_SNAPSHOT_VERIFY") != nullptr) {
uint64_t draft_hash = 0;
if (!strata::core::conversation_kv_verify(incoming->kv.back(), mtp.kv_state(), g,
int64_t(incoming->live.ids.size()), false, draft_hash, err)) {
std::printf("ERR verifying restored draft KV: %s\n", err.c_str());
return 1;
}
std::fprintf(stderr, "strata serve: SNAPSHOT_VERIFY draft=%016llx cells=%lld mode=%d source=%s resident=%lld\n",
(unsigned long long) draft_hash, (long long) incoming->kv.back().cells,
mtp.kv_state().kv_mode, "ram",
(long long) (mtp.kv_state().n_slots * strata::kernels::qsa_real_shapes().page_size));
}
live = std::move(incoming->live.ids);
live_imgs = std::move(incoming->live.imgs);
checks = std::move(incoming->checkpoints);
cvec_cached = incoming->cvec;
resume = parked.tokens;
from_live = parked.live;
if (std::getenv("STRATA_SNAPSHOT_FULL_CAPTURE") == nullptr)
conversations.retain(std::move(incoming->kv), int64_t(live.size()));
incoming.reset(); // Running-state/checkpoint copies are no longer needed.
std::fprintf(stderr, "strata serve: conversation cache: restored %lld tokens (%s) in %.1f ms; parked=%zu bytes=%zu\n",
(long long) resume, from_live ? "live" : "checkpoint",
std::chrono::duration<double, std::milli>(Clock::now() - t0).count(),
conversations.size(), conversations.bytes());
}
if (want_cvec != cvec_cached) {
live_ok = false;
checks.clear();
cvec_cached = want_cvec;
}
if (strata::kernels::cvec().loaded()) strata::kernels::cvec_set_enabled(want_cvec);
// this request rewrites every cell from `resume` on, so a checkpoint past it (or not on this prompt's
// path) no longer has its cells; the ones kept are prefixes of both the old tokens and the new
checks.erase(std::remove_if(checks.begin(), checks.end(), [&](const ConvCheckpoint& c) {
return (int64_t) c.ids.size() > resume || !starts_with(c.ids, c.imgs);
}), checks.end());
live_ok = false; // until this request has finished, the session is in between
int64_t reread_to = -1; // STRATA_CKPT_REREAD only: read [0, reread_to) again instead of restoring
if (resume == 0) {
strata::core::session_zero(ss, g, nullptr, main_cs);
cudaStreamSynchronize(main_stream);
for (auto& st : stages) {
const strata::core::OnDevice on(st->dev);
strata::core::session_zero(st->ss, g, nullptr, (void*) st->stream);
cudaStreamSynchronize(st->stream);
}
checks.clear();
} else if (!from_live) {
ConvCheckpoint* c = nullptr;
for (ConvCheckpoint& k : checks) if ((int64_t) k.ids.size() == resume) c = &k;
if (c != nullptr) c->used = ++check_clock; // mounting through it is the use LRU counts
static const bool reread = std::getenv("STRATA_CKPT_REREAD") != nullptr;
if (reread && c != nullptr) {
// THE CHECK OF THE CHECKPOINT: instead of restoring it, read its tokens again from position 0 in
// one run (below, with the prompt path's slots lent like any read) - the same chunks the request
// that saved it read them in, when that request started at 0. With the VRAM expert set fixed
// (--adapt-swaps 0) the answer must match the restored one token for token; anything the
// checkpoint missed shows up as a difference.
strata::core::session_zero(ss, g, nullptr, main_cs);
cudaStreamSynchronize(main_stream);
for (auto& st : stages) {
const strata::core::OnDevice on(st->dev);
strata::core::session_zero(st->ss, g, nullptr, (void*) st->stream);
cudaStreamSynchronize(st->stream);
}
reread_to = resume;
std::fprintf(stderr, "strata serve: STRATA_CKPT_REREAD: reading %lld tokens again instead of "
"restoring\n", (long long) resume);
} else if (c == nullptr || !checkpoint_restore(*c, ss, g) || c->stage_parts.size() != stages.size() ||
[&] {
for (size_t i = 0; i < stages.size(); ++i) {
const strata::core::OnDevice on(stages[i]->dev);
if (!checkpoint_restore(c->stage_parts[i], stages[i]->ss, g)) return true;
}
return false;
}()) {
std::printf("ERR restoring a conversation checkpoint failed\n");
return 1;
}
}
// KV streaming: the drafter's ring may hold cells past `resume` from a longer turn; the main layers'
// host copies and slots are always current (every writer writes both), so they need nothing
if (resume > 0 && reread_to <= 0) mtp.kv_restore(resume);
tr("request", n, geni ? 1 : 0);
mtp.set_prompt_len(n);
const int64_t read_from = reread_to > 0 ? 0 : resume;
conversations.limit_reuse(read_from);
pp_total = n;
pp_from = read_from;
pp_reached = read_from;
pp_t0 = r0;
pp_next_check = reread_to > 0 ? INT64_MAX : resume + o.prompt_cache_every;
{
std::lock_guard<std::mutex> lk(part_mu);
part_at.clear();
std::fill(part_next.begin(), part_next.end(), pp_next_check);
}
std::printf("RESUME %lld\n", (long long) resume); // before reading: this many prompt tokens are reused
strata::core::progress_at("reading the prompt, from token", read_from);
std::fflush(stdout);
// A SHORT PART OF THE PROMPT - the new message of a chat that continues from a checkpoint, the assistant
// header - goes through the verify windows, S tokens at a time, as decode reads them. The batched path
// costs ~300 ms per run however few tokens it has (it streams every expert the chunk routes to that is
// not in VRAM over PCIe), and it borrows slots it must refill after (~180 ms); a window costs ~16 ms a
// token, with the misses on the CPU. Each part below is decided on its own, so a long first message is
// read batched and its header still goes through the windows. Picture rows need the batched path.
// STRATA_CKPT_REREAD compares a restored checkpoint with a batched re-read, so it keeps every read batched.
static const bool no_short = std::getenv("STRATA_CKPT_REREAD") != nullptr;
auto windows_ok = [&](int64_t a, int64_t b) -> bool {
if (no_short || b - a > o.short_read) return false;
if (sp.embd_rows != nullptr)
for (int64_t i = a; i < b; ++i)
if (sp.embd_rows[i] != nullptr) return false;
return true;
};
// tokens [a, b) through the windows: commit all of them, then give the draft layer their residuals
auto read_windows = [&](int64_t a, int64_t b, std::string& e) -> bool {
strata::core::progress_at("reading the prompt (verify windows), from token", a); // #217: not "batched"
// every token is committed and the picks are discarded: no head sampling (see set_head_sampling)
struct NoHeadSampling {
strata::core::Verifier& v;
explicit NoHeadSampling(strata::core::Verifier& x) : v(x) { v.set_head_sampling(false); }
~NoHeadSampling() { v.set_head_sampling(true); }
} no_head_sampling(ver);
std::vector<int32_t> win((size_t) S), outw((size_t) S), nxt((size_t) S);
for (int64_t q = a; q < b;) {
if (stop_req.load()) { e = "cancelled"; return false; }
const int T = (int) std::min<int64_t>(S, b - q);
for (int t = 0; t < T; ++t) {
win[(size_t) t] = (int32_t) cur[(size_t) (q + t)];
nxt[(size_t) t] = (int32_t) cur[(size_t) (q + t + 1)];
}
drive.d.layers = 0;
drive.d.experts = 0;
drive.d.failed = false;
if (!ver.run(T, win.data(), q, win_pool_fn, win_pool_user, outw.data(), e) || drive.d.failed) {
if (drive.d.failed && drive.d.fail) e = drive.d.fail;
return false;
}
// STRATA_LOGPOS=<path>: the teacher-forced log-probability of every token read here (every
// token is committed and nxt[t] is the prompt's own next token), appended to <path>; the last
// column is the log-probability of STRATA_LOGPOS_EXTRA (default 248046, <|im_end|>), so the
// end-of-turn mass a chat model puts on raw text can be taken out of the measurement
static std::FILE* logpos = [] {
const char* p = std::getenv("STRATA_LOGPOS");
return p != nullptr ? std::fopen(p, "ab") : nullptr;
}();
static const int32_t logpos_extra = [] {
const char* p = std::getenv("STRATA_LOGPOS_EXTRA");
return p != nullptr ? (int32_t) std::atoi(p) : (int32_t) 248046;
}();
if (logpos != nullptr && !ver.window_logprobs(nxt.data(), T, q, logpos_extra, logpos, e))
return false;
if (!ver.commit(T, e) || !mtp.prefill(ver.final_R_all(), nxt.data(), T, q, e)) return false;
q += T;
pp_reached = q; // #471
}
// the batched prompt path (other streams), checkpoints and snapshots may follow: the last commit first
if (!ver.wait_commit(e)) return false;
const double ms = std::chrono::duration<double, std::milli>(Clock::now() - pp_t0).count();
std::printf("PP %lld %lld %.0f %.1f\n", (long long) b, (long long) pp_total, ms,
ms > 0.0 ? 1000.0 * (double) (b - pp_from) / ms : 0.0);
strata::core::progress_beat();
std::fflush(stdout);
return true;
};
// the batched path's slots are lent just before its first run and given back (refilled) before a window
// reads - so the windows always see the whole expert cache - or once the prompt is read
// Every participant gives its loan back here: the rows it lent are refilled into the SAME slots from
// the arena, the residency table is restored, and one upload puts it on every device. A stage refills
// through its own cache and its own device - a slot refilled into the wrong cache would leave that
// stage's cache holding an expert it does not own, which is silent and produces plausible tokens.
// (split in two halves so a layer split can queue every stage's copies before it waits for any: #340)
auto refill_issue = [&](PfPart& p, std::string& e) -> bool {
tr("refill start", (long long) p.lent.size());
const strata::core::OnDevice on(p.dev);
for (const auto& [i, slot] : p.lent) { // D-4: queued, one wait (STRATA_REFILL_BLOCKING=1: each)
const uint8_t* b = srcp->blob(i / g.n_expert, i % g.n_expert);
const int64_t nb = (int64_t) strata::kernels::cpu::expert_layout().blob_bytes(i / g.n_expert);
if (b == nullptr || !(refill_blocking() ? p.cache->fill_slot_blocking(slot, b, e, nb)
: p.cache->fill_slot_queued(slot, b, e, nb)))
return false;
host_res[(size_t) i] = slot;
}
return true;
};
auto refill_wait = [&](PfPart& p, std::string& e) -> bool {
const strata::core::OnDevice on(p.dev);
if (!p.cache->sync_queued(e)) return false;
p.lent.clear();
p.lent_chunk = 0;
return true;
};
auto refill_one = [&](PfPart& p, std::string& e) -> bool {
if (p.lent.empty()) return true;
return refill_issue(p, e) && refill_wait(p, e);
};
// #340: with several stages every stage's copies go out first (each card on its own link), then each is
// waited for - the same copies into the same slots, so the same cache; one stage: refill_one exactly.
// STRATA_REFILL_SERIAL=1: stage after stage, as 0.1.30/0.1.31.
static const bool refill_serial = std::getenv("STRATA_REFILL_SERIAL") != nullptr;
auto refill = [&](std::string& e) -> bool {
const auto t_rf = Clock::now();
int64_t n_lent = 0, n_parts = 0;
for (PfPart& p : pf_parts)
if (!p.lent.empty()) { n_lent += (int64_t) p.lent.size(); ++n_parts; }
if (n_parts > 1 && !refill_serial) {
for (PfPart& p : pf_parts)
if (!p.lent.empty() && !refill_issue(p, e)) return false;
for (PfPart& p : pf_parts)
if (!p.lent.empty() && !refill_wait(p, e)) return false;
} else {
for (PfPart& p : pf_parts)
if (!p.lent.empty() && !refill_one(p, e)) return false;
}
if (n_parts > 0) res_upload();
if (trace && n_parts > 0) {
std::fprintf(stderr, "strata trace: refilled %lld slots on %lld stage(s) in %.1f ms\n",
(long long) n_lent, (long long) n_parts,
std::chrono::duration<double, std::milli>(Clock::now() - t_rf).count());
std::fflush(stderr);
}
return true;
};
// lend the slots `tokens` batched prompt tokens need: the prompt path's buffers for min(chunk, tokens
// rounded up to 256), laid out in the last of the slots it may borrow - per participant, out of that
// participant's own cache, and marking only that participant's own layers
auto lend = [&](int64_t tokens, std::string& e) -> bool {
if (pf_parts.empty()) return true; // its own buffers: nothing to lend
const auto t_ln = Clock::now();
// what this segment needs, capped by the configured chunk: a request lends only what its own
// prompt needs, so a large chunk costs a short prompt nothing
const int64_t want_full = request_chunk(tokens, o.prefill_chunk);
// #340: a short enough request reads in the stages' own S-token chunks (nothing lent)
const int64_t want = split_small > 0 && tokens <= split_small_max ? std::min(want_full, split_small)
: want_full;
if (want <= 0) {
e = "prefill: cannot lend buffers for an empty request segment";
return false;
}
bool any = false;
for (PfPart& p : pf_parts) {
if (p.first < 0) continue;
if (!p.lent.empty()) {
if (want <= p.lent_chunk) continue; // its current loan already covers this
// ONLY this participant's loan goes back: `refill` would return the other participants'
// loans too, and their buffers are still laid out in their caches - marking those slots
// resident again would hand the next window a prompt buffer in place of an expert
if (!refill_one(p, e)) return false;
}
const strata::core::OnDevice on(p.dev);
const int32_t first = std::max<int32_t>(p.first, (int32_t) (p.cache->slots() - part_slots(p, want)));
if (want != p.sp->chunk() || first != p.first_now) {
if (!p.sp->relayout(want, p.cache->device_slot(first), part_bytes(p, first), e)) return false;
p.first_now = first;
}
for (int64_t l = p.lb; l < p.le; ++l) // THIS participant's layers only
for (int64_t ex = 0; ex < g.n_expert; ++ex) {
const size_t i = (size_t) (l * g.n_expert + ex);
if (host_res[i] >= first) {
p.lent.emplace_back((int32_t) i, host_res[i]);
host_res[i] = strata::core::kNotResident;
any = true;
}
}
p.lent_chunk = want;
}
if (any) res_upload();
if (trace) {
int64_t n_lent = 0;
for (const PfPart& p : pf_parts) n_lent += (int64_t) p.lent.size();
std::fprintf(stderr, "strata trace: lent %lld slots for %lld tokens in %.1f ms\n", (long long) n_lent,
(long long) want, std::chrono::duration<double, std::milli>(Clock::now() - t_ln).count());
std::fflush(stderr);
}
return true;
};
apply_pending(true);
// per-request sampling for the verify window's head (greedy when temperature is absent)
strata::kernels::SamplerParams req_sp;
req_sp.greedy = req_temperature <= 0.0f;
req_sp.temperature = req_temperature;
req_sp.top_p = req_top_p;
req_sp.top_k = req_top_k;
req_sp.seed = req_seed ? req_seed
: (unsigned long long) std::chrono::steady_clock::now().time_since_epoch().count();
req_sp.min_p = std::clamp(req_min_p, 0.0f, 1.0f);
req_sp.penalty_last_n = std::max(req_penalty_last_n, 0);
req_sp.penalty_repeat = req_penalty_repeat;
req_sp.penalty_freq = req_penalty_freq;
req_sp.penalty_present = req_penalty_present;
req_sp.counter = 0;
ver.set_sampling(req_sp);
mtp.set_draft_sampling(req_sp); // STRATA_SPEC_COUPLED=1: sampled drafts (a no-op otherwise)
drive.d.pcie_num = std::max(0, std::min(256, (int) (req_pcie_frac * 256.0 + 0.5)));
// a layer split: CUDA0's share as asked; a later GPU keeps its own (its link) unless the request sets one
for (int st = 0; st < split_drive.n; ++st)
split_drive.pcie_num[st] = (st == 0 || split_same || req_pcie_frac != o.pcie_frac)
? drive.d.pcie_num : pcie_num_of(stages[(size_t) st - 1]->pcie_frac);
const int hist_n = std::min(req_sp.penalty_last_n, kPenaltyWindowCap);
ver.set_history(hist_n > 0 ? d_hist : nullptr, hist_n);
bool cancelled = false;
tr("prompt start", n - 1);
// The prompt is read in two parts when it has a turn boundary past `resume`: up to the last <|im_start|>
// (the conversation so far), a checkpoint there, then the new turn's header. The next request of the same
// chat renders the same history - but not always the same header or the thinking of this reply - so that
// checkpoint is the one it reuses.
int64_t turn_at = -1;
if (o.prompt_cache > 0 && o.turn_token >= 0)
for (int64_t i = n - 1; i > resume; --i)
if (ids[(size_t) i] == o.turn_token) { turn_at = i; break; }
// A prompt read from token 0 also stops at its FIRST turn boundary: the end of the system prompt (with
// the tools), which every new chat of the same client shares. That checkpoint becomes the chain's root,
// which the retention policy pins (conv_cache.hpp), so the next new chat reads only what comes after it.
// (PR #65, code-martin.) Only for a system prompt of --prompt-cache-root tokens or more: a small one
// is cheaper to read again than the extra part costs (~0.3 s).
int64_t root_at = -1;
if (o.prompt_cache > 0 && o.turn_token >= 0 && o.prompt_cache_root > 0 && read_from == 0)
for (int64_t i = 1; i < turn_at; ++i)
if (ids[(size_t) i] == o.turn_token) {
if (i >= o.prompt_cache_root) root_at = i;
break;
}
int64_t at = read_from;
for (const int64_t to : {reread_to, root_at, turn_at, n - 1}) {
if (to <= at) continue;
err.clear();
const bool win = windows_ok(at, to);
if (win && !refill(err)) {
std::printf("ERR refilling a lent slot failed: %s\n", err.c_str());
return 1;
}
if (!win && !lend(to - at, err)) {
std::printf("ERR lending the prompt path its slots failed: %s\n", err.c_str());
return 1;
}
const auto tsp = Clock::now();
const bool sp_ok = win ? read_windows(at, to, err) : sp.run(ids.data() + at, to - at, at, err);
if (trace) {
std::fprintf(stderr, "strata trace: read %lld tokens (%s) in %.1f ms\n", (long long) (to - at),
win ? "windows" : "batched",
std::chrono::duration<double, std::milli>(Clock::now() - tsp).count());
std::fflush(stderr);
}
if (!sp_ok) {
if (!stop_req.load()) {
std::fprintf(stderr, "strata serve: %s\n", err.c_str());
std::printf("ERR %s\n", err.c_str());
// #224: a CUDA fault (an illegal address) poisons the context for the whole process, and
// unwinding the destructors on it could hang until the 60 s watchdog: leave at once
if (cudaPeekAtLastError() != cudaSuccess) {
std::fflush(stdout);
std::fflush(stderr);
std::_Exit(1);
}
return 1;
}
cancelled = true; // stopped while reading the prompt: refill the lent slots below, then DONE cancel
break;
}
at = to;
if ((to == turn_at || to == root_at) && !checkpoint_at(to)) {
std::printf("ERR saving a conversation checkpoint failed\n");
return 1;
}
}
if (!refill(err)) {
std::printf("ERR refilling a lent slot failed: %s\n", err.c_str());
return 1;
}
tr("prompt done (slots refilled)");
const double prompt_ms = std::chrono::duration<double, std::milli>(Clock::now() - r0).count();
std::printf("REUSED %lld\n", (long long) resume); // the prompt is read; the first window comes next
std::fflush(stdout);
// the verify windows: the first holds the last prompt token alone
int64_t p = n - 1;
int32_t x = (int32_t) ids[(size_t) (n - 1)];
std::vector<int32_t> drafts((size_t) S, 0), window((size_t) S), outv((size_t) S);
std::vector<float> dprob((size_t) S, 0.0f);
std::vector<int32_t> sbuf((size_t) S, 0);
if (o.suffix_draft > 0) {
sfx.reset();
for (int64_t t : ids) sfx.append((int32_t) t);
}
bool first_window = true;
int64_t produced_n = 0, sfx_windows = 0, sfx_drafts = 0, sfx_ok = 0;
int64_t draft_offered = 0, draft_accepted = 0;
// what the session holds once this request is done: the prompt read so far, then every committed token
std::vector<int32_t> consumed;
consumed.reserve((size_t) (n + max_new + S));
for (int64_t i = 0; i < n - 1; ++i) consumed.push_back((int32_t) ids[(size_t) i]);
const char* finish = "length";
const Clock::time_point d0 = Clock::now();
// STRATA_DECODE_TIMING=1: where a request's decode time goes (one line per request)
static const bool dec_timing = std::getenv("STRATA_DECODE_TIMING") != nullptr;
struct DecSnap {
double wait, pool, host, plan, actq, jobs, run;
int64_t misses, entries, hits, pcie;
};
auto dec_snap = [&]() {
return DecSnap{ver.ms_wait, ver.ms_pool, ver.ms_host, drive.d.ms_plan, drive.d.ms_actq, drive.d.ms_jobs,
drive.d.ms_run, drive.d.multi_misses, drive.d.multi_entries, drive.d.cache_hits,
drive.d.pcie_experts};
};
const DecSnap ds0 = dec_snap();
double dt_run = 0, dt_commit = 0, dt_draft = 0;
int64_t dec_windows = 0, dec_T = 0;
const int64_t decode_hits0 = drive.d.cache_hits;
// CS-T: the RAM and file tiers of this request (the mmap source; 0 with the arena)
const int64_t ram0 = src.ram_reads(), files0 = src.file_reads();
const uint64_t file_bytes0 = src.file_read_bytes();
const int64_t decode_look0 = drive.d.cache_hits + drive.d.cache_admitted + drive.d.cache_refused;
if (cancelled) finish = "cancel";
while (!cancelled && produced_n < max_new) {
int T = S_mtp;
if (req_spec_min_p > 0.0) {
T = 1;
while (T < S_mtp && dprob[(size_t) T - 1] >= (float) req_spec_min_p) ++T;
}
if (first_window) T = 1;
// a repeat of earlier context (prompt lookup) where the MTP's own first guess agrees: the policy takes it
// when its expected tokens per ms, from the measured acceptance and window costs, beat the MTP window's
bool from_sfx = false;
int sfx_match = 0;
if (o.suffix_draft > 0 && !first_window) {
const int k = sfx.propose(S - 1, sbuf.data());
sfx_match = sfx.last_match();
if (k > 0 && sbuf[0] == drafts[0]) {
const strata::spec::DraftPolicy::Pick pk = policy.choose(T, k, sfx_match);
if (pk.lookup) { T = pk.t; from_sfx = true; }
}
}
const bool timed_round = !first_window;
const Clock::time_point round0 = Clock::now();
if (p + T > o.max_context) break;
window[0] = x;
for (int i = 1; i < T; ++i) window[(size_t) i] = from_sfx ? sbuf[(size_t) i - 1] : drafts[(size_t) i - 1];
drive.d.layers = 0;
drive.d.experts = 0;
drive.d.failed = false;
// #463: the previous adapt round's copies land first - with a non-blocking query, whether a swapped-in
// expert ran on the GPU or the CPU (they round differently) depended on the copy's timing
// (STRATA_ADAPT_NOWAIT=1: 0.1.37's non-blocking query, the A/B)
apply_pending(!adapt_nowait());
if (hist_n > 0) {
// the tails the penalties count over, ONE PER ROW: the tokens the state has consumed, the
// fed-back head `x` (it joins `consumed` only after this window commits), then the drafts
// before that row - what plain decode would have counted there. (Until 0.1.19 only row 0
// was staged, and the drafted rows read unwritten slots.)
strata::kernels::penalty_rows(consumed.data(), (int64_t) consumed.size(), window.data(), T,
hist_n, hist_stage.data());
const strata::core::OnDevice on_h(hist_dev);
cudaMemcpy(d_hist, hist_stage.data(), (size_t) T * (size_t) hist_n * sizeof(int32_t),
cudaMemcpyHostToDevice);
}
tr("window", p, T);
const Clock::time_point tw0 = Clock::now();
if (!ver.run(T, window.data(), p, win_pool_fn, win_pool_user, outv.data(), err) || drive.d.failed) {
std::printf("ERR %s\n", drive.d.failed && drive.d.fail ? drive.d.fail : err.c_str());
return 1;
}
int a = 0;
while (a < T - 1 && window[(size_t) a + 1] == outv[(size_t) a]) ++a;
if (from_sfx) { ++sfx_windows; sfx_drafts += T - 1; sfx_ok += a; }
const Clock::time_point tw1 = Clock::now();
std::thread adapt_thr; // the adaptive tier beside the commit and the draft (as in generate)
bool adapt_ok = true;
if (!drive.d.usage.empty() && ((rounds + 1) % o.adapt_every) == 0)
adapt_thr = std::thread([&] { adapt_ok = adapt(); });
if (!ver.commit(a + 1, err)) {
if (adapt_thr.joinable()) adapt_thr.join();
std::printf("ERR %s\n", err.c_str());
return 1;
}
// the window's first a + 1 tokens are in the session now (the last output is not: it is next x)
for (int i = 0; i <= a; ++i) consumed.push_back(window[(size_t) i]);
draft_offered += T - 1;
draft_accepted += a;
first_window = false;
bool eos = false;
for (int i = 0; i <= a && produced_n < max_new && !eos; ++i) {
std::printf("T %d\n", (int) outv[(size_t) i]);
strata::core::progress_beat();
++produced_n;
if (o.suffix_draft > 0) sfx.append(outv[(size_t) i]);
eos = std::find(o.eos_ids.begin(), o.eos_ids.end(), (int64_t) outv[(size_t) i]) != o.eos_ids.end();
}
std::fflush(stdout);
++rounds;
const Clock::time_point tw2 = Clock::now();
// coupled drafts with penalties: the next window's row-0 history (`consumed` holds this window's
// commit, outv[a] is its row 0) - the drafts extend it on the device as the verify rows will
if (hist_n > 0 && mtp.coupled() && !eos && produced_n < max_new)
mtp.set_draft_history(consumed.data(), (int64_t) consumed.size(), outv[(size_t) a]);
const bool drafted = eos || produced_n >= max_new ||
mtp.draft(T, outv.data(), p, a, drafts.data(), err, dprob.data(), (float) req_spec_min_p);
{
const Clock::time_point tw3 = Clock::now();
auto msd = [](Clock::time_point a0, Clock::time_point b0) { return std::chrono::duration<double, std::milli>(b0 - a0).count(); };
dt_run += msd(tw0, tw1); dt_commit += msd(tw1, tw2); dt_draft += msd(tw2, tw3);
++dec_windows; dec_T += T;
}
if (adapt_thr.joinable()) adapt_thr.join();
if (!adapt_ok) {
std::printf("ERR an adaptive refill failed\n");
return 1;
}
if (!drafted) {
std::printf("ERR %s\n", err.c_str());
return 1;
}
if (timed_round && !eos)
policy.observe(from_sfx, T, a, sfx_match,
std::chrono::duration<double, std::milli>(Clock::now() - round0).count());
if (eos) { finish = "stop"; break; }
if (stop_req.load()) { finish = "cancel"; break; }
x = outv[(size_t) a];
p += a + 1;
}
const double decode_ms = std::chrono::duration<double, std::milli>(Clock::now() - d0).count();
// the last commit (set_commit_async): the session is complete before anything reads or copies it
if (!ver.wait_commit(err)) {
std::printf("ERR %s\n", err.c_str());
return 1;
}
if (dec_timing && dec_windows > 0) {
const DecSnap d1 = dec_snap();
const double w = (double) dec_windows, L = (double) g.n_layers;
std::fprintf(stderr, "strata decode timing: %lld windows, avg T %.2f, %.2f tokens/window, %.2f ms/window = "
"verify %.2f (GPU-reach wait %.2f + per-layer host %.2f [plan %.2f actq %.2f jobs %.2f "
"CPU %.2f] + stage %.2f) + commit/emit %.2f + draft %.2f; per layer-window: CPU experts "
"%.2f (%.2f entries), VRAM hits %.2f, PCIe %.2f\n",
(long long) dec_windows, dec_T / w, produced_n / w, decode_ms / w, dt_run / w,
(d1.wait - ds0.wait) / w, (d1.pool - ds0.pool) / w, (d1.plan - ds0.plan) / w,
(d1.actq - ds0.actq) / w, (d1.jobs - ds0.jobs) / w, (d1.run - ds0.run) / w,
(d1.host - ds0.host) / w, dt_commit / w, dt_draft / w, (d1.misses - ds0.misses) / (w * L),
(d1.entries - ds0.entries) / (w * L), (d1.hits - ds0.hits) / (w * L), (d1.pcie - ds0.pcie) / (w * L));
const std::string pr = ver.profile_report();
if (!pr.empty()) std::fprintf(stderr, "strata decode GPU stages (ms/window):%s\n", pr.c_str());
}
if (!cancelled) {
// a prompt stopped halfway leaves the session somewhere between two chunks: nothing to continue from
// (the checkpoints taken while reading it are still good)
live.swap(consumed);
live_imgs = imgs_below(req_imgs, (int64_t) live.size());
live_ok = o.prompt_cache > 0;
}
static const bool state_hash = std::getenv("STRATA_STATE_HASH") != nullptr;
if (state_hash && live_ok) {
// DEBUG: a fingerprint of every part of the session over the positions it holds ([0, L)), and
// separately of what lies past them in the last KV page (stale cells, fine unless something reads them)
if (cudaDeviceSynchronize() != cudaSuccess) {
std::printf("ERR synchronizing state fingerprint\n");
return 1;
}
const int64_t L = (int64_t) live.size();
const strata::kernels::QsaShapes qs = [&] {
strata::kernels::QsaShapes s = strata::kernels::qsa_real_shapes();
s.n_head_kv = g.n_head_kv; s.head_dim = g.head_dim; s.idx_dim = g.idx_key_dim;
return s;
}();
bool hash_ok = true;
std::array<uint8_t, 65536> hash_buffer;
auto hash_dev = [&](const void* p, size_t bytes, uint64_t h) {
for (size_t offset = 0; hash_ok && offset < bytes;) {
const size_t n = std::min(hash_buffer.size(), bytes - offset);
// VRAM or a streamed host copy, with fixed diagnostic workspace.
if (cudaMemcpy(hash_buffer.data(), static_cast<const uint8_t*>(p) + offset, n, cudaMemcpyDefault) != cudaSuccess) {
hash_ok = false;
break;
}
h = fnv1a(hash_buffer.data(), n, h);
offset += n;
}
return h;
};
// the cells [c0, c1) of one int8 K or V pool ([page][kv_head][page_size][head_dim]), bytes per value `w`
auto hash_cells = [&](const void* pool, int64_t per_cell, int64_t c0, int64_t c1, uint64_t h) {
const int64_t ps = qs.page_size;
for (int64_t pg = c0 / ps; pg * ps < c1; ++pg)
for (int64_t hd = 0; hd < qs.n_head_kv; ++hd) {
const int64_t a = std::max(c0, pg * ps) - pg * ps, e = std::min(c1, (pg + 1) * ps) - pg * ps;
const size_t off = (size_t) (((pg * qs.n_head_kv + hd) * ps + a) * per_cell);
h = hash_dev((const uint8_t*) pool + off, (size_t) ((e - a) * per_cell), h);
}
return h;
};
const ConvStateSizes z = conv_state_sizes(g, ss);
uint64_t h_gdn = hash_dev(ss.gdn_state, z.gdn, 1469598103934665603ull);
if (std::getenv("STRATA_STATE_HASH_GDN") != nullptr && ss.gdn_alloc > 0) { // per GDN layer: which one differs first
const size_t per = z.gdn / (size_t) ss.gdn_alloc;
std::string s;
char b[8];
for (int64_t i = 0; i < ss.gdn_alloc; ++i) {
std::snprintf(b, sizeof(b), "%04llx ", (unsigned long long) (hash_dev((const uint8_t*) ss.gdn_state + i * per, per, 1469598103934665603ull) & 0xffff));
s += b;
}
std::fprintf(stderr, "strata serve: STATE_HASH_GDN %s\n", s.c_str());
}
uint64_t h_ple = hash_dev(ss.ple_hist, ss.ple_hist ? z.ple : 0, 1469598103934665603ull);
uint64_t h_tail = 1469598103934665603ull, h_pool = h_tail, h_kv = h_tail, h_stale = h_tail;
// pooled= keeps its 0.1.29 meaning: the completed rows [0, L / idx_block) only. pooled_full= adds the
// spare row at L / idx_block (the `dead` key the next block completion overwrites), which a
// conversation restore writes back; dead= is the spare key itself
uint64_t h_dead = h_tail, h_pool_full = h_tail;
const int64_t kvb = qs.head_dim, scb = (qs.head_dim / 64) * 2;
// a state's K/V arrays and their bytes per (cell, head) row: the host copy when it has one
auto kv_arrays = [&](const strata::core::QsaState& st) {
const bool h = st.kv_mode != 0;
std::vector<std::pair<const void*, int64_t>> a;
if (st.kv_q4) {
const int64_t q4b = (int64_t) strata::kernels::kv_q4_bytes_per_head((int) qs.head_dim);
a = {{h ? st.host.k_q4 : st.k_q4, q4b}, {h ? st.host.v_q4 : st.v_q4, q4b}};
} else if (st.kv_hybrid) {
const int64_t q4b = (int64_t) strata::kernels::kv_q4_bytes_per_head((int) qs.head_dim);
a = {{h ? st.host.k_q : st.k_q, kvb}, {h ? st.host.v_q4 : st.v_q4, q4b},
{h ? st.host.k_scale : st.k_scale, scb}};
} else if (st.kv_int8) {
a = {{h ? st.host.k_q : st.k_q, kvb}, {h ? st.host.v_q : st.v_q, kvb},
{h ? st.host.k_scale : st.k_scale, scb}, {h ? st.host.v_scale : st.v_scale, scb}};
} else {
a = {{h ? st.host.k_pool : st.k_pool, qs.head_dim * 2},
{h ? st.host.v_pool : st.v_pool, qs.head_dim * 2}};
}
return a;
};
const int64_t end_cell = std::min<int64_t>(((L + qs.page_size - 1) / qs.page_size) * qs.page_size,
ss.max_cells); // = the primary state's max_cells
for (int64_t j = 0; j < ss.qsa_alloc; ++j) { // this session's owned QSA ordinals only
const strata::core::QsaState& st = ss.qsa_states[ss.qsa_ord0 + j];
h_tail = hash_dev(st.idx_tail, z.tail, h_tail);
h_dead = hash_dev(st.idx_dead, z.dead, h_dead);
h_pool = hash_dev(st.idx_pooled, (size_t) (L / qs.idx_block) * qs.idx_dim * 4, h_pool);
h_pool_full = hash_dev(st.idx_pooled, (size_t) (L > 0 ? L / qs.idx_block + 1 : 0) * qs.idx_dim * 4,
h_pool_full);
// KV streaming: the host copy is the identity layout and holds every cell
for (const auto& [pool, w] : kv_arrays(st)) {
h_kv = hash_cells(pool, w, 0, L, h_kv);
h_stale = hash_cells(pool, w, L, end_cell, h_stale);
}
}
const strata::core::QsaState& ms = mtp.kv_state();
uint64_t h_mtp = 1469598103934665603ull;
const int64_t mL = std::min<int64_t>(L, ms.max_cells);
for (const auto& [pool, w] : kv_arrays(ms))
if (pool != nullptr) h_mtp = hash_cells(pool, w, 0, mL, h_mtp);
if (!hash_ok) {
std::printf("ERR reading state fingerprint\n");
return 1;
}
std::fprintf(stderr, "strata serve: STATE_HASH L=%lld gdn=%016llx ple=%016llx tail=%016llx pooled=%016llx "
"kv=%016llx mtp=%016llx stale=%016llx dead=%016llx pooled_full=%016llx ple_prev=%d,%d\n", (long long) L,
(unsigned long long) h_gdn, (unsigned long long) h_ple, (unsigned long long) h_tail,
(unsigned long long) h_pool, (unsigned long long) h_kv, (unsigned long long) h_mtp,
(unsigned long long) h_stale, (unsigned long long) h_dead,
(unsigned long long) h_pool_full, ss.ple_prev[0], ss.ple_prev[1]);
}
const int64_t req_hits = drive.d.cache_hits - decode_hits0;
const int64_t req_look = (drive.d.cache_hits + drive.d.cache_admitted + drive.d.cache_refused) - decode_look0;
// #471: the prompt tokens this request read - all the fresh ones, or as far as the prompt pass got when a
// cancel stopped it part-way (a cancelled request used to be logged and counted as having read them all)
const int64_t fresh = n - resume;
const int64_t read_n = cancelled ? std::clamp<int64_t>(pp_reached - resume, 0, fresh) : fresh;
// DONE <generated> <prompt> <prompt ms> <decode ms> <finish> <drafts accepted> <drafts offered> <reused> [hits] [lookups]
// [RAM blobs] [file blobs] [file MB] (CS-T tiers; appended, so an older server reads the rest)
// [prompt tokens read] (#471: fewer than <prompt> - <reused> when a cancel stopped the read)
std::printf("DONE %lld %lld %.1f %.1f %s %lld %lld %lld %lld %lld %lld %lld %.1f %lld\n", (long long) produced_n,
(long long) n, prompt_ms, decode_ms, finish, (long long) draft_accepted, (long long) draft_offered,
(long long) resume, (long long) req_hits, (long long) req_look,
(long long) (src.ram_reads() - ram0), (long long) (src.file_reads() - files0),
(double) (src.file_read_bytes() - file_bytes0) / 1e6, (long long) read_n);
std::fflush(stdout);
if (drive.routing != nullptr) std::fflush(drive.routing); // the routing trace survives a crash and is watchable mid-session
// "12288 of 98179" when cancelled mid-read (#471), the rate from what was read
char read_txt[64];
if (cancelled)
std::snprintf(read_txt, sizeof(read_txt), "%lld of %lld", (long long) read_n, (long long) fresh);
else
std::snprintf(read_txt, sizeof(read_txt), "%lld", (long long) fresh);
std::fprintf(stderr, "strata serve: prompt %lld tokens = %lld reused + %s read in %.0f ms (%.1f tok/s), "
"%lld generated in %.0f ms (%.1f tok/s), drafts accepted %lld of %lld, %zu checkpoints%s\n",
(long long) n, (long long) resume, read_txt, prompt_ms,
prompt_ms > 0 ? 1000.0 * read_n / prompt_ms : 0.0, (long long) produced_n, decode_ms,
decode_ms > 0 ? 1000.0 * produced_n / decode_ms : 0.0, (long long) draft_accepted,
(long long) draft_offered, checks.size(), cancelled ? " (cancelled)" : "");
// the VRAM share of the experts the pool looked up while decoding; experts it sent over PCIe for the GPU
// to read (--pcie-frac) are in neither count
if (req_look > 0) {
std::fprintf(stderr, "strata serve: decode expert cache hit rate: %.1f%% (%lld hits / %lld lookups)\n",
100.0 * (double) req_hits / (double) req_look,
(long long) req_hits, (long long) req_look);
}
// the resident RAM mode, cumulative: experts read from experts.bin since the copy was made (what the plain
// mmap mode reads through the OS file cache, from the SSD when the RAM could not keep it)
if (src.complement_ready())
std::fprintf(stderr, "strata serve: resident RAM: %.2f GiB of experts in RAM, %lld exchanged with the "
"VRAM tier, %lld blob reads from the file\n",
(double) src.resident_bytes() / 1073741824.0, (long long) src.exchanges(),
(long long) src.file_reads());
// CS-T: the tiers, cumulative - GPU cache hits (the decode lookups above), RAM copy, files (SSD / OS cache)
if (srcp == &src)
std::fprintf(stderr, "strata serve: expert tiers: GPU %lld hits this request; since the start RAM %lld blobs, files %lld blobs "
"%.1f MB read%s\n", (long long) req_hits, (long long) src.ram_reads(),
(long long) src.file_reads(), (double) src.file_read_bytes() / 1e6,
src.gguf_mode() ? " (the GGUF in place)" : "");
// STRATA_SPLIT_TIMING: where each verify stage's host time went, cumulative per window since the start
// (waiting for its GPU to ring a layer, the CPU pool and plan per layer, staging the window)
if (static const bool st_timing = std::getenv("STRATA_SPLIT_TIMING") != nullptr; st_timing)
for (int st = 0; st < n_stages; ++st) {
const strata::core::Verifier& v = stage_ver(st);
const double w = v.windows > 0 ? (double) v.windows : 1.0;
std::fprintf(stderr, "strata serve: stage %d: %lld windows; per window: wait for the GPU %.3f ms, "
"pool + plan %.3f ms, host staging %.3f ms, commit %.3f ms\n", st,
(long long) v.windows, v.ms_wait / w, v.ms_pool / w, v.ms_host / w, v.ms_commit / w);
}
if (g.n_qsa_layers() > 0 && ss.qsa_states[ss.qsa_primary()].kv_mode == 1) {
// KV streaming, cumulative over the process: blocks the selections named vs blocks read from RAM
// (this device's owned ordinals; a split's other stages hold theirs)
uint64_t miss = 0, look = 0;
bool over = false;
for (int64_t j = 0; j < ss.qsa_alloc; ++j) {
const strata::kernels::KvStreamCounters c =
strata::kernels::kv_stream_counters(ss.qsa_states[ss.qsa_ord0 + j].map);
miss += c.misses; look += c.lookups; over = over || c.overflow;
}
std::fprintf(stderr, "strata serve: KV streaming: %.2f%% of %llu block reads hit VRAM, %.1f MiB read "
"from RAM%s\n", look ? 100.0 * (double) (look - miss) / (double) look : 100.0,
(unsigned long long) look, (double) miss * 4224.0 / 1048576.0,
over ? " - OVERFLOW (too few resident cells)" : "");
}
if (sfx_windows > 0)
std::fprintf(stderr, "strata serve: suffix drafts: %lld windows, %lld of %lld drafts accepted\n",
(long long) sfx_windows, (long long) sfx_ok, (long long) sfx_drafts);
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0)
std::fprintf(stderr, "strata serve: CUDA%d: %lld expert entries, %lld active layer launches, %.1f MiB returned "
"(%.1f MiB with full rows) in this request; host %.0f ms staging+launching, %.0f ms "
"waiting for it in this request\n", r + 1,
(long long) (remote_experts[(size_t) r].computed() - remote_before[(size_t) r]),
(long long) (remote_experts[(size_t) r].launched_layers() - launches_before[(size_t) r]),
(double) (remote_experts[(size_t) r].returned_bytes() - compact_before[(size_t) r]) / 1048576.0,
(double) (remote_experts[(size_t) r].full_row_bytes() - full_before[(size_t) r]) / 1048576.0,
remote_experts[(size_t) r].ms_begin() - begin_before[(size_t) r],
remote_experts[(size_t) r].ms_wait() - wait_before[(size_t) r]);
}
save_profile("exit"); // #477: QUIT, or the server closed stdin
return 0;
}
// ---- plan v0.3 P5: the prompt's conditioning positions [0, n_prompt - 1) in batched chunks. The token loop
// then starts at the last prompt position, whose prediction is the first generated token.
int64_t pos_start = 0;
int64_t spec_pos = 0; // plan v0.3 P6: where the speculative loop starts (0 = not used)
strata::prefill::Prefill prefill;
double prefill_batched_ms = 0;
std::FILE* final_r = o.dump_final_r.empty() ? nullptr : std::fopen(o.dump_final_r.c_str(), "wb");
std::vector<float> final_r_host(final_r ? (size_t) (g.hc * g.n_embd) : 0);
std::vector<std::pair<int32_t, int32_t>> lent; // (residency index, slot) lent to the prompt path
const int64_t n_batched = (o.prefill_until > 0 && o.prefill_until < n_prompt - 1) ? o.prefill_until : n_prompt - 1;
if (o.prefill_chunk > 0 && n_prompt > 1) {
void* borrow = nullptr;
uint64_t borrow_bytes = 0;
if (!o.no_prefill_borrow && !host_res.empty() && d_res != nullptr) {
int64_t chunk = o.prefill_chunk;
int64_t k = plan_lend(chunk); // auto: the largest chunk that fits; fixed: halved to fit
const int64_t request_sized = request_chunk(n_batched, chunk);
if (k > 0 && request_sized < chunk) { // no bigger than this prompt segment needs
chunk = request_sized;
k = lend_slots(chunk);
if (k + 128 > xcache.slots()) k = 0;
if (!o.prefill_auto) o.prefill_chunk = chunk;
}
if (o.prefill_auto) {
o.prefill_chunk = k > 0 ? chunk : request_chunk(n_batched, 1024);
std::fprintf(stderr, "strata generate: prompt chunk auto: %lld tokens\n", (long long) o.prefill_chunk);
} else if (chunk != o.prefill_chunk) {
k = 0; // a fixed chunk that does not fit: its own buffers, as before
}
const int64_t blob = (int64_t) strata::kernels::cpu::expert_layout().max_blob;
if (k > 0) { // the lent slots are refilled after the prompt
const int32_t first = (int32_t) (xcache.slots() - k);
for (size_t i = 0; i < host_res.size(); ++i)
if (host_res[i] >= first) {
lent.emplace_back((int32_t) i, host_res[i]);
host_res[i] = strata::core::kNotResident;
}
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
borrow = xcache.device_slot(first);
borrow_bytes = xcache.slot_offsets() ? (uint64_t) (xcache.bytes() - (int64_t) xcache.slot_offsets()[first])
: (uint64_t) k * (uint64_t) blob;
std::fprintf(stderr, "strata generate: prompt path borrows %lld cache slots (%.2f GiB)\n", (long long) k,
(double) borrow_bytes / 1073741824.0);
}
}
if (borrow == nullptr)
std::fprintf(stderr, "strata generate: prompt path allocates its own buffers (no cache slots to borrow)\n");
if (!prefill.init(wt, g, ss, srcp, o.expert_cache > 0 ? &xcache : nullptr,
host_res.empty() ? nullptr : host_res.data(), o.prefill_chunk, main_cs, err, borrow,
borrow_bytes)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
if (!o.mtp.empty()) {
if (!mtp.bind(wt, &native_head, nullptr, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
prefill.on_chunk = [&](const float* R_rows, int64_t T, int64_t p0, std::string& e) -> bool {
// cell i pairs R_i with the token at i + 1 (every such token is in the prompt)
std::vector<int32_t> nxt((size_t) T);
for (int64_t t = 0; t < T; ++t) nxt[(size_t) t] = (int32_t) o.tokens[(size_t) (p0 + t + 1)];
if (prefill.draft_kv(mtp, R_rows, nxt.data(), T, p0, e)) return true; // E-9
return e.empty() && mtp.prefill(R_rows, nxt.data(), T, p0, e);
};
}
const Clock::time_point tp0 = Clock::now();
if (!prefill.run(o.tokens.data(), n_batched, 0, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
// refill the lent slots from the arena and give them back to the decode tier
if (!lent.empty()) {
const Clock::time_point tr = Clock::now();
for (const auto& [i, slot] : lent) { // D-4: queued, one wait (STRATA_REFILL_BLOCKING=1: each)
const uint8_t* b = srcp->blob(i / g.n_expert, i % g.n_expert);
const int64_t nb = (int64_t) strata::kernels::cpu::expert_layout().blob_bytes(i / g.n_expert);
if (b == nullptr || !(refill_blocking() ? xcache.fill_slot_blocking(slot, b, err, nb)
: xcache.fill_slot_queued(slot, b, err, nb))) {
std::fprintf(stderr, "strata generate: refilling a lent slot failed: %s\n", err.c_str());
return 1;
}
host_res[(size_t) i] = slot;
}
if (!xcache.sync_queued(err)) {
std::fprintf(stderr, "strata generate: refilling the lent slots failed: %s\n", err.c_str());
return 1;
}
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
std::fprintf(stderr, "strata generate: %zu lent slots refilled in %.1f ms\n", lent.size(),
std::chrono::duration<double, std::milli>(Clock::now() - tr).count());
}
prefill_batched_ms = std::chrono::duration<double, std::milli>(Clock::now() - tp0).count();
prefill_ms += prefill_batched_ms;
pos_start = n_batched;
tok = o.tokens[(size_t) pos_start];
// the PLE window of the token path: the two tokens before `pos_start`
ss.ple_prev[0] = pos_start >= 2 ? (int32_t) o.tokens[(size_t) (pos_start - 2)] : -1;
ss.ple_prev[1] = pos_start >= 1 ? (int32_t) o.tokens[(size_t) (pos_start - 1)] : -1;
const strata::prefill::PrefillStats& ps = prefill.stats();
std::fprintf(stderr, "strata generate: prefill %lld tokens in %lld chunks, %.1f ms (%.1f tok/s); experts "
"streamed %lld (%lld by DMA, host %.1f ms), resident %lld; PLE %.1f ms\n",
(long long) ps.tokens, (long long) ps.chunks, ps.ms_total,
ps.ms_total > 0 ? 1000.0 * (double) ps.tokens / ps.ms_total : 0.0, (long long) ps.experts_streamed,
(long long) ps.experts_dma, ps.ms_experts_host, (long long) ps.experts_resident, ps.ms_ple);
}
for (int64_t pos = pos_start;; ++pos) {
// plan v0.3 P6: a native pack's last prompt token is the first verify window (T = 1)
if (native_pack) { spec_pos = pos; break; }
if (pos >= o.max_context) {
std::fprintf(stderr, "strata generate: ran out of context at position %lld\n", (long long) pos);
return 2;
}
// **THE TOKEN TIMER STARTS HERE, BEFORE ANY OF THE TOKEN'S WORK (A7).** It used to start after
// `put_input`/`embed_row`, which excluded the embedding and the PLE window advance from the reported
// rate, and it stopped before the NaN scan, the logits dump and the sampler. The published tok/s
// figure is a WALL-CLOCK rate: everything one token costs, PLE advance to sampled id. A rate that
// excludes real per-token work is not a rate anyone can plan against.
const Clock::time_point t0 = Clock::now();
// **THE PLE'S TOKEN WINDOW ADVANCES HERE, ONCE PER TOKEN, AND `ple_stage_token` RUNS OUTSIDE THE
// CAPTURE.** Both are the driver's job: `ngram_rows` is a host hash over the last three tokens and the
// table gather is a host read, so either one inside a captured graph would run once at capture time and
// replay forever. `ple_prev` is OLDEST FIRST and `-1` means "no predecessor", which `ngram_rows`
// treats as the EOS cut - a sequence boundary.
ss.ple_token = (int32_t) tok;
Clock::time_point tp = Clock::now();
// Plan v0.3 P2: the 16 SSD reads start here and complete while the embedding is staged; `ms_ple` is
// the issue plus the time still spent WAITING afterwards, i.e. the part the embedding did not hide.
if (ss.ple.ready() && !strata::core::ple_issue_token(ss.ple, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
{
const Clock::time_point n = Clock::now();
ms_ple += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
if (!put_input(tok, pos)) return 1;
{
const Clock::time_point n = Clock::now();
ms_embed += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
if (ss.ple.ready() && !strata::core::ple_finish_token(ss.ple, token_stream, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
{
const Clock::time_point n = Clock::now();
ms_ple += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
if (pos % 256 == 0 || pos + 1 >= n_prompt - 1)
std::fprintf(stderr, "strata generate: position %lld, token %lld%s\n", (long long) pos, (long long) tok,
pos < n_prompt ? " (prompt)" : "");
// **`d.layers` IS THE BLOB'S LAYER AXIS, NOT A COUNTER.** The adapter uses it to index
// `experts.bin` as `layer * n_expert + expert`, so it MUST restart at 0 for every token. Leaving it
// running across tokens asks for layer 48 of a 48-layer file on the second token - which
// `FileExpertSource` REFUSES rather than wrapping into layer 0's experts, and that refusal is the only
// reason this was a clean error instead of a silently wrong second token.
drive.d.layers = 0;
drive.d.experts = 0;
drive.d.failed = false;
err.clear();
if (o.no_capture) {
if (!strata::core::session_token(wt, g, pos, /*pos_base=*/0, ss, d_parts, main_cs,
o.sync_every_layer, err)) {
std::fprintf(stderr, "strata generate: session_token: %s\n", err.c_str());
return 1;
}
} else {
strata::core::doorbell_reset(db);
if (tgraph.captured) {
if (!strata::core::session_run_token(g, pos, /*pos_base=*/0, ss, tgraph, pool_fn, pool_user,
loop_scratch.y_miss, main_cs, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
} else if (!strata::core::session_loop(g, pos, /*pos_base=*/0, ss, gr, pool_fn, hit_fn, pool_user, /*overlap=*/true, main_cs,
err, layer_stage, &loop_scratch)) {
std::fprintf(stderr, "strata generate: session_loop: %s\n", err.c_str());
return 1;
}
}
if (final_r != nullptr) {
cudaMemcpy(final_r_host.data(), ss.R, final_r_host.size() * sizeof(float), cudaMemcpyDeviceToHost);
const int64_t posrec[2] = {pos, tok};
std::fwrite(posrec, sizeof posrec, 1, final_r);
std::fwrite(final_r_host.data(), sizeof(float), final_r_host.size(), final_r);
}
if (drive.d.failed) {
std::fprintf(stderr, "strata generate: the expert pool failed at layer %lld expert %lld: %s\n",
(long long) drive.d.fail_layer, (long long) drive.d.fail_expert,
drive.d.fail ? drive.d.fail : "(no message)");
return 1;
}
{
// **THE LAYER LOOP ITSELF, WHICH IS WHAT `--gpu-only-full` HAS TO BE COMPARED AGAINST.**
const Clock::time_point n = Clock::now();
ms_layers += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
// ---- the ladder for THIS position, in the order `session_loop` filled it: layer 0 first.
if (layer_dump != nullptr) std::fwrite(layer_stage, sizeof(float), layer_floats, layer_dump);
if (half_dump != nullptr) {
std::fwrite(half_stage, sizeof(float), (size_t) g.n_layers * (size_t) half_stride, half_dump);
}
if (!run_head(token_stream)) {
std::fprintf(stderr, "strata generate: lm_head: %s\n", err.c_str());
return 1;
}
// **A CHECKPOINT AFTER THE HEAD, BECAUSE AN ASYNC FAULT IS STICKY AND LIES ABOUT WHERE IT HAPPENED.**
// Measured, and it cost an hour: without this, `embed_row`'s D2H on the NEXT token reported "an illegal
// memory access" at a plane offset that has nothing to do with the fault, and the layer that actually
// faulted had completed its own error checks successfully - because its kernels had not run yet. A
// sticky error surfaces at the next SYNCHRONISING call, which is whatever happens to come next.
if (!o.stream_token && cudaDeviceSynchronize() != cudaSuccess) {
std::fprintf(stderr, "strata generate: the device faulted in lm_head at position %lld: %s\n",
(long long) pos, cudaGetErrorString(cudaGetLastError()));
return 1;
}
{
const Clock::time_point n = Clock::now();
ms_head += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
const bool emit_logits = dump != nullptr &&
strata::program::logits_selection::selected(pos, dump_positions, o.logits_stride);
const bool read_logits = !o.stream_token || o.check_logits || emit_logits;
if (read_logits && (cudaMemcpyAsync(logits.data(), d_logits, (size_t) n_vocab * 4,
cudaMemcpyDeviceToHost, (cudaStream_t) token_stream) != cudaSuccess ||
cudaStreamSynchronize((cudaStream_t) token_stream) != cudaSuccess)) {
std::fprintf(stderr, "strata generate: reading the logits back failed\n");
return 1;
}
int bad = 0;
if (read_logits) for (float v : logits) if (!std::isfinite(v)) ++bad;
if (bad != 0) {
std::fprintf(stderr, "strata generate: %d of %lld logits are not finite at position %lld\n", bad,
(long long) n_vocab, (long long) pos);
return 1;
}
// the header with the first row (see `hdr`): a run that dumps nothing leaves an empty file
if (emit_logits && !hdr_written && std::fwrite(hdr, sizeof hdr, 1, dump) != 1) {
std::fprintf(stderr, "strata generate: cannot write logits header\n");
std::fclose(dump);
return 1;
}
if (emit_logits) hdr_written = true;
if (emit_logits && std::fwrite(logits.data(), sizeof(float), (size_t) n_vocab, dump) != (size_t) n_vocab) {
std::fprintf(stderr, "strata generate: cannot write logits at position %lld\n", (long long) pos);
std::fclose(dump);
return 1;
}
{
// **993 KB OF SYNCHRONOUS D2H AND A 248,320-FLOAT HOST SCAN, EVERY TOKEN.** (The review's notes
// say 151,936 floats; the artifact's `output.weight` is 248,320 rows, so the real figure is 1.6x
// that - a number nobody had checked because nothing measured this term.) R2.6 asks for the dump
// and the scan to be behind flags; round 36 did exactly that and measured it SLOWER, because on
// this driver a large blocking readback is also what flushes the pipeline for the sampler that
// follows. Timed so the claim can be re-checked rather than remembered.
const Clock::time_point n = Clock::now();
ms_readback += std::chrono::duration<double, std::milli>(n - tp).count();
tp = n;
}
int next = 0;
// The draw is Philox(seed, position), as in a verify window (row t at pos0 draws pos0 + t): a seed gives
// the same text whether a token comes from this path or from the speculative loop below.
sp.counter = (uint64_t) pos;
strata::kernels::sample_tokens(d_logits, 1, (int) n_vocab, nullptr, 0, sp, d_next, token_stream);
if (cudaMemcpyAsync(&next, d_next, sizeof(int), cudaMemcpyDeviceToHost,
(cudaStream_t) token_stream) != cudaSuccess ||
cudaStreamSynchronize((cudaStream_t) token_stream) != cudaSuccess) {
std::fprintf(stderr, "strata generate: reading the sampled token back failed: %s\n",
cudaGetErrorString(cudaGetLastError()));
return 1;
}
// The sampled-token synchronization also completes every captured QSA
// status readback. Retain one status per layer so a later layer cannot
// hide an earlier failure; no extra synchronization or token allocation.
if (o.native_flash_attn_short) for (int64_t j = 0; j < ss.qsa_alloc; ++j) {
const int64_t i = ss.qsa_ord0 + j; // the global ordinal, as the session's carve names it
const int32_t status = ss.qsa_states[i].host_step[strata::kernels::kStepCount];
if (status != 0) {
std::fprintf(stderr, "strata generate: native attention status %d at QSA layer %lld, position %lld\n",
status, (long long) i, (long long) pos);
return 1;
}
}
if (next < 0 || next >= n_vocab) {
std::fprintf(stderr, "strata generate: the sampler returned %d, outside 0..%lld\n", next,
(long long) (n_vocab - 1));
return 1;
}
{
// **TWO DEVICE-WIDE SYNCS FOR FOUR BYTES.** `sample_tokens(nullptr)` ends in
// `cudaDeviceSynchronize()` (`sampler.cu:245`) and the blocking 4-byte read below is the second.
const Clock::time_point n = Clock::now();
ms_sample += std::chrono::duration<double, std::milli>(n - tp).count();
++phase_tokens;
}
// CHARGED HERE, AFTER THE SAMPLER, so the wall-clock rate covers the whole token including the embedding,
// the NaN scan, the logits readback and the sample (A7). Only DECODE positions count; prefill is
// measured separately.
if (pos >= n_prompt - 1) total_ms += std::chrono::duration<double, std::milli>(Clock::now() - t0).count();
else prefill_ms += std::chrono::duration<double, std::milli>(Clock::now() - t0).count();
if (pos == n_prompt - 1) ttft_ms = std::chrono::duration<double, std::milli>(Clock::now() - t_start).count();
// **THE PREDICTION AT THE LAST PROMPT POSITION *IS* THE FIRST GENERATED TOKEN.** Sampling on every
// position and recording only from `n_prompt - 1` onward is what keeps the two cases from needing
// separate handling - and the version that "obviously" only samples after the prompt loses exactly one
// token's worth of conditioning.
if (pos >= n_prompt - 1) produced.push_back(next);
if ((int64_t) produced.size() >= o.max_new) break;
if (o.stop_eos && pos >= n_prompt - 1 &&
std::find(o.eos_ids.begin(), o.eos_ids.end(), (int64_t) next) != o.eos_ids.end()) break;
// TEACHER FORCING while the prompt lasts: the next input is the prompt's own next token, not the
// model's guess. Feeding the guess would make the run depend on the model's own errors from position
// 1, which is a different (and worse) measurement of the same prompt.
// and the window advances: the token just decoded becomes the newest predecessor.
ss.ple_prev[0] = ss.ple_prev[1];
ss.ple_prev[1] = (int32_t) tok;
tok = (pos + 1 < n_prompt) ? o.tokens[(size_t) (pos + 1)] : next;
// Plan v0.3 P6: from the first generated token on, the speculative loop below takes over.
if (o.spec > 0 && pos >= n_prompt - 1) { spec_pos = pos + 1; break; }
}
// ================================ plan v0.3 P6: SPECULATIVE DECODING ================================
//
// Each round verifies [the last emitted token, drafts...] in one window; the window's argmax after token t
// is exactly what greedy decode would emit there, so the first draft that differs ends the round and the
// round emits (accepted drafts + 1) tokens. `commit` keeps the state of the tokens that were emitted.
const bool ended = o.stop_eos && !produced.empty() &&
std::find(o.eos_ids.begin(), o.eos_ids.end(), (int64_t) produced.back()) != o.eos_ids.end();
if (spec_pos > 0 && (int64_t) produced.size() < o.max_new && !ended) {
std::vector<int64_t> oracle;
if (!o.spec_oracle.empty()) {
std::ifstream in(o.spec_oracle);
std::string text((std::istreambuf_iterator<char>(in)), std::istreambuf_iterator<char>());
std::string e;
if (!in || !parse_i64_list(text.c_str(), oracle, e)) {
std::fprintf(stderr, "strata generate: cannot read --spec-oracle %s\n", o.spec_oracle.c_str());
return 2;
}
}
if (thits.d_res == nullptr) {
std::fprintf(stderr, "strata generate: --spec needs the device residency table (--expert-profile, "
"--expert-cache and the token graph)\n");
return 2;
}
mem_mark("the head and the prompt path");
strata::core::Verifier ver;
strata::core::VerifyHits vh;
vh.d_res = thits.d_res;
vh.cache_base = thits.cache_base;
vh.blob = thits.blob;
vh.slot_off = xcache.slot_offsets(); // E-6: the device plan's pointers
vh.n_slots = xcache.slots();
if (!ver.init(wt, g, ss, vh, native_head.loaded() ? &native_head : nullptr, o.spec, err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
const bool use_mtp = !o.mtp.empty();
if (use_mtp && !mtp.bind(wt, &native_head, ver.final_R_all(), err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
mem_mark("the verifier and the drafter's binding");
ver.set_sampling(sp); // the CLI's own sampling (until 0.1.19 this loop was always greedy); no penalties here
if (use_mtp) mtp.set_draft_sampling(sp); // STRATA_SPEC_COUPLED=1: sampled drafts (a no-op otherwise)
ver.set_split(o.spec_split);
// auto: the copy kernel for every pack. DMA (the native packs' default until 0.1.13) has the host call
// cudaMemcpyAsync + cudaLaunchHostFunc inside a verify window while the GPU spins on the flag they raise;
// issue #31's thread dumps show the host stuck in that cudaMemcpyAsync on a driver lock for good. The copy
// kernel needs no host CUDA call there, and costs ~1-3% decode on IQ3_S (45.3 -> 44.8 tok/s, 8 requests).
ver.set_pcie_mode(o.pcie_mode == "dma" ? 0 : o.pcie_mode == "direct" ? 1 : 2);
drive.d.plan = ver.plan_sink();
drive.d.pcie_num = (int) (o.pcie_frac * 256.0 + 0.5);
if (drive.d.pcie_num < 0) drive.d.pcie_num = 0;
if (drive.d.pcie_num > 256) drive.d.pcie_num = 256;
const int64_t pcie0 = drive.d.pcie_experts;
// CS-T: the file tier since the decode began (the prompt path's copies are before this)
const int64_t files0 = src.file_reads(), ram0 = src.ram_reads();
const uint64_t fbytes0 = src.file_blob_bytes(), fall0 = src.file_read_bytes();
const double fms0 = src.file_ms();
if (o.adapt_every > 0 && o.adapt_swaps > 0) drive.d.usage.assign((size_t) (g.n_layers * g.n_expert), 0.0f);
int64_t swaps_total = 0;
double ms_adapt = 0;
cudaStream_t adapt_stream = nullptr;
if (!drive.d.usage.empty() && cudaStreamCreateWithFlags(&adapt_stream, cudaStreamNonBlocking) != cudaSuccess) {
std::fprintf(stderr, "strata generate: cannot create the refill stream\n");
return 1;
}
// plan v0.3 P6: swaps in flight - (residency index, slot) admitted when adapt_ev has completed
std::vector<std::pair<int32_t, int32_t>> pending;
cudaEvent_t adapt_ev = nullptr;
cudaEventCreateWithFlags(&adapt_ev, cudaEventDisableTiming);
int64_t adapt_rounds = 0; // counted here: `rounds` is declared below the adapt lambda
auto apply_pending = [&](bool wait) {
if (pending.empty()) return;
// STRATA_TRACE_ADAPT=1: whether a round's copies had landed when the next window read the table
static const bool trace_pending = std::getenv("STRATA_TRACE_ADAPT") != nullptr;
if (wait) cudaEventSynchronize(adapt_ev);
else if (cudaEventQuery(adapt_ev) != cudaSuccess) {
if (trace_pending)
std::fprintf(stderr, "strata: PENDING not landed, %zu stay non-resident this window\n",
pending.size());
return;
}
if (trace_pending)
std::fprintf(stderr, "strata: PENDING landed, %zu experts become resident\n", pending.size());
src.commit_exchanges(); // the resident RAM mode: the evicted experts take their places in RAM
for (const auto& [i, slot] : pending) host_res[(size_t) i] = slot;
pending.clear();
if (d_res != nullptr)
cudaMemcpy(d_res, host_res.data(), host_res.size() * sizeof(int32_t), cudaMemcpyHostToDevice);
};
// Plan v0.3 P6: the VRAM tier follows the conversation. Candidates are missing experts routed at least
// twice (decayed); each is paired with its layer's least-routed resident expert and swapped when it was
// routed clearly more often. Copies run between rounds, when the GPU is idle.
auto adapt = [&]() -> bool {
const Clock::time_point ta = Clock::now();
++adapt_rounds;
// STRATA_TRACE_ADAPT: why an adapt round did or did not swap. Default off, one getenv, and it
// reports the only thing that can make an adapt round a coin flip: whether the PREVIOUS round's
// asynchronous expert copies had landed by the time this round started.
static const bool trace_adapt = std::getenv("STRATA_TRACE_ADAPT") != nullptr;
if (!pending.empty()) {
if (trace_adapt)
std::fprintf(stderr, "strata: ADAPT round=%lld SKIPPED, %zu swaps still in flight\n",
(long long) adapt_rounds, pending.size());
return true; // the previous swaps are still in flight
}
if (trace_adapt) std::fprintf(stderr, "strata: ADAPT round=%lld considering\n", (long long) adapt_rounds);
struct Swap { float gain; int32_t layer, in, out; };
std::vector<Swap> swaps;
std::vector<std::pair<float, int32_t>> cand, vict;
for (int64_t l = 0; l < g.n_layers; ++l) {
cand.clear();
vict.clear();
const float* u = drive.d.usage.data() + l * g.n_expert;
const int32_t* r = host_res.data() + l * g.n_expert;
for (int32_t e = 0; e < (int32_t) g.n_expert; ++e) {
if (r[e] < 0) { if (u[e] >= 2.0f) cand.emplace_back(u[e], e); }
else vict.emplace_back(u[e], e);
}
if (cand.empty() || vict.empty()) continue;
std::sort(cand.begin(), cand.end(), [](auto& a, auto& b) { return a.first > b.first; });
const size_t nc = std::min(cand.size(), vict.size());
std::partial_sort(vict.begin(), vict.begin() + (ptrdiff_t) nc, vict.end(),
[](auto& a, auto& b) { return a.first < b.first; });
for (size_t i = 0; i < nc; ++i) {
if (cand[i].first < vict[i].first + 1.5f) break;
swaps.push_back({cand[i].first - vict[i].first, (int32_t) l, cand[i].second, vict[i].second});
}
}
std::sort(swaps.begin(), swaps.end(), [](const Swap& a, const Swap& b) { return a.gain > b.gain; });
if ((int) swaps.size() > o.adapt_swaps) swaps.resize((size_t) o.adapt_swaps);
if (!resident_stage_swaps(src, xcache, host_res, g.n_expert, swaps, adapt_stream)) {
std::fprintf(stderr, "strata generate: an adaptive refill failed (copying evicted experts back)\n");
return false;
}
for (const Swap& s : swaps) {
const size_t in = (size_t) s.layer * g.n_expert + s.in, out = (size_t) s.layer * g.n_expert + s.out;
const int32_t slot = host_res[out];
const uint8_t* b = srcp->blob(s.layer, s.in);
// asynchronous: the copies run while the MTP drafts; the next window waits for them
if (slot < 0 || b == nullptr ||
cudaMemcpyAsync(xcache.device_slot(slot), b, (size_t) strata::kernels::cpu::expert_layout().blob_bytes(s.layer),
cudaMemcpyHostToDevice, adapt_stream) != cudaSuccess) {
std::fprintf(stderr, "strata generate: an adaptive refill failed\n");
return false;
}
host_res[out] = strata::core::kNotResident; // evicted now: the CPU computes it meanwhile
pending.emplace_back((int32_t) in, slot); // resident once the copy has landed
}
if (!swaps.empty()) cudaEventRecord(adapt_ev, adapt_stream);
if (trace_adapt)
std::fprintf(stderr, "strata: ADAPT round=%lld swapped %zu of %d slots, usage decayed\n",
(long long) adapt_rounds, swaps.size(), o.adapt_swaps);
for (float& v : drive.d.usage) v *= o.adapt_decay;
swaps_total += (int64_t) swaps.size();
ms_adapt += std::chrono::duration<double, std::milli>(Clock::now() - ta).count();
return true;
};
int64_t p = spec_pos;
int32_t x = (int32_t) tok;
std::vector<int32_t> drafts((size_t) o.spec, 0);
std::vector<float> dprob((size_t) o.spec, 1.0f);
std::vector<int64_t> window_hist((size_t) o.spec + 1, 0);
// plan v0.3 P6: with a native pack the first window is the last prompt token alone (it produces the first
// generated token and the MTP's first cell); otherwise the token loop already did that.
bool first_window = native_pack;
if (use_mtp && !first_window &&
!mtp.draft_first(o.spec, ss.R, x, p - 1, drafts.data(), err, dprob.data(), (float) o.spec_min_p)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::vector<int32_t> window((size_t) o.spec), outv((size_t) o.spec);
std::vector<int64_t> accepted_hist((size_t) o.spec, 0);
int64_t rounds = 0, drafts_total = 0, drafts_ok = 0, corrupt_counter = 0;
const int S_mtp = o.mtp_max_t > 0 ? std::min(o.mtp_max_t, o.spec) : o.spec;
if (use_mtp && S_mtp < o.spec) mtp.set_max_drafts(S_mtp - 1);
strata::spec::SuffixDrafter sfx(std::max(1, o.suffix_draft), 64, (size_t) o.max_context + 4096);
strata::spec::DraftPolicy policy(o.spec); // MTP or lookup window (see draft_policy.hpp)
std::vector<int32_t> sbuf((size_t) o.spec, 0);
int64_t sfx_windows = 0, sfx_drafts = 0, sfx_ok = 0;
if (o.suffix_draft > 0) {
for (int64_t t : o.tokens) sfx.append((int32_t) t);
for (int64_t t : produced) sfx.append((int32_t) t);
}
const double pool_ms0 = drive.cpu_ms;
const int64_t misses0 = drive.d.multi_misses, entries0 = drive.d.multi_entries;
while ((int64_t) produced.size() < o.max_new) {
const Clock::time_point t0 = Clock::now();
int T = S_mtp;
if (use_mtp && o.spec_min_p > 0.0) {
T = 1;
while (T < S_mtp && dprob[(size_t) T - 1] >= (float) o.spec_min_p) ++T;
}
if (first_window) T = 1;
bool from_sfx = false;
int sfx_match = 0;
if (o.suffix_draft > 0 && !first_window) {
const int k = sfx.propose(o.spec - 1, sbuf.data());
sfx_match = sfx.last_match();
if (k > 0 && (!use_mtp || sbuf[0] == drafts[0])) {
const strata::spec::DraftPolicy::Pick pk = policy.choose(T, k, sfx_match);
if (pk.lookup) { T = pk.t; from_sfx = true; }
}
}
const bool timed_round = !first_window;
++window_hist[(size_t) T];
if (p + T > o.max_context) {
std::fprintf(stderr, "strata generate: ran out of context at position %lld\n", (long long) p);
return 2;
}
window[0] = x;
for (int i = 1; i < T; ++i) {
const size_t at = produced.size() - 1 + (size_t) i;
int32_t d = from_sfx ? sbuf[(size_t) i - 1] : use_mtp ? drafts[(size_t) i - 1]
: at < oracle.size() ? (int32_t) oracle[at] : 0;
if (o.spec_corrupt > 0 && (++corrupt_counter % o.spec_corrupt) == 0) d = (d + 1) % (int32_t) n_vocab;
window[(size_t) i] = d;
}
drive.d.layers = 0;
drive.d.experts = 0;
drive.d.failed = false;
// #463: the previous adapt round's copies land first - with a non-blocking query, whether a swapped-in
// expert ran on the GPU or the CPU (they round differently) depended on the copy's timing
// (STRATA_ADAPT_NOWAIT=1: 0.1.37's non-blocking query, the A/B)
apply_pending(!adapt_nowait());
if (!ver.run(T, window.data(), p, &drive_pool_multi, &drive, outv.data(), err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
if (drive.d.failed) {
std::fprintf(stderr, "strata generate: the expert pool failed at layer %lld expert %lld: %s\n",
(long long) drive.d.fail_layer, (long long) drive.d.fail_expert,
drive.d.fail ? drive.d.fail : "(no message)");
return 1;
}
int a = 0;
while (a < T - 1 && window[(size_t) a + 1] == outv[(size_t) a]) ++a;
if (first_window) {
first_window = false;
ttft_ms = std::chrono::duration<double, std::milli>(Clock::now() - t_start).count();
// Diagnostics: the first window runs the prompt's last token over the state the prompt path left
// (keys, values, recurrent state), so its logits carry whatever that path did. A native pack never
// runs the per-token loop --dump-logits reads; this is where prompt paths can be compared by output.
if (const char* fl = std::getenv("STRATA_DUMP_FIRST_LOGITS")) {
std::vector<float> row((size_t) ver.vocab());
std::FILE* f = ver.copy_logits(0, row.data()) ? std::fopen(fl, "wb") : nullptr;
if (f == nullptr || std::fwrite(row.data(), sizeof(float), row.size(), f) != row.size())
std::fprintf(stderr, "strata generate: STRATA_DUMP_FIRST_LOGITS: cannot write %s\n", fl);
if (f) std::fclose(f);
}
}
// plan v0.3 P6: the adaptive tier's host work (ranking, copy submission) runs on its own thread while the
// GPU commits and drafts; it touches only the residency tables, which nothing reads until the next window
std::thread adapt_thr;
bool adapt_ok = true;
if (!drive.d.usage.empty() && ((rounds + 1) % o.adapt_every) == 0)
adapt_thr = std::thread([&] { adapt_ok = adapt(); });
if (!ver.commit(a + 1, err)) {
if (adapt_thr.joinable()) adapt_thr.join();
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
++rounds;
drafts_total += T - 1;
drafts_ok += a;
++accepted_hist[(size_t) a];
if (from_sfx) { ++sfx_windows; sfx_drafts += T - 1; sfx_ok += a; }
bool eos = false;
for (int i = 0; i <= a && (int64_t) produced.size() < o.max_new && !eos; ++i) {
produced.push_back(outv[(size_t) i]);
if (o.suffix_draft > 0) sfx.append(outv[(size_t) i]);
eos = o.stop_eos && std::find(o.eos_ids.begin(), o.eos_ids.end(), (int64_t) outv[(size_t) i]) != o.eos_ids.end();
}
if (eos) {
if (adapt_thr.joinable()) adapt_thr.join();
total_ms += std::chrono::duration<double, std::milli>(Clock::now() - t0).count();
break;
}
const bool drafted = !use_mtp || (int64_t) produced.size() >= o.max_new ||
mtp.draft(T, outv.data(), p, a, drafts.data(), err, dprob.data(), (float) o.spec_min_p);
if (adapt_thr.joinable()) adapt_thr.join();
if (!adapt_ok) return 1;
if (!drafted) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
x = outv[(size_t) a];
p += a + 1;
const double round_ms = std::chrono::duration<double, std::milli>(Clock::now() - t0).count();
total_ms += round_ms;
if (timed_round) policy.observe(from_sfx, T, a, sfx_match, round_ms);
if (rounds % 64 == 0)
std::fprintf(stderr, "strata generate: position %lld, %lld tokens, %lld rounds\n", (long long) p,
(long long) produced.size(), (long long) rounds);
}
// the last commit (set_commit_async) before anything reads the session again
if (!ver.wait_commit(err)) {
std::fprintf(stderr, "strata generate: %s\n", err.c_str());
return 1;
}
std::printf("%-24s %lld rounds of %d, drafts accepted %lld of %lld (%.3f), %.2f tokens per round\n",
"speculation", (long long) rounds, o.spec, (long long) drafts_ok, (long long) drafts_total,
drafts_total > 0 ? (double) drafts_ok / (double) drafts_total : 0.0,
rounds > 0 ? (double) (drafts_ok + rounds) / (double) rounds : 0.0);
if (o.spec_min_p > 0.0) {
std::printf("%-24s", "window sizes");
for (size_t i = 1; i < window_hist.size(); ++i) std::printf(" T%zu:%lld", i, (long long) window_hist[i]);
std::printf(" (min draft probability %.2f)\n", o.spec_min_p);
}
if (o.suffix_draft > 0)
std::printf("%-24s %lld windows, drafts accepted %lld of %lld\n", "suffix drafts", (long long) sfx_windows,
(long long) sfx_ok, (long long) sfx_drafts);
std::printf("%-24s", "accepted per round");
for (size_t i = 0; i < accepted_hist.size(); ++i) std::printf(" %zu:%lld", i, (long long) accepted_hist[i]);
std::printf("\n");
if (rounds > 0)
std::printf("%-24s wait for rings %.3f pool %.3f host %.3f commit %.3f ms/round; CPU experts %.2f "
"distinct / %.2f routed per layer\n",
"verify window", ver.ms_wait / rounds, ver.ms_pool / rounds, ver.ms_host / rounds,
ver.ms_commit / rounds,
(double) (drive.d.multi_misses - misses0) / (double) (rounds * g.n_layers),
(double) (drive.d.multi_entries - entries0) / (double) (rounds * g.n_layers));
if (rounds > 0)
std::printf("%-24s gate/up %.3f quantize %.3f down %.3f ms/round; %.1f GB/s over the rows phases; "
"CPU pool call %.3f ms/round\n", "pool multi", pool.ms_multi_gu / rounds,
pool.ms_multi_q / rounds, pool.ms_multi_down / rounds,
(double) pool.multi_bytes / 1e6 / std::max(1e-9, pool.ms_multi_gu + pool.ms_multi_down),
(drive.cpu_ms - pool_ms0) / rounds);
if (rounds > 0)
std::printf("%-24s plan %.3f activation quantize %.3f jobs %.3f run %.3f ms/round\n", "dispatch",
drive.d.ms_plan / rounds, drive.d.ms_actq / rounds, drive.d.ms_jobs / rounds,
drive.d.ms_run / rounds);
if (rounds > 0 && !drive.d.usage.empty())
std::printf("%-24s %lld experts swapped into the VRAM tier (every %d rounds, %.3f ms/round)\n", "adaptive tier",
(long long) swaps_total, o.adapt_every, ms_adapt / rounds);
if (src.complement_ready())
std::printf("%-24s %.2f GiB of experts in RAM, %lld exchanged with the VRAM tier, %lld blob reads from "
"the file\n", "resident RAM", (double) src.resident_bytes() / 1073741824.0,
(long long) src.exchanges(), (long long) src.file_reads());
if (srcp == &src) { // CS-T: the RAM and file tiers (the GPU cache's share is the hit rate above)
const double fms = src.file_ms() - fms0, fmb = (double) (src.file_blob_bytes() - fbytes0) / 1e6;
std::printf("%-24s decode: RAM %lld blobs, files %lld blobs, %.1f MB read from the files (%.2f MB/round, "
"%.1f ms/round of reading, %.2f GB/s per reading thread)%s; prompt copies %.1f MB\n",
"expert tiers", (long long) (src.ram_reads() - ram0), (long long) (src.file_reads() - files0),
fmb, rounds > 0 ? fmb / rounds : 0.0, rounds > 0 ? fms / rounds : 0.0,
fms > 0 ? fmb / fms : 0.0, src.gguf_mode() ? " (the GGUF in place)" : "",
(double) (fall0 - fbytes0) / 1e6);
if (drive.d.lookahead != nullptr)
std::printf("%-24s %lld experts warmed, %lld of the file tier's %lld blob reads had been warmed (%.1f%%); "
"predictor %.1f ms/round on its thread, %lld layers skipped (still busy)\n", "routing prefetch",
(long long) src.warmed(), (long long) src.warmed_hits(),
(long long) (src.file_reads() - files0),
src.file_reads() > files0 ? 100.0 * (double) src.warmed_hits() / (double) (src.file_reads() - files0) : 0.0,
rounds > 0 ? drive.d.lookahead->busy_ms() / rounds : 0.0, (long long) drive.d.lookahead->skipped());
}
if (rounds > 0 && drive.d.pcie_num > 0)
std::printf("%-24s %.2f distinct experts per layer read over PCIe (share %d/256 of the misses)\n",
"pcie experts", (double) (drive.d.pcie_experts - pcie0) / (double) (rounds * g.n_layers),
drive.d.pcie_num);
(void) pool_ms0;
if (use_mtp && rounds > 0)
std::printf("%-24s %.3f ms/round drafting (%lld rounds), MTP prompt %.1f ms, %.0f MiB of VRAM\n", "mtp",
mtp.ms_draft / (double) mtp.rounds, (long long) mtp.rounds, mtp.ms_prefill,
(double) mtp.vram_bytes() / 1048576.0);
}
if (dump != nullptr && std::fclose(dump) != 0) {
std::fprintf(stderr, "strata generate: cannot finish logits dump\n");
return 1;
}
if (layer_dump != nullptr) {
std::fclose(layer_dump);
cudaFreeHost(layer_stage);
std::printf("%-24s %s (%lld layers + the input x %d streams x %lld per position)\n", "layers dumped",
o.dump_layers.c_str(), (long long) g.n_layers, (int) g.hc, (long long) g.n_embd);
}
if (half_dump != nullptr) {
std::fclose(half_dump);
cudaFreeHost(half_stage);
std::printf("%-24s %s (%lld layers x %llu per position)\n", "halves dumped", o.dump_halves.c_str(),
(long long) g.n_layers, (unsigned long long) half_stride);
}
if (routing != nullptr) {
std::fclose(routing);
drive.routing = nullptr;
std::printf("%-24s %s (%lld records of layer, k, ids, weights)\n", "routing dumped",
o.dump_routing.c_str(), (long long) drive.calls);
}
if (o.stage_timing) strata::core::stage_timing_report(g.n_layers);
const int64_t decoded = (int64_t) produced.size();
std::printf("prompt :");
for (int64_t t : o.tokens) std::printf(" %lld", (long long) t);
std::printf("\noutput :");
for (int64_t t : produced) std::printf(" %lld", (long long) t);
std::printf("\n");
const double decode_ms = decoded > 0 ? total_ms / (double) decoded : 0.0;
std::printf("%-24s %lld tokens in %.1f ms -> %.2f tok/s\n", "decode", (long long) decoded, total_ms,
decode_ms > 0.0 ? 1000.0 / decode_ms : 0.0);
if (n_prompt > 1)
std::printf("%-24s %lld tokens in %.1f ms -> %.2f tok/s (time to first token %.1f ms)\n", "prefill",
(long long) (n_prompt - 1), prefill_ms,
prefill_ms > 0 ? 1000.0 * (double) (n_prompt - 1) / prefill_ms : 0.0, ttft_ms);
if (!o.dump_mixed.empty()) {
std::vector<float> mx((size_t) g.n_embd);
if (cudaMemcpy(mx.data(), ss.block.mixed, mx.size() * sizeof(float), cudaMemcpyDeviceToHost) !=
cudaSuccess) {
std::fprintf(stderr, "strata generate: reading mixed back failed\n");
return 1;
}
std::FILE* mf = std::fopen(o.dump_mixed.c_str(), "wb");
if (mf == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_mixed.c_str());
return 1;
}
std::fwrite(mx.data(), sizeof(float), mx.size(), mf);
std::fclose(mf);
double s2 = 0, mag = 0;
for (float v : mx) { s2 += (double) v * (double) v; mag += std::fabs((double) v); }
std::printf("%-24s %s (n_embd %lld, rms %.5g, mean|.| %.5g)\n", "mixed dumped", o.dump_mixed.c_str(),
(long long) g.n_embd, std::sqrt(s2 / (double) mx.size()), mag / (double) mx.size());
}
// ---- the residual, for bisecting the head against the layers (see `dump_residual`'s note)
if (!o.dump_residual.empty()) {
std::vector<float> R((size_t) g.hc * g.n_embd);
if (cudaMemcpy(R.data(), ss.R, R.size() * sizeof(float), cudaMemcpyDeviceToHost) != cudaSuccess) {
std::fprintf(stderr, "strata generate: reading R back failed\n");
return 1;
}
std::FILE* rf = std::fopen(o.dump_residual.c_str(), "wb");
if (rf == nullptr) {
std::fprintf(stderr, "strata generate: cannot write %s\n", o.dump_residual.c_str());
return 1;
}
const int32_t hdr[2] = {(int32_t) g.hc, (int32_t) g.n_embd};
std::fwrite(hdr, sizeof hdr, 1, rf);
std::fwrite(R.data(), sizeof(float), R.size(), rf);
std::fclose(rf);
double mag = 0, mx = 0;
int bad = 0;
for (float v : R) {
if (!std::isfinite(v)) ++bad;
else { mag += std::fabs((double) v); mx = std::max(mx, (double) std::fabs((double) v)); }
}
std::printf("%-24s %s (%d x %lld, nonfinite %d, mean|.| %.4g, max|.| %.4g)\n", "residual dumped",
o.dump_residual.c_str(), (int) g.hc, (long long) g.n_embd, bad, mag / (double) R.size(), mx);
}
if (o.stats) {
std::printf("%-24s %.3f ms/token (WALL CLOCK: embed, layers, head, sample)\n", " per token", decode_ms);
// **THE PER-TOKEN HOST TERM, WHICH `--gpu-only-full` CANNOT SEE.** That measurement never enters the
// token loop, so it excludes all six of these. On the 78-token fixture + 200 generated tokens the six
// sum to ~9 ms of non-layer work against ~1.5 ms of actual head GPU work - 16% of the token, and it is
// not the pool.
if (phase_tokens > 0) {
const double pt = (double) phase_tokens;
std::printf("%-24s PLE %.3f embed %.3f LAYERS %.3f head %.3f readback %.3f sample %.3f "
"(sum %.3f of %.3f ms)\n",
" token host phases", ms_ple / pt, ms_embed / pt, ms_layers / pt, ms_head / pt,
ms_readback / pt, ms_sample / pt,
(ms_ple + ms_embed + ms_layers + ms_head + ms_readback + ms_sample) / pt, decode_ms);
}
if (const std::string io = ple_table.io_report(); !io.empty()) std::printf(" %s\n", io.c_str());
// **THE DENOMINATOR IS THE POSITIONS THE POOL ACTUALLY RAN ON, NOT THE DECODED TOKENS (A6).**
// `drive_pool` is called once per layer per position and PREFILL runs the loop too, so accumulating
// `cpu_ms` over prefill and then dividing by `decoded` inflates this figure. `drive.calls / n_layers`
// is the number of positions - the same correction the ring counters below already received, which is
// why they print "of 192" rather than "240 of 192".
const double pool_positions = g.n_layers > 0 ? (double) drive.calls / (double) g.n_layers : 0.0;
std::printf("%-24s %.3f ms/token over %lld layers (%.0f positions, %lld dispatches)\n",
" the CPU expert pool", pool_positions > 0.0 ? drive.cpu_ms / pool_positions : 0.0,
(long long) g.n_layers, pool_positions, (long long) drive.calls);
// **AND WHERE INSIDE `run()` IT WENT.** Three phases per layer and they were one number, which cannot
// tell a pool that is slow at the WORK from one that is slow at the SYNCHRONISATION - opposite fixes.
// Wait-for-park is expected to be ~0 (the workers re-parked at the end of the previous layer); the
// question is whether the time is in the drain or in the re-park barrier.
if (pool_positions > 0.0) {
double wp = 0, dr = 0, rp = 0;
pool.phase_ms(wp, dr, rp);
const double per = pool_positions;
std::printf("%-24s wait-park %.3f drain %.3f re-park %.3f ms/token\n",
" pool phases", wp / per, dr / per, rp / per);
}
std::printf("%-24s %lld blobs read\n", " expert blobs", (long long) srcp->reads());
for (int r = 0; r < 3; ++r) if (o.expert_cache_remote[(size_t) r] > 0)
std::printf(" CUDA%d experts %lld routed entries computed\n",
r + 1, (long long) remote_experts[(size_t) r].computed());
// ---- **R4's DISPATCH MEASUREMENT: h, ON THE ENGINE'S OWN ROUTING.** No offline trace, no corpus
// question, no k-fold - these are the ids the router actually produced on this run. Reported as
// hits/lookups so it can be read directly as the h the cache would deliver, and alongside `refused`
// so a full cache is visible rather than silently capping the rate.
if (o.expert_cache > 0) {
const int64_t look = drive.d.cache_hits + drive.d.cache_admitted + drive.d.cache_refused;
const int64_t hl = drive.d.hit_ready + drive.d.hit_late;
std::printf("%-24s %lld of %lld layers the hit work was DONE when the pool returned\n",
" R4 overlap", (long long) drive.d.hit_ready, (long long) hl);
std::printf("%-24s %lld of %lld = %.4f (%lld admitted, %lld refused, cache %.4f%% full)\n",
" R4 expert-cache hits", (long long) drive.d.cache_hits, (long long) look,
look > 0 ? (double) drive.d.cache_hits / (double) look : 0.0,
(long long) drive.d.cache_admitted, (long long) drive.d.cache_refused,
100.0 * (double) (drive.d.cache_admitted + drive.d.cache_hits > 0
? (double) xcache.resident() / (double) xcache.slots()
: 0.0));
}
if (tgraph.captured && tgraph.calls > 0) {
const double per = (double) tgraph.calls;
std::printf("%-24s wait for rings %.3f pool %.3f ms/token (%lld flushes over %lld positions)\n",
" token graph", tgraph.ms_wait / per, tgraph.ms_pool / per, (long long) tgraph.flushes,
(long long) tgraph.calls);
}
if (gr.captured && gr.calls_total > 0) {
// The counters are CUMULATIVE over every `session_loop` call, and PREFILL runs the loop too - so
// the denominator is the number of positions, not the number of generated tokens. Dividing by
// `n_layers * decoded` printed "240 of 192", which is a reporting bug that looks like a ring
// firing more often than it should.
const int64_t positions = gr.calls_total;
std::printf("%-24s %lld of %lld over %lld positions\n", " rings seen MID-GRAPH",
(long long) gr.rings_mid_graph, (long long) (g.n_layers * positions),
(long long) positions);
std::printf("%-24s %.3f ms of a %.3f ms layer\n", " ring latency",
gr.ms_to_ring / (double) (g.n_layers * positions), decode_ms / (double) g.n_layers);
// **THE ROUND TRIP, SPLIT AT THE RING.** `ring latency` is the first half and stops when the ring
// is seen; this is the second half - the driver calls after it, during which the GPU is IDLE
// because `post[l]` has not been launched yet. `--no-pool` is the arm that isolates it: 38.73
// ms/token against a 26.32 ms pure-GPU floor is 12.4 ms of round trip with no expert work at all.
//
// Same denominator as the pool line above (the positions the loop actually ran on), so the two can
// be added without one of them being inflated by prefill.
const double perlap = (double) (g.n_layers * positions);
std::printf("%-24s %.3f ms/token over %.0f positions (%.3f ms/layer, after the ring)\n",
" host after ring", pool_positions > 0.0 ? gr.ms_host / pool_positions : 0.0,
pool_positions, gr.ms_host / perlap);
}
}
if (dump != nullptr) std::printf("%-24s %s\n", "logits dumped", o.dump_logits.c_str());
strata::core::session_graphs_free(gr);
strata::core::doorbell_free(db);
cudaFree(d_next);
cudaFree(d_logits);
cudaFree(d_emb);
cudaFree(d_parts);
cudaFree(sbuf);
cudaFree(arena);
return 0;
}