/** * What this pipeline is missing, as sentences a reader can act on. * Empty means a burst is safe to run. */ export function missingPipelineMembers(pipeline: any): string[]; /** * Replaces `engine.decode` with a burst-and-drain version. * * `decode` stays a one-token call — the caller's loop still checks * `pipeline.stopped()` between tokens and still emits one chunk per token — but * only one call in K actually touches the GPU. The rest drain a buffer. * * @param {object} engine an MLCEngine; in this project the one inside the worker * @param {object} [options] * @param {number} [options.steps] forward steps per sync; 1 disables the path * @param {(info: {steps: number, tokens: number, ms: number}) => void} [options.onBurst] * @param {(info: {missing: string[]}) => void} [options.onFallback] fired once * per pipeline that fails the contract, before it is routed to stock decoding * @returns {{setSteps: (n: number) => void, readonly steps: number, * readonly fallbacks: number}} */ export function installMultiStepDecoding(engine: object, { steps, onBurst, onFallback }?: { steps?: number; onBurst?: (info: { steps: number; tokens: number; ms: number; }) => void; onFallback?: (info: { missing: string[]; }) => void; }): { setSteps: (n: number) => void; readonly steps: number; readonly fallbacks: number; }; /** * How many steps may run before the next stop condition *has* to be checked. * * `max_tokens` and the context window are countable, so they are clamped rather * than overshot — which leaves stop tokens as the only reason a burst is ever * rewound. Returns 1 when multi-step cannot be used at all, routing the caller * to stock single-step decoding. */ export function burstSize(pipeline: any, genConfig: any, steps: any): number; /** * Multi-step decoding: N forward steps per GPU->CPU sync. * * Why this exists: decode here is not compute-bound, it is *sync*-bound. Firefox * resolves `onSubmittedWorkDone()` / `mapAsync()` only on a 100 ms poll tick * (AI.md, "The 10 tok/s ceiling"), and stock WebLLM needs exactly one sync per * token — it reads the sampled token id back to JS before it can build the next * step's input. One token per tick = 9.6 tok/s, of which ~7 ms is real compute. * * The fix is the one vLLM ships as `--num-scheduler-steps`: run K steps before * paying the per-batch cost once. What makes it possible here without touching * the compiled model is that WebLLM's sampling path is *already* on the GPU — * `softmax_with_temperature`, `argsort_probs` and `sample_with_top_p` hand back * an int32[1] device tensor, and `Tensor.copyFrom(Tensor)` is a device-to-device * copy. So the sampled id feeds straight back into `embed` without ever becoming * a JS number: * * embed -> decode -> penalties -> softmax -> argsort -> sample -> embed -> ... * * Each step stages its id into its own CPU tensor, and the burst ends with * **one** `device.sync()`. tvmjs queues GPU->CPU copies into `pendingGPUToCPUCopy` * and only awaits them in `sync()`, so K readbacks still cost one tick. * * Two things follow from the 100 ms grid, and they are why `steps` is a dial: * * - The win is quantized, not linear. A burst costs `ceil(K * perStepMs / 100)` * ticks, so throughput is a sawtooth and the good values of K are the ones * landing just under a boundary. On a 0.8B at ~7.3 ms/step that is K=13 * (~130 tok/s); K=14 already spills into a second tick and halves it. * - The best K shrinks as the model grows, because `perStepMs` grows. A model * at 25 ms/step wants K=4, not K=15. * * Cost of the trick: the sampler cannot see its own output mid-burst. Repetition * and presence/frequency penalties use the token history as it stood when the * burst started, and stop conditions are only checked after the readback, so a * burst can overshoot a stop token and must then be rewound. Both are the same * trade vLLM makes. Anything needing per-token CPU feedback (grammar-constrained * JSON, logprobs, a logit processor) falls back to single-step, where behaviour * is identical to stock WebLLM. * * The other cost is that all of this drives ~30 undocumented tvmjs internals. A * WebLLM upgrade that renames one does not break generation — it turns the fast * path off and takes the throughput with it, silently. `PIPELINE_CONTRACT` below * is that surface written down and checked against the live pipeline before the * first burst, so the failure announces itself instead of being measured months * later. */ /** vLLM's documented sweet spot, and the value this extension ships. */ export const DEFAULT_DECODE_STEPS: 15; /** * Above this, the lookahead thrown away at a stop token outweighs the tick it * saves, and the transient logits/argsort buffers stop being free. */ export const MAX_DECODE_STEPS: 32; export function clampSteps(n: any): number; export namespace PIPELINE_CONTRACT { let calls: string[]; let numbers: string[]; let reads: string[]; let optional: string[]; }