Download types/engine/multistep.d.ts from nyaaorick/everything-webgpu: direct link, hf CLI and curl.
- Browser
- Download file 5.12 kB
-
https://huggingface.co/nyaaorick/everything-webgpu/resolve/main/types/engine/multistep.d.ts
- Command line
-
hf download hf://nyaaorick/everything-webgpu/types/engine/multistep.d.ts
-
curl -L -o multistep.d.ts https://huggingface.co/nyaaorick/everything-webgpu/resolve/main/types/engine/multistep.d.ts
5.12 kB
| /** | |
| * What this pipeline is missing, as sentences a reader can act on. | |
| * Empty means a burst is safe to run. | |
| */ | |
| export function missingPipelineMembers(pipeline: any): string[]; | |
| /** | |
| * Replaces `engine.decode` with a burst-and-drain version. | |
| * | |
| * `decode` stays a one-token call β the caller's loop still checks | |
| * `pipeline.stopped()` between tokens and still emits one chunk per token β but | |
| * only one call in K actually touches the GPU. The rest drain a buffer. | |
| * | |
| * @param {object} engine an MLCEngine; in this project the one inside the worker | |
| * @param {object} [options] | |
| * @param {number} [options.steps] forward steps per sync; 1 disables the path | |
| * @param {(info: {steps: number, tokens: number, ms: number}) => void} [options.onBurst] | |
| * @param {(info: {missing: string[]}) => void} [options.onFallback] fired once | |
| * per pipeline that fails the contract, before it is routed to stock decoding | |
| * @returns {{setSteps: (n: number) => void, readonly steps: number, | |
| * readonly fallbacks: number}} | |
| */ | |
| export function installMultiStepDecoding(engine: object, { steps, onBurst, onFallback }?: { | |
| steps?: number; | |
| onBurst?: (info: { | |
| steps: number; | |
| tokens: number; | |
| ms: number; | |
| }) => void; | |
| onFallback?: (info: { | |
| missing: string[]; | |
| }) => void; | |
| }): { | |
| setSteps: (n: number) => void; | |
| readonly steps: number; | |
| readonly fallbacks: number; | |
| }; | |
| /** | |
| * How many steps may run before the next stop condition *has* to be checked. | |
| * | |
| * `max_tokens` and the context window are countable, so they are clamped rather | |
| * than overshot β which leaves stop tokens as the only reason a burst is ever | |
| * rewound. Returns 1 when multi-step cannot be used at all, routing the caller | |
| * to stock single-step decoding. | |
| */ | |
| export function burstSize(pipeline: any, genConfig: any, steps: any): number; | |
| /** | |
| * Multi-step decoding: N forward steps per GPU->CPU sync. | |
| * | |
| * Why this exists: decode here is not compute-bound, it is *sync*-bound. Firefox | |
| * resolves `onSubmittedWorkDone()` / `mapAsync()` only on a 100 ms poll tick | |
| * (AI.md, "The 10 tok/s ceiling"), and stock WebLLM needs exactly one sync per | |
| * token β it reads the sampled token id back to JS before it can build the next | |
| * step's input. One token per tick = 9.6 tok/s, of which ~7 ms is real compute. | |
| * | |
| * The fix is the one vLLM ships as `--num-scheduler-steps`: run K steps before | |
| * paying the per-batch cost once. What makes it possible here without touching | |
| * the compiled model is that WebLLM's sampling path is *already* on the GPU β | |
| * `softmax_with_temperature`, `argsort_probs` and `sample_with_top_p` hand back | |
| * an int32[1] device tensor, and `Tensor.copyFrom(Tensor)` is a device-to-device | |
| * copy. So the sampled id feeds straight back into `embed` without ever becoming | |
| * a JS number: | |
| * | |
| * embed -> decode -> penalties -> softmax -> argsort -> sample -> embed -> ... | |
| * | |
| * Each step stages its id into its own CPU tensor, and the burst ends with | |
| * **one** `device.sync()`. tvmjs queues GPU->CPU copies into `pendingGPUToCPUCopy` | |
| * and only awaits them in `sync()`, so K readbacks still cost one tick. | |
| * | |
| * Two things follow from the 100 ms grid, and they are why `steps` is a dial: | |
| * | |
| * - The win is quantized, not linear. A burst costs `ceil(K * perStepMs / 100)` | |
| * ticks, so throughput is a sawtooth and the good values of K are the ones | |
| * landing just under a boundary. On a 0.8B at ~7.3 ms/step that is K=13 | |
| * (~130 tok/s); K=14 already spills into a second tick and halves it. | |
| * - The best K shrinks as the model grows, because `perStepMs` grows. A model | |
| * at 25 ms/step wants K=4, not K=15. | |
| * | |
| * Cost of the trick: the sampler cannot see its own output mid-burst. Repetition | |
| * and presence/frequency penalties use the token history as it stood when the | |
| * burst started, and stop conditions are only checked after the readback, so a | |
| * burst can overshoot a stop token and must then be rewound. Both are the same | |
| * trade vLLM makes. Anything needing per-token CPU feedback (grammar-constrained | |
| * JSON, logprobs, a logit processor) falls back to single-step, where behaviour | |
| * is identical to stock WebLLM. | |
| * | |
| * The other cost is that all of this drives ~30 undocumented tvmjs internals. A | |
| * WebLLM upgrade that renames one does not break generation β it turns the fast | |
| * path off and takes the throughput with it, silently. `PIPELINE_CONTRACT` below | |
| * is that surface written down and checked against the live pipeline before the | |
| * first burst, so the failure announces itself instead of being measured months | |
| * later. | |
| */ | |
| /** vLLM's documented sweet spot, and the value this extension ships. */ | |
| export const DEFAULT_DECODE_STEPS: 15; | |
| /** | |
| * Above this, the lookahead thrown away at a stop token outweighs the tick it | |
| * saves, and the transient logits/argsort buffers stop being free. | |
| */ | |
| export const MAX_DECODE_STEPS: 32; | |
| export function clampSteps(n: any): number; | |
| export namespace PIPELINE_CONTRACT { | |
| let calls: string[]; | |
| let numbers: string[]; | |
| let reads: string[]; | |
| let optional: string[]; | |
| } | |