everything-webgpu / types /engine /device.d.ts
nyaaorick's picture
feat: publish everything-webgpu package, engine source and documentation
1944112 verified
Raw History Blame Contribute Delete
5.22 kB
/**
* @typedef {object} DeviceProbe
* @property {boolean} webgpu
* @property {string} [reason] why WebGPU is unusable, when it is
* @property {object} [adapter] vendor / architecture / device, where exposed
* @property {{shaderF16: boolean}} [features]
* @property {object} [limits]
* @property {boolean} [kvReuse] whether cross-turn KV reuse can be used here
* @property {{quota?: number, usage?: number, persisted?: boolean}} storage
* @property {number} [deviceMemoryGB] Chrome only; absent is not "small"
*/
/**
* Reads what this machine will admit to. Never throws: an unusable device is a
* result, not an error — the caller's job is to explain it, not to crash.
*
* @returns {Promise<DeviceProbe>}
*/
export function probeDevice(): Promise<DeviceProbe>;
/**
* Whether a model can run on a probed device.
*
* `blockers` mean it will not work; `warnings` mean it will work worse, or
* might not fit. The split matters: a caller should refuse to start on a
* blocker and merely inform on a warning, and conflating the two is how you end
* up either crashing or refusing to run something that would have been fine.
*
* @param {{model_id: string, vram_required_MB?: number, sizeBytes?: number}} model
* @param {DeviceProbe} probe
* @returns {{ok: boolean, blockers: Array<{code: string, message: string}>,
* warnings: Array<{code: string, message: string}>}}
*/
export function canRun(model: {
model_id: string;
vram_required_MB?: number;
sizeBytes?: number;
}, probe: DeviceProbe): {
ok: boolean;
blockers: Array<{
code: string;
message: string;
}>;
warnings: Array<{
code: string;
message: string;
}>;
};
/**
* Rank a model list by what this device can actually run.
*
* The prebuilt list spans 239 MB to 31 GB, so "which model should I use" is the
* first question a developer has and the one they have least basis to answer.
* Runnable models come first, then fewest warnings; unrunnable ones are kept at
* the end carrying their reason rather than silently dropped, because "why
* can't I use that one" is the next question.
*
* **`prefer` is a real choice, not a default worth hiding.** Decode here is
* memory-bandwidth-bound — time per token scales with weight bytes (AI.md,
* "Why not llama.cpp/Ollama-class"), so the largest model that fits is also the
* slowest thing that fits. `"quality"` picks the biggest, `"speed"` the
* smallest. Neither is right for everyone, which is why it is a parameter.
*
* Vision models are excluded from a text ranking rather than merely deprioritised:
* a VLM answers text prompts perfectly well, but at several times the download
* for no benefit, so recommending one to a caller who did not ask is bad advice.
*
* @param {Array<object>} models `model_list` entries or registry records
* @param {{probe: DeviceProbe, maxVramMB?: number, needsVision?: boolean,
* prefer?: "quality" | "speed"}} opts
*/
export function rankModels(models: Array<object>, { probe, maxVramMB, needsVision, prefer }?: {
probe: DeviceProbe;
maxVramMB?: number;
needsVision?: boolean;
prefer?: "quality" | "speed";
}): {
ok: boolean;
blockers: Array<{
code: string;
message: string;
}>;
warnings: Array<{
code: string;
message: string;
}>;
model: any;
}[];
/**
* @param {number} modelBytes
* @param {number} [bytesPerSecond] this machine's measured rate, when known
* @returns {{tokensPerSecond: number, basis: "measured" | "extrapolated",
* modelBytes: number, bytesPerSecond: number, reference?: string}}
*/
export function projectSpeed(modelBytes: number, bytesPerSecond?: number): {
tokensPerSecond: number;
basis: "measured" | "extrapolated";
modelBytes: number;
bytesPerSecond: number;
reference?: string;
};
/**
* Decode throughput is memory bandwidth divided by weight bytes.
*
* This project measured the whole chain: decode reaches ~16 GB/s of the M4's
* ~120 GB/s, and the 1.06 GB build runs 16.6–18.1 tok/s — which is that
* quotient. So a projection needs one number, the *achieved* bandwidth, and
* everything else follows from model size.
*
* The constant below is that machine's figure and is only a starting point. The
* moment this engine has decoded anything it knows the real number for the
* machine it is on, and `ScheduledEngine.estimateSpeed()` switches to it — so
* this is a cold-start default, not a claim about anyone's hardware.
*/
export const REFERENCE_DECODE_BYTES_PER_SECOND: 17000000000;
export const REFERENCE_DEVICE: "M4 MacBook Air (16 GB), Firefox";
export type DeviceProbe = {
webgpu: boolean;
/**
* why WebGPU is unusable, when it is
*/
reason?: string;
/**
* vendor / architecture / device, where exposed
*/
adapter?: object;
features?: {
shaderF16: boolean;
};
limits?: object;
/**
* whether cross-turn KV reuse can be used here
*/
kvReuse?: boolean;
storage: {
quota?: number;
usage?: number;
persisted?: boolean;
};
/**
* Chrome only; absent is not "small"
*/
deviceMemoryGB?: number;
};