Download types/engine/device.d.ts from nyaaorick/everything-webgpu: direct link, hf CLI and curl.
- Browser
- Download file 5.22 kB
-
https://huggingface.co/nyaaorick/everything-webgpu/resolve/main/types/engine/device.d.ts
- Command line
-
hf download hf://nyaaorick/everything-webgpu/types/engine/device.d.ts
-
curl -L -o device.d.ts https://huggingface.co/nyaaorick/everything-webgpu/resolve/main/types/engine/device.d.ts
5.22 kB
| /** | |
| * @typedef {object} DeviceProbe | |
| * @property {boolean} webgpu | |
| * @property {string} [reason] why WebGPU is unusable, when it is | |
| * @property {object} [adapter] vendor / architecture / device, where exposed | |
| * @property {{shaderF16: boolean}} [features] | |
| * @property {object} [limits] | |
| * @property {boolean} [kvReuse] whether cross-turn KV reuse can be used here | |
| * @property {{quota?: number, usage?: number, persisted?: boolean}} storage | |
| * @property {number} [deviceMemoryGB] Chrome only; absent is not "small" | |
| */ | |
| /** | |
| * Reads what this machine will admit to. Never throws: an unusable device is a | |
| * result, not an error — the caller's job is to explain it, not to crash. | |
| * | |
| * @returns {Promise<DeviceProbe>} | |
| */ | |
| export function probeDevice(): Promise<DeviceProbe>; | |
| /** | |
| * Whether a model can run on a probed device. | |
| * | |
| * `blockers` mean it will not work; `warnings` mean it will work worse, or | |
| * might not fit. The split matters: a caller should refuse to start on a | |
| * blocker and merely inform on a warning, and conflating the two is how you end | |
| * up either crashing or refusing to run something that would have been fine. | |
| * | |
| * @param {{model_id: string, vram_required_MB?: number, sizeBytes?: number}} model | |
| * @param {DeviceProbe} probe | |
| * @returns {{ok: boolean, blockers: Array<{code: string, message: string}>, | |
| * warnings: Array<{code: string, message: string}>}} | |
| */ | |
| export function canRun(model: { | |
| model_id: string; | |
| vram_required_MB?: number; | |
| sizeBytes?: number; | |
| }, probe: DeviceProbe): { | |
| ok: boolean; | |
| blockers: Array<{ | |
| code: string; | |
| message: string; | |
| }>; | |
| warnings: Array<{ | |
| code: string; | |
| message: string; | |
| }>; | |
| }; | |
| /** | |
| * Rank a model list by what this device can actually run. | |
| * | |
| * The prebuilt list spans 239 MB to 31 GB, so "which model should I use" is the | |
| * first question a developer has and the one they have least basis to answer. | |
| * Runnable models come first, then fewest warnings; unrunnable ones are kept at | |
| * the end carrying their reason rather than silently dropped, because "why | |
| * can't I use that one" is the next question. | |
| * | |
| * **`prefer` is a real choice, not a default worth hiding.** Decode here is | |
| * memory-bandwidth-bound — time per token scales with weight bytes (AI.md, | |
| * "Why not llama.cpp/Ollama-class"), so the largest model that fits is also the | |
| * slowest thing that fits. `"quality"` picks the biggest, `"speed"` the | |
| * smallest. Neither is right for everyone, which is why it is a parameter. | |
| * | |
| * Vision models are excluded from a text ranking rather than merely deprioritised: | |
| * a VLM answers text prompts perfectly well, but at several times the download | |
| * for no benefit, so recommending one to a caller who did not ask is bad advice. | |
| * | |
| * @param {Array<object>} models `model_list` entries or registry records | |
| * @param {{probe: DeviceProbe, maxVramMB?: number, needsVision?: boolean, | |
| * prefer?: "quality" | "speed"}} opts | |
| */ | |
| export function rankModels(models: Array<object>, { probe, maxVramMB, needsVision, prefer }?: { | |
| probe: DeviceProbe; | |
| maxVramMB?: number; | |
| needsVision?: boolean; | |
| prefer?: "quality" | "speed"; | |
| }): { | |
| ok: boolean; | |
| blockers: Array<{ | |
| code: string; | |
| message: string; | |
| }>; | |
| warnings: Array<{ | |
| code: string; | |
| message: string; | |
| }>; | |
| model: any; | |
| }[]; | |
| /** | |
| * @param {number} modelBytes | |
| * @param {number} [bytesPerSecond] this machine's measured rate, when known | |
| * @returns {{tokensPerSecond: number, basis: "measured" | "extrapolated", | |
| * modelBytes: number, bytesPerSecond: number, reference?: string}} | |
| */ | |
| export function projectSpeed(modelBytes: number, bytesPerSecond?: number): { | |
| tokensPerSecond: number; | |
| basis: "measured" | "extrapolated"; | |
| modelBytes: number; | |
| bytesPerSecond: number; | |
| reference?: string; | |
| }; | |
| /** | |
| * Decode throughput is memory bandwidth divided by weight bytes. | |
| * | |
| * This project measured the whole chain: decode reaches ~16 GB/s of the M4's | |
| * ~120 GB/s, and the 1.06 GB build runs 16.6–18.1 tok/s — which is that | |
| * quotient. So a projection needs one number, the *achieved* bandwidth, and | |
| * everything else follows from model size. | |
| * | |
| * The constant below is that machine's figure and is only a starting point. The | |
| * moment this engine has decoded anything it knows the real number for the | |
| * machine it is on, and `ScheduledEngine.estimateSpeed()` switches to it — so | |
| * this is a cold-start default, not a claim about anyone's hardware. | |
| */ | |
| export const REFERENCE_DECODE_BYTES_PER_SECOND: 17000000000; | |
| export const REFERENCE_DEVICE: "M4 MacBook Air (16 GB), Firefox"; | |
| export type DeviceProbe = { | |
| webgpu: boolean; | |
| /** | |
| * why WebGPU is unusable, when it is | |
| */ | |
| reason?: string; | |
| /** | |
| * vendor / architecture / device, where exposed | |
| */ | |
| adapter?: object; | |
| features?: { | |
| shaderF16: boolean; | |
| }; | |
| limits?: object; | |
| /** | |
| * whether cross-turn KV reuse can be used here | |
| */ | |
| kvReuse?: boolean; | |
| storage: { | |
| quota?: number; | |
| usage?: number; | |
| persisted?: boolean; | |
| }; | |
| /** | |
| * Chrome only; absent is not "small" | |
| */ | |
| deviceMemoryGB?: number; | |
| }; | |