/** * @typedef {object} DeviceProbe * @property {boolean} webgpu * @property {string} [reason] why WebGPU is unusable, when it is * @property {object} [adapter] vendor / architecture / device, where exposed * @property {{shaderF16: boolean}} [features] * @property {object} [limits] * @property {boolean} [kvReuse] whether cross-turn KV reuse can be used here * @property {{quota?: number, usage?: number, persisted?: boolean}} storage * @property {number} [deviceMemoryGB] Chrome only; absent is not "small" */ /** * Reads what this machine will admit to. Never throws: an unusable device is a * result, not an error — the caller's job is to explain it, not to crash. * * @returns {Promise} */ export function probeDevice(): Promise; /** * Whether a model can run on a probed device. * * `blockers` mean it will not work; `warnings` mean it will work worse, or * might not fit. The split matters: a caller should refuse to start on a * blocker and merely inform on a warning, and conflating the two is how you end * up either crashing or refusing to run something that would have been fine. * * @param {{model_id: string, vram_required_MB?: number, sizeBytes?: number}} model * @param {DeviceProbe} probe * @returns {{ok: boolean, blockers: Array<{code: string, message: string}>, * warnings: Array<{code: string, message: string}>}} */ export function canRun(model: { model_id: string; vram_required_MB?: number; sizeBytes?: number; }, probe: DeviceProbe): { ok: boolean; blockers: Array<{ code: string; message: string; }>; warnings: Array<{ code: string; message: string; }>; }; /** * Rank a model list by what this device can actually run. * * The prebuilt list spans 239 MB to 31 GB, so "which model should I use" is the * first question a developer has and the one they have least basis to answer. * Runnable models come first, then fewest warnings; unrunnable ones are kept at * the end carrying their reason rather than silently dropped, because "why * can't I use that one" is the next question. * * **`prefer` is a real choice, not a default worth hiding.** Decode here is * memory-bandwidth-bound — time per token scales with weight bytes (AI.md, * "Why not llama.cpp/Ollama-class"), so the largest model that fits is also the * slowest thing that fits. `"quality"` picks the biggest, `"speed"` the * smallest. Neither is right for everyone, which is why it is a parameter. * * Vision models are excluded from a text ranking rather than merely deprioritised: * a VLM answers text prompts perfectly well, but at several times the download * for no benefit, so recommending one to a caller who did not ask is bad advice. * * @param {Array} models `model_list` entries or registry records * @param {{probe: DeviceProbe, maxVramMB?: number, needsVision?: boolean, * prefer?: "quality" | "speed"}} opts */ export function rankModels(models: Array, { probe, maxVramMB, needsVision, prefer }?: { probe: DeviceProbe; maxVramMB?: number; needsVision?: boolean; prefer?: "quality" | "speed"; }): { ok: boolean; blockers: Array<{ code: string; message: string; }>; warnings: Array<{ code: string; message: string; }>; model: any; }[]; /** * @param {number} modelBytes * @param {number} [bytesPerSecond] this machine's measured rate, when known * @returns {{tokensPerSecond: number, basis: "measured" | "extrapolated", * modelBytes: number, bytesPerSecond: number, reference?: string}} */ export function projectSpeed(modelBytes: number, bytesPerSecond?: number): { tokensPerSecond: number; basis: "measured" | "extrapolated"; modelBytes: number; bytesPerSecond: number; reference?: string; }; /** * Decode throughput is memory bandwidth divided by weight bytes. * * This project measured the whole chain: decode reaches ~16 GB/s of the M4's * ~120 GB/s, and the 1.06 GB build runs 16.6–18.1 tok/s — which is that * quotient. So a projection needs one number, the *achieved* bandwidth, and * everything else follows from model size. * * The constant below is that machine's figure and is only a starting point. The * moment this engine has decoded anything it knows the real number for the * machine it is on, and `ScheduledEngine.estimateSpeed()` switches to it — so * this is a cold-start default, not a claim about anyone's hardware. */ export const REFERENCE_DECODE_BYTES_PER_SECOND: 17000000000; export const REFERENCE_DEVICE: "M4 MacBook Air (16 GB), Firefox"; export type DeviceProbe = { webgpu: boolean; /** * why WebGPU is unusable, when it is */ reason?: string; /** * vendor / architecture / device, where exposed */ adapter?: object; features?: { shaderF16: boolean; }; limits?: object; /** * whether cross-turn KV reuse can be used here */ kvReuse?: boolean; storage: { quota?: number; usage?: number; persisted?: boolean; }; /** * Chrome only; absent is not "small" */ deviceMemoryGB?: number; };