File size: 5,219 Bytes
1944112 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 | /**
* @typedef {object} DeviceProbe
* @property {boolean} webgpu
* @property {string} [reason] why WebGPU is unusable, when it is
* @property {object} [adapter] vendor / architecture / device, where exposed
* @property {{shaderF16: boolean}} [features]
* @property {object} [limits]
* @property {boolean} [kvReuse] whether cross-turn KV reuse can be used here
* @property {{quota?: number, usage?: number, persisted?: boolean}} storage
* @property {number} [deviceMemoryGB] Chrome only; absent is not "small"
*/
/**
* Reads what this machine will admit to. Never throws: an unusable device is a
* result, not an error — the caller's job is to explain it, not to crash.
*
* @returns {Promise<DeviceProbe>}
*/
export function probeDevice(): Promise<DeviceProbe>;
/**
* Whether a model can run on a probed device.
*
* `blockers` mean it will not work; `warnings` mean it will work worse, or
* might not fit. The split matters: a caller should refuse to start on a
* blocker and merely inform on a warning, and conflating the two is how you end
* up either crashing or refusing to run something that would have been fine.
*
* @param {{model_id: string, vram_required_MB?: number, sizeBytes?: number}} model
* @param {DeviceProbe} probe
* @returns {{ok: boolean, blockers: Array<{code: string, message: string}>,
* warnings: Array<{code: string, message: string}>}}
*/
export function canRun(model: {
model_id: string;
vram_required_MB?: number;
sizeBytes?: number;
}, probe: DeviceProbe): {
ok: boolean;
blockers: Array<{
code: string;
message: string;
}>;
warnings: Array<{
code: string;
message: string;
}>;
};
/**
* Rank a model list by what this device can actually run.
*
* The prebuilt list spans 239 MB to 31 GB, so "which model should I use" is the
* first question a developer has and the one they have least basis to answer.
* Runnable models come first, then fewest warnings; unrunnable ones are kept at
* the end carrying their reason rather than silently dropped, because "why
* can't I use that one" is the next question.
*
* **`prefer` is a real choice, not a default worth hiding.** Decode here is
* memory-bandwidth-bound — time per token scales with weight bytes (AI.md,
* "Why not llama.cpp/Ollama-class"), so the largest model that fits is also the
* slowest thing that fits. `"quality"` picks the biggest, `"speed"` the
* smallest. Neither is right for everyone, which is why it is a parameter.
*
* Vision models are excluded from a text ranking rather than merely deprioritised:
* a VLM answers text prompts perfectly well, but at several times the download
* for no benefit, so recommending one to a caller who did not ask is bad advice.
*
* @param {Array<object>} models `model_list` entries or registry records
* @param {{probe: DeviceProbe, maxVramMB?: number, needsVision?: boolean,
* prefer?: "quality" | "speed"}} opts
*/
export function rankModels(models: Array<object>, { probe, maxVramMB, needsVision, prefer }?: {
probe: DeviceProbe;
maxVramMB?: number;
needsVision?: boolean;
prefer?: "quality" | "speed";
}): {
ok: boolean;
blockers: Array<{
code: string;
message: string;
}>;
warnings: Array<{
code: string;
message: string;
}>;
model: any;
}[];
/**
* @param {number} modelBytes
* @param {number} [bytesPerSecond] this machine's measured rate, when known
* @returns {{tokensPerSecond: number, basis: "measured" | "extrapolated",
* modelBytes: number, bytesPerSecond: number, reference?: string}}
*/
export function projectSpeed(modelBytes: number, bytesPerSecond?: number): {
tokensPerSecond: number;
basis: "measured" | "extrapolated";
modelBytes: number;
bytesPerSecond: number;
reference?: string;
};
/**
* Decode throughput is memory bandwidth divided by weight bytes.
*
* This project measured the whole chain: decode reaches ~16 GB/s of the M4's
* ~120 GB/s, and the 1.06 GB build runs 16.6–18.1 tok/s — which is that
* quotient. So a projection needs one number, the *achieved* bandwidth, and
* everything else follows from model size.
*
* The constant below is that machine's figure and is only a starting point. The
* moment this engine has decoded anything it knows the real number for the
* machine it is on, and `ScheduledEngine.estimateSpeed()` switches to it — so
* this is a cold-start default, not a claim about anyone's hardware.
*/
export const REFERENCE_DECODE_BYTES_PER_SECOND: 17000000000;
export const REFERENCE_DEVICE: "M4 MacBook Air (16 GB), Firefox";
export type DeviceProbe = {
webgpu: boolean;
/**
* why WebGPU is unusable, when it is
*/
reason?: string;
/**
* vendor / architecture / device, where exposed
*/
adapter?: object;
features?: {
shaderF16: boolean;
};
limits?: object;
/**
* whether cross-turn KV reuse can be used here
*/
kvReuse?: boolean;
storage: {
quota?: number;
usage?: number;
persisted?: boolean;
};
/**
* Chrome only; absent is not "small"
*/
deviceMemoryGB?: number;
};
|