File size: 5,219 Bytes
1944112
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
/**
 * @typedef {object} DeviceProbe
 * @property {boolean} webgpu
 * @property {string} [reason] why WebGPU is unusable, when it is
 * @property {object} [adapter] vendor / architecture / device, where exposed
 * @property {{shaderF16: boolean}} [features]
 * @property {object} [limits]
 * @property {boolean} [kvReuse] whether cross-turn KV reuse can be used here
 * @property {{quota?: number, usage?: number, persisted?: boolean}} storage
 * @property {number} [deviceMemoryGB] Chrome only; absent is not "small"
 */
/**
 * Reads what this machine will admit to. Never throws: an unusable device is a
 * result, not an error — the caller's job is to explain it, not to crash.
 *
 * @returns {Promise<DeviceProbe>}
 */
export function probeDevice(): Promise<DeviceProbe>;
/**
 * Whether a model can run on a probed device.
 *
 * `blockers` mean it will not work; `warnings` mean it will work worse, or
 * might not fit. The split matters: a caller should refuse to start on a
 * blocker and merely inform on a warning, and conflating the two is how you end
 * up either crashing or refusing to run something that would have been fine.
 *
 * @param {{model_id: string, vram_required_MB?: number, sizeBytes?: number}} model
 * @param {DeviceProbe} probe
 * @returns {{ok: boolean, blockers: Array<{code: string, message: string}>,
 *   warnings: Array<{code: string, message: string}>}}
 */
export function canRun(model: {
    model_id: string;
    vram_required_MB?: number;
    sizeBytes?: number;
}, probe: DeviceProbe): {
    ok: boolean;
    blockers: Array<{
        code: string;
        message: string;
    }>;
    warnings: Array<{
        code: string;
        message: string;
    }>;
};
/**
 * Rank a model list by what this device can actually run.
 *
 * The prebuilt list spans 239 MB to 31 GB, so "which model should I use" is the
 * first question a developer has and the one they have least basis to answer.
 * Runnable models come first, then fewest warnings; unrunnable ones are kept at
 * the end carrying their reason rather than silently dropped, because "why
 * can't I use that one" is the next question.
 *
 * **`prefer` is a real choice, not a default worth hiding.** Decode here is
 * memory-bandwidth-bound — time per token scales with weight bytes (AI.md,
 * "Why not llama.cpp/Ollama-class"), so the largest model that fits is also the
 * slowest thing that fits. `"quality"` picks the biggest, `"speed"` the
 * smallest. Neither is right for everyone, which is why it is a parameter.
 *
 * Vision models are excluded from a text ranking rather than merely deprioritised:
 * a VLM answers text prompts perfectly well, but at several times the download
 * for no benefit, so recommending one to a caller who did not ask is bad advice.
 *
 * @param {Array<object>} models `model_list` entries or registry records
 * @param {{probe: DeviceProbe, maxVramMB?: number, needsVision?: boolean,
 *   prefer?: "quality" | "speed"}} opts
 */
export function rankModels(models: Array<object>, { probe, maxVramMB, needsVision, prefer }?: {
    probe: DeviceProbe;
    maxVramMB?: number;
    needsVision?: boolean;
    prefer?: "quality" | "speed";
}): {
    ok: boolean;
    blockers: Array<{
        code: string;
        message: string;
    }>;
    warnings: Array<{
        code: string;
        message: string;
    }>;
    model: any;
}[];
/**
 * @param {number} modelBytes
 * @param {number} [bytesPerSecond] this machine's measured rate, when known
 * @returns {{tokensPerSecond: number, basis: "measured" | "extrapolated",
 *   modelBytes: number, bytesPerSecond: number, reference?: string}}
 */
export function projectSpeed(modelBytes: number, bytesPerSecond?: number): {
    tokensPerSecond: number;
    basis: "measured" | "extrapolated";
    modelBytes: number;
    bytesPerSecond: number;
    reference?: string;
};
/**
 * Decode throughput is memory bandwidth divided by weight bytes.
 *
 * This project measured the whole chain: decode reaches ~16 GB/s of the M4's
 * ~120 GB/s, and the 1.06 GB build runs 16.6–18.1 tok/s — which is that
 * quotient. So a projection needs one number, the *achieved* bandwidth, and
 * everything else follows from model size.
 *
 * The constant below is that machine's figure and is only a starting point. The
 * moment this engine has decoded anything it knows the real number for the
 * machine it is on, and `ScheduledEngine.estimateSpeed()` switches to it — so
 * this is a cold-start default, not a claim about anyone's hardware.
 */
export const REFERENCE_DECODE_BYTES_PER_SECOND: 17000000000;
export const REFERENCE_DEVICE: "M4 MacBook Air (16 GB), Firefox";
export type DeviceProbe = {
    webgpu: boolean;
    /**
     * why WebGPU is unusable, when it is
     */
    reason?: string;
    /**
     * vendor / architecture / device, where exposed
     */
    adapter?: object;
    features?: {
        shaderF16: boolean;
    };
    limits?: object;
    /**
     * whether cross-turn KV reuse can be used here
     */
    kvReuse?: boolean;
    storage: {
        quota?: number;
        usage?: number;
        persisted?: boolean;
    };
    /**
     * Chrome only; absent is not "small"
     */
    deviceMemoryGB?: number;
};