nyaaorick's picture
feat: publish everything-webgpu package, engine source and documentation
1944112 verified
Raw History Blame Contribute Delete
2.62 kB
export class EnginePool {
constructor({ size, createEngine, onStateChange }: {
size?: number;
createEngine: any;
onStateChange?: () => void;
});
/** Engines that exist right now, which is not the same as the cap. */
get size(): number;
get maxSize(): number;
get loaded(): boolean;
/**
* Brings up the first engine, and only the first.
*
* The pool used to build every engine here. It no longer does, because an
* engine is a full copy of the weights - measured ~1.6 GB steady state for
* the 0.8B, ~2.4 GB for a 2B - and staging a second one costs that much host
* memory before the GPU ever sees it. On a 16 GB machine that is the
* difference between working and swapping, and it was being paid up front
* whether or not two tasks ever ran at once. `#grow` earns the rest.
*/
load(onProgress?: () => void): Promise<any>;
unload(): Promise<void>;
/**
* @param {object} spec
* @param {object} spec.params passed straight to `engine.chat.completions.create`
* @param {string} [spec.task] the unit that owns an engine; a whole batch shares one
* @param {string} [spec.session] later jobs with this session supersede earlier ones
* @param {string} [spec.priority] one of PRIORITY
* @param {boolean} [spec.preemptible] may be interrupted by an interactive job
* @param {(chunk: object) => void} [spec.onChunk] WebLLM's chunk, verbatim
* @returns {Promise<{text: string, usage?: object, cancelled?: boolean, preempted?: boolean}>}
*/
submit(spec: {
params: object;
task?: string;
session?: string;
priority?: string;
preemptible?: boolean;
onChunk?: (chunk: object) => void;
}): Promise<{
text: string;
usage?: object;
cancelled?: boolean;
preempted?: boolean;
}>;
/** Cancels by job id or by session key. Returns how many jobs it stopped. */
cancel(idOrSession: any): number;
/**
* Pushes a runtime setting to every engine that accepts one.
*
* Separate from `load()` because the settings this carries — `decodeSteps` so
* far — are per-generation knobs, not per-model ones: retuning them must not
* cost a reload of the weights.
*/
configure(patch: any): number;
status(): {
size: number;
maxSize: number;
growing: boolean;
growthBlocked: any;
busy: number;
queued: number;
queuedByPriority: {
[k: string]: number;
};
};
#private;
}