File size: 2,616 Bytes
1944112
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
export class EnginePool {
    constructor({ size, createEngine, onStateChange }: {
        size?: number;
        createEngine: any;
        onStateChange?: () => void;
    });
    /** Engines that exist right now, which is not the same as the cap. */
    get size(): number;
    get maxSize(): number;
    get loaded(): boolean;
    /**
     * Brings up the first engine, and only the first.
     *
     * The pool used to build every engine here. It no longer does, because an
     * engine is a full copy of the weights - measured ~1.6 GB steady state for
     * the 0.8B, ~2.4 GB for a 2B - and staging a second one costs that much host
     * memory before the GPU ever sees it. On a 16 GB machine that is the
     * difference between working and swapping, and it was being paid up front
     * whether or not two tasks ever ran at once. `#grow` earns the rest.
     */
    load(onProgress?: () => void): Promise<any>;
    unload(): Promise<void>;
    /**
     * @param {object} spec
     * @param {object} spec.params  passed straight to `engine.chat.completions.create`
     * @param {string} [spec.task] the unit that owns an engine; a whole batch shares one
     * @param {string} [spec.session] later jobs with this session supersede earlier ones
     * @param {string} [spec.priority] one of PRIORITY
     * @param {boolean} [spec.preemptible] may be interrupted by an interactive job
     * @param {(chunk: object) => void} [spec.onChunk] WebLLM's chunk, verbatim
     * @returns {Promise<{text: string, usage?: object, cancelled?: boolean, preempted?: boolean}>}
     */
    submit(spec: {
        params: object;
        task?: string;
        session?: string;
        priority?: string;
        preemptible?: boolean;
        onChunk?: (chunk: object) => void;
    }): Promise<{
        text: string;
        usage?: object;
        cancelled?: boolean;
        preempted?: boolean;
    }>;
    /** Cancels by job id or by session key. Returns how many jobs it stopped. */
    cancel(idOrSession: any): number;
    /**
     * Pushes a runtime setting to every engine that accepts one.
     *
     * Separate from `load()` because the settings this carries — `decodeSteps` so
     * far — are per-generation knobs, not per-model ones: retuning them must not
     * cost a reload of the weights.
     */
    configure(patch: any): number;
    status(): {
        size: number;
        maxSize: number;
        growing: boolean;
        growthBlocked: any;
        busy: number;
        queued: number;
        queuedByPriority: {
            [k: string]: number;
        };
    };
    #private;
}