File size: 2,726 Bytes
39371ea
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
// Local audit harness; no UI state or production configuration is changed.
import { AutoTokenizer, AutoModelForCausalLM, env, random } from '@huggingface/transformers';
import { prepareModelCache } from '../src/download.mjs';
import { nucleusProcessor } from '../src/sampling.mjs';
let tokenizer, model;
export async function loadProbe({ name = 'minicpm5-webgpu', limit128 = false } = {}) {
  if (!['127.0.0.1', 'localhost'].includes(location.hostname)) throw Error('Local audit only.');
  if (!['minicpm5-webgpu', 'minicpm5-head-fp16'].includes(name)) throw Error('Unknown audit model.');
  if (limit128) {
    const requestDevice = GPUAdapter.prototype.requestDevice;
    GPUAdapter.prototype.requestDevice = function (descriptor = {}) {
      return requestDevice.call(this, { ...descriptor, requiredLimits: { ...descriptor.requiredLimits,
        maxStorageBufferBindingSize: 128 * 1024 ** 2, maxBufferSize: 256 * 1024 ** 2 } });
    };
  }
  env.allowLocalModels = false; env.allowRemoteModels = true;
  env.remoteHost = location.origin + '/'; env.remotePathTemplate = 'models/{model}/';
  if (name === 'minicpm5-webgpu') env.customCache = await prepareModelCache(location.origin + '/models/minicpm5-webgpu/', {signal: new AbortController().signal});
  env.useCustomCache = name === 'minicpm5-webgpu'; env.useBrowserCache = false;
  env.backends.onnx.wasm.wasmPaths = location.origin + '/runtime/'; env.backends.onnx.wasm.numThreads = 1;
  tokenizer = await AutoTokenizer.from_pretrained(name);
  model = await AutoModelForCausalLM.from_pretrained(name, {device:'webgpu',dtype:'q4f16'});
  const limits = env.backends.onnx.webgpu.device.limits;
  return {name, maxStorageBufferBindingSize:limits.maxStorageBufferBindingSize, maxBufferSize:limits.maxBufferSize};
}
export async function runProbe(fixture, {thinking, sample, penalty=1, seed=42, maxTokens=1024}) {
  random.seed(seed);
  const inputs = tokenizer.apply_chat_template(fixture.messages, {tools:fixture.tools, enable_thinking:thinking, add_generation_prompt:true, return_dict:true});
  const started = performance.now();
  const output = await model.generate({...inputs,max_new_tokens:maxTokens,do_sample:sample,temperature:1,top_k:0,top_p:.95,
    repetition_penalty:penalty,logits_processor:sample?[nucleusProcessor(.95)]:[],eos_token_id:[1,130073]});
  const ids=output.tolist()[0].slice(inputs.input_ids.dims[1]).map(Number);
  return {case:fixture.name,thinking,sample,penalty,seed,input_tokens:inputs.input_ids.dims[1],output_tokens:ids.length,
    elapsed_s:(performance.now()-started)/1000,ended:[1,130073].includes(ids.at(-1)),raw:tokenizer.decode(ids,{skip_special_tokens:false}),ids};
}
export async function disposeProbe(){await model?.dispose();}