MiniCPM5-2B-WebGPU-Pi-HTTP / app /tests /model-probe.mjs
Mike0021's picture
Enable browser HTTP commands with CORS-aware errors and bounded requests
39371ea verified
Raw History Blame Contribute Delete
2.73 kB
// Local audit harness; no UI state or production configuration is changed.
import { AutoTokenizer, AutoModelForCausalLM, env, random } from '@huggingface/transformers';
import { prepareModelCache } from '../src/download.mjs';
import { nucleusProcessor } from '../src/sampling.mjs';
let tokenizer, model;
export async function loadProbe({ name = 'minicpm5-webgpu', limit128 = false } = {}) {
if (!['127.0.0.1', 'localhost'].includes(location.hostname)) throw Error('Local audit only.');
if (!['minicpm5-webgpu', 'minicpm5-head-fp16'].includes(name)) throw Error('Unknown audit model.');
if (limit128) {
const requestDevice = GPUAdapter.prototype.requestDevice;
GPUAdapter.prototype.requestDevice = function (descriptor = {}) {
return requestDevice.call(this, { ...descriptor, requiredLimits: { ...descriptor.requiredLimits,
maxStorageBufferBindingSize: 128 * 1024 ** 2, maxBufferSize: 256 * 1024 ** 2 } });
};
}
env.allowLocalModels = false; env.allowRemoteModels = true;
env.remoteHost = location.origin + '/'; env.remotePathTemplate = 'models/{model}/';
if (name === 'minicpm5-webgpu') env.customCache = await prepareModelCache(location.origin + '/models/minicpm5-webgpu/', {signal: new AbortController().signal});
env.useCustomCache = name === 'minicpm5-webgpu'; env.useBrowserCache = false;
env.backends.onnx.wasm.wasmPaths = location.origin + '/runtime/'; env.backends.onnx.wasm.numThreads = 1;
tokenizer = await AutoTokenizer.from_pretrained(name);
model = await AutoModelForCausalLM.from_pretrained(name, {device:'webgpu',dtype:'q4f16'});
const limits = env.backends.onnx.webgpu.device.limits;
return {name, maxStorageBufferBindingSize:limits.maxStorageBufferBindingSize, maxBufferSize:limits.maxBufferSize};
}
export async function runProbe(fixture, {thinking, sample, penalty=1, seed=42, maxTokens=1024}) {
random.seed(seed);
const inputs = tokenizer.apply_chat_template(fixture.messages, {tools:fixture.tools, enable_thinking:thinking, add_generation_prompt:true, return_dict:true});
const started = performance.now();
const output = await model.generate({...inputs,max_new_tokens:maxTokens,do_sample:sample,temperature:1,top_k:0,top_p:.95,
repetition_penalty:penalty,logits_processor:sample?[nucleusProcessor(.95)]:[],eos_token_id:[1,130073]});
const ids=output.tolist()[0].slice(inputs.input_ids.dims[1]).map(Number);
return {case:fixture.name,thinking,sample,penalty,seed,input_tokens:inputs.input_ids.dims[1],output_tokens:ids.length,
elapsed_s:(performance.now()-started)/1000,ended:[1,130073].includes(ids.at(-1)),raw:tokenizer.decode(ids,{skip_special_tokens:false}),ids};
}
export async function disposeProbe(){await model?.dispose();}