// Local audit harness; no UI state or production configuration is changed. import { AutoTokenizer, AutoModelForCausalLM, env, random } from '@huggingface/transformers'; import { prepareModelCache } from '../src/download.mjs'; import { nucleusProcessor } from '../src/sampling.mjs'; let tokenizer, model; export async function loadProbe({ name = 'minicpm5-webgpu', limit128 = false } = {}) { if (!['127.0.0.1', 'localhost'].includes(location.hostname)) throw Error('Local audit only.'); if (!['minicpm5-webgpu', 'minicpm5-head-fp16'].includes(name)) throw Error('Unknown audit model.'); if (limit128) { const requestDevice = GPUAdapter.prototype.requestDevice; GPUAdapter.prototype.requestDevice = function (descriptor = {}) { return requestDevice.call(this, { ...descriptor, requiredLimits: { ...descriptor.requiredLimits, maxStorageBufferBindingSize: 128 * 1024 ** 2, maxBufferSize: 256 * 1024 ** 2 } }); }; } env.allowLocalModels = false; env.allowRemoteModels = true; env.remoteHost = location.origin + '/'; env.remotePathTemplate = 'models/{model}/'; if (name === 'minicpm5-webgpu') env.customCache = await prepareModelCache(location.origin + '/models/minicpm5-webgpu/', {signal: new AbortController().signal}); env.useCustomCache = name === 'minicpm5-webgpu'; env.useBrowserCache = false; env.backends.onnx.wasm.wasmPaths = location.origin + '/runtime/'; env.backends.onnx.wasm.numThreads = 1; tokenizer = await AutoTokenizer.from_pretrained(name); model = await AutoModelForCausalLM.from_pretrained(name, {device:'webgpu',dtype:'q4f16'}); const limits = env.backends.onnx.webgpu.device.limits; return {name, maxStorageBufferBindingSize:limits.maxStorageBufferBindingSize, maxBufferSize:limits.maxBufferSize}; } export async function runProbe(fixture, {thinking, sample, penalty=1, seed=42, maxTokens=1024}) { random.seed(seed); const inputs = tokenizer.apply_chat_template(fixture.messages, {tools:fixture.tools, enable_thinking:thinking, add_generation_prompt:true, return_dict:true}); const started = performance.now(); const output = await model.generate({...inputs,max_new_tokens:maxTokens,do_sample:sample,temperature:1,top_k:0,top_p:.95, repetition_penalty:penalty,logits_processor:sample?[nucleusProcessor(.95)]:[],eos_token_id:[1,130073]}); const ids=output.tolist()[0].slice(inputs.input_ids.dims[1]).map(Number); return {case:fixture.name,thinking,sample,penalty,seed,input_tokens:inputs.input_ids.dims[1],output_tokens:ids.length, elapsed_s:(performance.now()-started)/1000,ended:[1,130073].includes(ids.at(-1)),raw:tokenizer.decode(ids,{skip_special_tokens:false}),ids}; } export async function disposeProbe(){await model?.dispose();}