Download app/tests/model-probe.mjs from Mike0021/MiniCPM5-2B-WebGPU-Pi-HTTP: direct link, hf CLI and curl.
- Browser
- Download file 2.73 kB
-
https://huggingface.co/spaces/Mike0021/MiniCPM5-2B-WebGPU-Pi-HTTP/resolve/main/app/tests/model-probe.mjs
- Command line
-
hf download hf://spaces/Mike0021/MiniCPM5-2B-WebGPU-Pi-HTTP/app/tests/model-probe.mjs
-
curl -L -o model-probe.mjs https://huggingface.co/spaces/Mike0021/MiniCPM5-2B-WebGPU-Pi-HTTP/resolve/main/app/tests/model-probe.mjs
2.73 kB
| // Local audit harness; no UI state or production configuration is changed. | |
| import { AutoTokenizer, AutoModelForCausalLM, env, random } from '@huggingface/transformers'; | |
| import { prepareModelCache } from '../src/download.mjs'; | |
| import { nucleusProcessor } from '../src/sampling.mjs'; | |
| let tokenizer, model; | |
| export async function loadProbe({ name = 'minicpm5-webgpu', limit128 = false } = {}) { | |
| if (!['127.0.0.1', 'localhost'].includes(location.hostname)) throw Error('Local audit only.'); | |
| if (!['minicpm5-webgpu', 'minicpm5-head-fp16'].includes(name)) throw Error('Unknown audit model.'); | |
| if (limit128) { | |
| const requestDevice = GPUAdapter.prototype.requestDevice; | |
| GPUAdapter.prototype.requestDevice = function (descriptor = {}) { | |
| return requestDevice.call(this, { ...descriptor, requiredLimits: { ...descriptor.requiredLimits, | |
| maxStorageBufferBindingSize: 128 * 1024 ** 2, maxBufferSize: 256 * 1024 ** 2 } }); | |
| }; | |
| } | |
| env.allowLocalModels = false; env.allowRemoteModels = true; | |
| env.remoteHost = location.origin + '/'; env.remotePathTemplate = 'models/{model}/'; | |
| if (name === 'minicpm5-webgpu') env.customCache = await prepareModelCache(location.origin + '/models/minicpm5-webgpu/', {signal: new AbortController().signal}); | |
| env.useCustomCache = name === 'minicpm5-webgpu'; env.useBrowserCache = false; | |
| env.backends.onnx.wasm.wasmPaths = location.origin + '/runtime/'; env.backends.onnx.wasm.numThreads = 1; | |
| tokenizer = await AutoTokenizer.from_pretrained(name); | |
| model = await AutoModelForCausalLM.from_pretrained(name, {device:'webgpu',dtype:'q4f16'}); | |
| const limits = env.backends.onnx.webgpu.device.limits; | |
| return {name, maxStorageBufferBindingSize:limits.maxStorageBufferBindingSize, maxBufferSize:limits.maxBufferSize}; | |
| } | |
| export async function runProbe(fixture, {thinking, sample, penalty=1, seed=42, maxTokens=1024}) { | |
| random.seed(seed); | |
| const inputs = tokenizer.apply_chat_template(fixture.messages, {tools:fixture.tools, enable_thinking:thinking, add_generation_prompt:true, return_dict:true}); | |
| const started = performance.now(); | |
| const output = await model.generate({...inputs,max_new_tokens:maxTokens,do_sample:sample,temperature:1,top_k:0,top_p:.95, | |
| repetition_penalty:penalty,logits_processor:sample?[nucleusProcessor(.95)]:[],eos_token_id:[1,130073]}); | |
| const ids=output.tolist()[0].slice(inputs.input_ids.dims[1]).map(Number); | |
| return {case:fixture.name,thinking,sample,penalty,seed,input_tokens:inputs.input_ids.dims[1],output_tokens:ids.length, | |
| elapsed_s:(performance.now()-started)/1000,ended:[1,130073].includes(ids.at(-1)),raw:tokenizer.decode(ids,{skip_special_tokens:false}),ids}; | |
| } | |
| export async function disposeProbe(){await model?.dispose();} | |