example-router / worker.mjs
Future-Labs's picture
Load and validate pinned encoder configuration at inference initialization
8c5bbab verified
Raw History Blame Contribute Delete
3.71 kB
import { Tokenizer } from 'https://cdn.jsdelivr.net/npm/@huggingface/tokenizers@0.2.0/dist/tokenizers.mjs';
import * as ort from 'https://cdn.jsdelivr.net/npm/onnxruntime-web@1.23.2/dist/ort.wasm.min.mjs';
import {encodeMessage} from './tokenize.mjs';
import {buildRouter,validateRouter,predict} from './router.mjs';
ort.env.wasm.numThreads=1;
ort.env.wasm.wasmPaths='https://cdn.jsdelivr.net/npm/onnxruntime-web@1.23.2/dist/';
let engine,router;
async function getJSON(url){const r=await fetch(url);if(!r.ok)throw Error('Could not load model files.');return r.json();}
async function load(){
const runtime=await getJSON('./runtime.json');
const config=await getJSON(runtime.config_url);
if(config.format!=='example-router-config-v1'||config.embedding_model!=='minilm-l6-v2-fp16-v1'||config.dimensions!==384||config.max_length!==256||config.pooling!=='attention_mask_mean_l2')throw Error('Unsupported encoder configuration.');
self.postMessage({type:'status',text:'Downloading the 45 MB encoder. Examples stay in this tab.'});
const [vocab,tc,response]=await Promise.all([getJSON('./tokenizer.json'),getJSON('./tokenizer_config.json'),fetch(config.model_url)]);
if(!response.ok)throw Error('Model download failed. Retry when connected.');
const bytes=await response.arrayBuffer();
const hash=Array.from(new Uint8Array(await crypto.subtle.digest('SHA-256',bytes)),x=>x.toString(16).padStart(2,'0')).join('');
if(hash!==config.model_sha256)throw Error('Model checksum mismatch.');
return {tokenizer:new Tokenizer(vocab,tc),session:await ort.InferenceSession.create(bytes,{executionProviders:['wasm']})};
}
async function embed(text){
if(typeof text!=='string'||!text.trim()||text.length>4000)throw Error('Use a nonempty message of at most 4,000 characters.');
engine??=load().catch(e=>{engine=null;throw e;});const {tokenizer,session}=await engine;const encoded=encodeMessage(tokenizer,text);const feeds={};
for(const name of ['input_ids','attention_mask','token_type_ids'])feeds[name]=new ort.Tensor('int64',BigInt64Array.from(encoded[name],BigInt),[1,encoded.input_ids.length]);
const outputs=await session.run(feeds);const vector=Array.from(outputs.embeddings.data);
Object.values(feeds).forEach(t=>t.dispose());Object.values(outputs).forEach(t=>t.dispose());return {vector,truncated:encoded.truncated};
}
self.onmessage=async({data})=>{try{
if(data.action==='build'){
const groups=data.groups;
if(!Array.isArray(groups)||groups.length<2||groups.length>20)throw Error('Build between 2 and 20 routes.');
if(new Set(groups.map(g=>g.label.trim())).size!==groups.length)throw Error('Route names must be unique.');
const texts=groups.flatMap(g=>g.examples);if(texts.length>200||groups.some(g=>!g.examples.length))throw Error('Use at least one example per route, up to 200 total.');
const vectors=[];let truncated=0;
for(let i=0;i<texts.length;i++){const e=await embed(texts[i]);vectors.push(e.vector);truncated+=Number(e.truncated);self.postMessage({type:'status',text:`Learning from example ${i+1} of ${texts.length} on your device…`});}
router=buildRouter(groups,vectors);self.postMessage({type:'built',router,truncated});
}else if(data.action==='import'){router=validateRouter(data.router);self.postMessage({type:'imported',router});
}else if(data.action==='predict'){if(!router)throw Error('Build or import a router first.');const e=await embed(data.text);self.postMessage({type:'prediction',...predict(router,e.vector,data.threshold),truncated:e.truncated});
}else if(data.action==='embed'){const e=await embed(data.text);self.postMessage({type:'embedding',...e});}
else throw Error('Unknown action.');
}catch(e){self.postMessage({type:'error',message:e.message});}};