File size: 2,646 Bytes
d0fb3e3
d636c52
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
// Hearlark worker: runs Whisper in the browser with transformers.js (WebGPU when available, else WASM).
// Audio arrives as 16 kHz mono Float32Array; nothing is uploaded anywhere. © 2026 CyberMax.
import { pipeline, WhisperTextStreamer, env } from 'https://cdn.jsdelivr.net/npm/@huggingface/transformers@4.3.0';

env.allowLocalModels = false;

let asr = null;
let loadedKey = null;

async function load(model, device, post) {
  const key = `${model}|${device}`;
  if (asr && loadedKey === key) return asr;
  if (asr) { try { await asr.dispose(); } catch {} asr = null; loadedKey = null; }
  const dtype = device === 'webgpu' ? { encoder_model: 'fp32', decoder_model_merged: 'q4' } : 'q8';
  asr = await pipeline('automatic-speech-recognition', model, {
    device,
    dtype,
    progress_callback: (p) => post({ type: 'load', status: p.status, file: p.file, loaded: p.loaded, total: p.total, progress: p.progress }),
  });
  loadedKey = key;
  return asr;
}

self.onmessage = async ({ data }) => {
  if (data?.type !== 'run') return;
  const post = (m) => self.postMessage(m);
  try {
    let device = data.device;
    let p;
    try {
      p = await load(data.model, device, post);
    } catch (e) {
      if (device !== 'webgpu') throw e;
      post({ type: 'info', message: `WebGPU could not start (${e?.message || e}); using the CPU instead.` });
      device = 'wasm';
      p = await load(data.model, device, post);
    }
    post({ type: 'ready', device });

    const englishOnly = /\.en$/.test(data.model);
    const opts = {
      chunk_length_s: 30,
      stride_length_s: 5,
      return_timestamps: true,
      ...(englishOnly ? {} : { language: data.language || null, task: data.task === 'translate' ? 'translate' : 'transcribe' }),
    };
    let finalized = 0;
    let out;
    try {
      const time_precision = p.processor.feature_extractor.config.chunk_length / p.model.config.max_source_positions;
      const streamer = new WhisperTextStreamer(p.tokenizer, {
        time_precision,
        skip_prompt: true,
        callback_function: (text) => post({ type: 'partial', text }),
        on_finalize: () => post({ type: 'chunk', done: ++finalized }),
      });
      out = await p(data.audio, { ...opts, streamer });
    } catch (e) {
      // Streaming is only for live progress; if this transformers.js build rejects it, run without it.
      if (!/stream/i.test(String(e?.message || e))) throw e;
      out = await p(data.audio, opts);
    }
    post({ type: 'done', result: { text: out.text, chunks: out.chunks || [] }, device });
  } catch (e) {
    post({ type: 'error', message: String(e?.message || e) });
  }
};