Download app.bundle.js from RobinsAIWorld/granite-speech-streaming-webgpu: direct link, hf CLI and curl.
- Browser
- Download file 70.9 kB
-
https://huggingface.co/spaces/RobinsAIWorld/granite-speech-streaming-webgpu/resolve/main/app.bundle.js
- Command line
-
hf download hf://spaces/RobinsAIWorld/granite-speech-streaming-webgpu/app.bundle.js
-
curl -L -o app.bundle.js https://huggingface.co/spaces/RobinsAIWorld/granite-speech-streaming-webgpu/resolve/main/app.bundle.js
70.9 kB
| /* | |
| * GENERATED by build_bundle.py -- do not edit. | |
| * | |
| * The app's ES modules concatenated into one classic script. A private | |
| * Space's cross-site iframe does not authenticate the module fetches issued | |
| * while the page is parsing (they 401, and the module map then caches that | |
| * failure), whereas classic scripts always carry the cookie. Edit the | |
| * sources listed below and re-run build_bundle.py. | |
| * | |
| * Sources, in dependency order: ctc-frontend.js, ctc-tokenizer.js, ctc-engine.js, ctc-streaming.js, app.js | |
| */ | |
| (() => { | |
| ; | |
| // Stands in for the module namespace: each module destructures what it | |
| // imports off this and assigns its exports back onto it. | |
| const __M = {}; | |
| // ==== ctc-frontend.js =================================================== | |
| (() => { | |
| /** | |
| * Audio front-end for the FastConformer CTC model. | |
| * | |
| * A port of the model's `CtcConformerProcessor._frontend` (log-mel + deltas + | |
| * frame stacking) plus the sidecar's `_agc`. The mel filterbank and the STFT | |
| * window are NOT recomputed here -- they are loaded from `mel_filters.bin` / | |
| * `stft_window.bin`, dumped by export_ctc_onnx.py straight out of the torchaudio | |
| * objects the Python path uses. Reimplementing torchaudio's htk mel scale | |
| * (norm=None) and torch.stft's window centring in JS is the most likely place for | |
| * a silent mismatch, and a front-end that is subtly wrong degrades WER without | |
| * ever raising an error. | |
| * | |
| * Verified bit-for-bit against the Python features by test-frontend.mjs. | |
| */ | |
| const EPS = 1e-10; | |
| /** In-place iterative radix-2 complex FFT (Cooley-Tukey, decimation in time). */ | |
| class FFT { | |
| constructor(size) { | |
| this.size = size; | |
| this.levels = Math.log2(size) | 0; | |
| if (2 ** this.levels !== size) throw new Error(`FFT size ${size} is not a power of two`); | |
| // Bit-reversal permutation and twiddle tables, precomputed once: the | |
| // front-end runs this transform ~100 times per second of audio. | |
| this.rev = new Uint32Array(size); | |
| for (let i = 0; i < size; i++) { | |
| let r = 0; | |
| for (let b = 0; b < this.levels; b++) r |= ((i >> b) & 1) << (this.levels - 1 - b); | |
| this.rev[i] = r; | |
| } | |
| this.cos = new Float32Array(size / 2); | |
| this.sin = new Float32Array(size / 2); | |
| for (let i = 0; i < size / 2; i++) { | |
| this.cos[i] = Math.cos((-2 * Math.PI * i) / size); | |
| this.sin[i] = Math.sin((-2 * Math.PI * i) / size); | |
| } | |
| } | |
| /** Transform `re`/`im` in place (both length `size`). */ | |
| run(re, im) { | |
| const { size, rev, cos, sin } = this; | |
| for (let i = 0; i < size; i++) { | |
| const j = rev[i]; | |
| if (j > i) { | |
| let t = re[i]; re[i] = re[j]; re[j] = t; | |
| t = im[i]; im[i] = im[j]; im[j] = t; | |
| } | |
| } | |
| for (let len = 2; len <= size; len <<= 1) { | |
| const half = len >> 1; | |
| const step = size / len; | |
| for (let i = 0; i < size; i += len) { | |
| for (let j = 0, k = 0; j < half; j++, k += step) { | |
| const c = cos[k], s = sin[k]; | |
| const a = i + j, b = a + half; | |
| const tre = re[b] * c - im[b] * s; | |
| const tim = re[b] * s + im[b] * c; | |
| re[b] = re[a] - tre; im[b] = im[a] - tim; | |
| re[a] += tre; im[a] += tim; | |
| } | |
| } | |
| } | |
| } | |
| } | |
| /** | |
| * Centred moving average, cumsum-based (O(n)). | |
| * | |
| * Mirrors the sidecar's `_moving_avg`, including its edge handling: the valid | |
| * region is padded by replicating the first/last valid value, with the extra | |
| * sample going on the right when the window is even. | |
| */ | |
| function movingAvg(v, win) { | |
| const n = v.length; | |
| win = Math.max(1, win | 0); | |
| if (win <= 1 || n <= 1) return v; | |
| win = Math.min(win, n); | |
| // float64 accumulator: a float32 cumsum over ~10 s of samples loses enough | |
| // precision to shift the gain envelope (the Python side uses float64 too). | |
| const csum = new Float64Array(n + 1); | |
| for (let i = 0; i < n; i++) csum[i + 1] = csum[i] + v[i]; | |
| const valid = n - win + 1; | |
| const padL = (win - 1) >> 1; | |
| const out = new Float32Array(n); | |
| const first = (csum[win] - csum[0]) / win; | |
| const last = (csum[n] - csum[n - win]) / win; | |
| for (let i = 0; i < n; i++) { | |
| const k = i - padL; | |
| out[i] = k < 0 ? first : k >= valid ? last : (csum[k + win] - csum[k]) / win; | |
| } | |
| return out; | |
| } | |
| /** | |
| * Automatic gain control -- port of the sidecar's `_agc`. | |
| * | |
| * Not cosmetic: the CTC front-end normalizes log-mel against the clip's *global | |
| * peak*, so a quiet tail 20-40 dB below a loud head gets floored and silently | |
| * dropped. A smoothed moving-RMS gain toward `target` equalizes the levels. | |
| */ | |
| function agc(x, { sampleRate = 16000, target = 0.12, winMs = 150, maxGain = 20 } = {}) { | |
| const n = x.length; | |
| if (n === 0) return x; | |
| const win = Math.max(1, Math.round((sampleRate * winMs) / 1000)); | |
| const sq = new Float32Array(n); | |
| for (let i = 0; i < n; i++) sq[i] = x[i] * x[i]; | |
| const ms = movingAvg(sq, win); | |
| const g = new Float32Array(n); | |
| for (let i = 0; i < n; i++) { | |
| const env = Math.sqrt(ms[i] + 1e-9); | |
| g[i] = Math.min(Math.max(target / Math.max(env, 1e-4), 0), maxGain); | |
| } | |
| const gs = movingAvg(g, win); // smooth the gain to avoid pumping | |
| const y = new Float32Array(n); | |
| let peak = 0; | |
| for (let i = 0; i < n; i++) { | |
| y[i] = x[i] * gs[i]; | |
| const a = Math.abs(y[i]); | |
| if (a > peak) peak = a; | |
| } | |
| if (peak > 0.97) { | |
| const k = 0.97 / peak; | |
| for (let i = 0; i < n; i++) y[i] *= k; | |
| } | |
| return y; | |
| } | |
| class CtcFrontend { | |
| constructor(config, melFilters, window) { | |
| this.config = config; | |
| this.melFilters = melFilters; // [n_mels][n_freqs], mel-major | |
| this.window = window; // n_fft, already centred by the exporter | |
| this.fft = new FFT(config.n_fft); | |
| this.nFreqs = config.n_fft / 2 + 1; | |
| } | |
| static async load(baseUrl = './onnx-ctc/') { | |
| const get = async (name, as) => { | |
| const r = await fetch(baseUrl + name); | |
| if (!r.ok) throw new Error(`front-end asset ${name}: HTTP ${r.status}`); | |
| return as === 'json' ? r.json() : new Float32Array(await r.arrayBuffer()); | |
| }; | |
| const [config, melFilters, window] = await Promise.all([ | |
| get('frontend_config.json', 'json'), | |
| get('mel_filters.bin'), | |
| get('stft_window.bin'), | |
| ]); | |
| return new CtcFrontend(config, melFilters, window); | |
| } | |
| /** Feature frames a clip of `nSamples` produces (before padding to pad_multiple). */ | |
| frameCount(nSamples) { | |
| const s = this.config.stack_factor; | |
| const melFrames = Math.floor(nSamples / this.config.hop_length); | |
| return s * Math.ceil(melFrames / s); | |
| } | |
| /** | |
| * waveform (Float32Array, 16 kHz mono) -> stacked log-mel(+delta) features. | |
| * | |
| * Returns `{ data, frames, dim }` where `data` is [frames, dim] row-major, | |
| * ready to become the model's `input_features` tensor. | |
| */ | |
| compute(wav) { | |
| const { hop_length: hop, n_fft: nFft, n_mels: nMels, stack_factor: stack, | |
| logmel_floor_db: floorDb, deltas: useDeltas } = this.config; | |
| const nFrames = this.frameCount(wav.length); | |
| if (nFrames === 0) return { data: new Float32Array(0), frames: 0, dim: this.config.input_dim }; | |
| // The processor right-pads the waveform so the trailing partial stack | |
| // group is filled rather than dropped, then slices to exactly nFrames. | |
| const need = (nFrames - 1) * hop + 1; | |
| let x = wav; | |
| if (x.length < need) { | |
| x = new Float32Array(need); | |
| x.set(wav); | |
| } | |
| // torch.stft(center=True, pad_mode='reflect'): reflect-pad by n_fft/2 so | |
| // frame i is centred on sample i*hop. | |
| const half = nFft >> 1; | |
| const padded = new Float32Array(x.length + nFft); | |
| padded.set(x, half); | |
| for (let i = 0; i < half; i++) { | |
| padded[half - 1 - i] = x[Math.min(i + 1, x.length - 1)]; | |
| padded[half + x.length + i] = x[Math.max(x.length - 2 - i, 0)]; | |
| } | |
| // Mel spectrogram, mel-major ([nMels][nFrames]) because deltas run along time. | |
| const mel = new Float32Array(nMels * nFrames); | |
| const re = new Float32Array(nFft); | |
| const im = new Float32Array(nFft); | |
| const power = new Float32Array(this.nFreqs); | |
| for (let t = 0; t < nFrames; t++) { | |
| const off = t * hop; | |
| for (let i = 0; i < nFft; i++) { | |
| re[i] = padded[off + i] * this.window[i]; | |
| im[i] = 0; | |
| } | |
| this.fft.run(re, im); | |
| for (let k = 0; k < this.nFreqs; k++) power[k] = re[k] * re[k] + im[k] * im[k]; | |
| for (let m = 0; m < nMels; m++) { | |
| const fb = m * this.nFreqs; | |
| let acc = 0; | |
| for (let k = 0; k < this.nFreqs; k++) acc += this.melFilters[fb + k] * power[k]; | |
| mel[m * nFrames + t] = acc; | |
| } | |
| } | |
| // log10 with a floor relative to the clip's global peak, then the /4 +1 | |
| // scaling the model was trained with. | |
| let mx = -Infinity; | |
| for (let i = 0; i < mel.length; i++) { | |
| const v = Math.log10(Math.max(mel[i], EPS)); | |
| mel[i] = v; | |
| if (v > mx) mx = v; | |
| } | |
| const floor = mx - floorDb; | |
| for (let i = 0; i < mel.length; i++) mel[i] = Math.max(mel[i], floor) / 4 + 1; | |
| // Deltas: torchaudio.functional.compute_deltas(win_length=3) is | |
| // (x[t+1] - x[t-1]) / 2 with replicate edge padding. | |
| const nCh = useDeltas ? nMels * 2 : nMels; | |
| const chans = new Float32Array(nCh * nFrames); | |
| chans.set(mel); | |
| if (useDeltas) { | |
| for (let m = 0; m < nMels; m++) { | |
| const src = m * nFrames; | |
| const dst = (nMels + m) * nFrames; | |
| for (let t = 0; t < nFrames; t++) { | |
| const prev = mel[src + Math.max(t - 1, 0)]; | |
| const next = mel[src + Math.min(t + 1, nFrames - 1)]; | |
| chans[dst + t] = (next - prev) / 2; | |
| } | |
| } | |
| } | |
| // Stack: (nCh, nFrames) -> (nFrames/stack, stack*nCh), i.e. row j is the | |
| // channels of frame stack*j followed by those of frame stack*j+1. | |
| const outFrames = nFrames / stack; | |
| const dim = stack * nCh; | |
| const data = new Float32Array(outFrames * dim); | |
| for (let j = 0; j < outFrames; j++) { | |
| for (let k = 0; k < stack; k++) { | |
| const t = j * stack + k; | |
| const base = j * dim + k * nCh; | |
| for (let c = 0; c < nCh; c++) data[base + c] = chans[c * nFrames + t]; | |
| } | |
| } | |
| return { data, frames: outFrames, dim }; | |
| } | |
| } | |
| Object.assign(__M, { agc, CtcFrontend }); | |
| })(); | |
| // ==== ctc-tokenizer.js ================================================== | |
| (() => { | |
| /** | |
| * Decoder for the model's 16384-entry ByteLevel BPE vocabulary. | |
| * | |
| * Only decoding is needed (the model never encodes text), so export_ctc_onnx.py | |
| * dumps the id->piece slice that can reach the decoder as vocab.json (~170 KB) | |
| * instead of shipping the whole tokenizer.json. | |
| * | |
| * granite-speech-4.2-470m-turboctc uses a GPT-2 style **ByteLevel** tokenizer: | |
| * every byte 0-255 is represented by one printable codepoint (space is `Ġ`), so a | |
| * piece is a string of byte-stand-ins rather than text, and one UTF-8 character | |
| * may be split across several pieces ("é" arrives as "Ã" + "©"). Decoding is | |
| * therefore: map each character back to its byte, concatenate, then UTF-8 decode | |
| * once at the end -- never per piece, or multi-byte characters break. | |
| * | |
| * (The earlier C-T26.18 build used a SentencePiece tokenizer whose decoder was | |
| * Replace(U+2581 -> space) / ByteFallback / Fuse / Strip. That is a different | |
| * algorithm and would silently produce mojibake here, so the exporter asserts the | |
| * tokenizer's decoder type matches this implementation.) | |
| */ | |
| /** | |
| * GPT-2's byte <-> codepoint table, inverted to codepoint -> byte. | |
| * | |
| * The 188 bytes that are already printable map to themselves; the remaining 68 | |
| * are moved up to U+0100 and beyond so no piece ever contains a control | |
| * character or a bare space. | |
| */ | |
| function byteLevelCharToByte() { | |
| const bytes = []; | |
| for (let b = 0x21; b <= 0x7e; b++) bytes.push(b); // !..~ | |
| for (let b = 0xa1; b <= 0xac; b++) bytes.push(b); // ¡..¬ | |
| for (let b = 0xae; b <= 0xff; b++) bytes.push(b); // ®..ÿ | |
| const codepoints = bytes.slice(); | |
| const printable = new Set(bytes); | |
| let next = 0; | |
| for (let b = 0; b < 256; b++) { | |
| if (!printable.has(b)) { | |
| bytes.push(b); | |
| codepoints.push(256 + next); | |
| next++; | |
| } | |
| } | |
| const map = new Map(); | |
| for (let i = 0; i < bytes.length; i++) map.set(codepoints[i], bytes[i]); | |
| return map; | |
| } | |
| class CtcDecoder { | |
| constructor(pieces, numSpecialTokens) { | |
| this.numSpecialTokens = numSpecialTokens; | |
| const charToByte = byteLevelCharToByte(); | |
| // Pre-resolve every piece to its bytes, so decoding a frame of ids is a | |
| // few array copies rather than per-call string work. | |
| this.encoded = new Array(pieces.length); | |
| for (let i = 0; i < pieces.length; i++) { | |
| const piece = pieces[i]; | |
| const bytes = new Uint8Array(piece.length); | |
| let n = 0; | |
| for (const ch of piece) { | |
| const b = charToByte.get(ch.codePointAt(0)); | |
| // A codepoint outside the table means this vocab was not produced | |
| // by a ByteLevel tokenizer; skip rather than emit a wrong byte. | |
| if (b !== undefined) bytes[n++] = b; | |
| } | |
| this.encoded[i] = bytes.subarray(0, n); | |
| } | |
| this.utf8Decoder = new TextDecoder('utf-8'); | |
| } | |
| static async load(baseUrl = './onnx-ctc/') { | |
| const [vocab, config] = await Promise.all([ | |
| fetch(baseUrl + 'vocab.json').then((r) => { | |
| if (!r.ok) throw new Error(`vocab.json: HTTP ${r.status}`); | |
| return r.json(); | |
| }), | |
| fetch(baseUrl + 'frontend_config.json').then((r) => { | |
| if (!r.ok) throw new Error(`frontend_config.json: HTTP ${r.status}`); | |
| return r.json(); | |
| }), | |
| ]); | |
| return new CtcDecoder(vocab, config.num_special_tokens); | |
| } | |
| /** Decode content ids (all >= numSpecialTokens) to text. */ | |
| decode(ids) { | |
| let total = 0; | |
| for (const id of ids) { | |
| const e = this.encoded[id - this.numSpecialTokens]; | |
| if (e) total += e.length; | |
| } | |
| const buf = new Uint8Array(total); | |
| let at = 0; | |
| for (const id of ids) { | |
| const e = this.encoded[id - this.numSpecialTokens]; | |
| if (!e) continue; // out-of-range id: skip rather than poison the string | |
| buf.set(e, at); | |
| at += e.length; | |
| } | |
| return this.utf8Decoder.decode(buf); | |
| } | |
| } | |
| /** | |
| * Greedy CTC collapse: drop repeats, then the blank, then any remaining | |
| * below-content ids. | |
| * | |
| * Order matters and matches the Python path: repeats are collapsed against the | |
| * *raw* previous id (blanks included), which is what lets a blank between two | |
| * identical tokens preserve both. For this model `numSpecialTokens` is 1 -- the | |
| * CTC blank at id 0 is the only non-text id, and everything above it is a | |
| * tokenizer id as-is. | |
| */ | |
| function ctcCollapse(ids, { blankId = 0, numSpecialTokens = 1 } = {}) { | |
| const out = []; | |
| let prev = -1; | |
| for (let i = 0; i < ids.length; i++) { | |
| const id = ids[i]; | |
| if (id !== prev && id !== blankId && id >= numSpecialTokens) out.push(id); | |
| prev = id; | |
| } | |
| return out; | |
| } | |
| Object.assign(__M, { CtcDecoder, ctcCollapse }); | |
| })(); | |
| // ==== ctc-engine.js ===================================================== | |
| (() => { | |
| const { CtcFrontend, agc, CtcDecoder, ctcCollapse } = __M; | |
| /** | |
| * FastConformer CTC inference engine (onnxruntime-web, WebGPU). | |
| * | |
| * The model is a custom `ctc_conformer` architecture with no transformers.js | |
| * support, so this drives an ONNX session directly: features -> one forward pass | |
| * -> greedy CTC collapse. There is no autoregressive loop, no KV cache and no | |
| * chat template; a window is one `session.run`. | |
| * | |
| * Two contracts come from export_ctc_onnx.py and must hold here: | |
| * | |
| * - **Frames are padded to `pad_multiple`.** The encoder chunks attention into | |
| * context_size (128) blocks at every layer and subsamples time by 4, so a | |
| * frame count with no partial block anywhere is a multiple of 512 (10.24 s). | |
| * The exported graph therefore has no tail-block path, and short windows are | |
| * padded up with `padding_mask` marking the pad. Measured cost of the masked | |
| * padding on granite-speech-4.2-470m-turboctc: none -- all 14 fixture clips | |
| * decode identically to the unpadded path, with zero per-frame argmax flips. | |
| * (Its encoder masks inside the conv module specifically so a padded frame | |
| * cannot leak through the stride-2 kernel into the last valid frames.) | |
| * | |
| * - **The graph returns ids, not logits.** `[1, T/4, 16384]` float32 logits | |
| * would be an ~8 MB device-to-host copy per window; argmax and the | |
| * log-softmax max happen inside the graph instead, so a window costs ~1 KB of | |
| * output. `top_logprob` is the confidence signal the streaming gate uses. | |
| */ | |
| const CACHE_NAME = 'granite-ctc-models-v1'; | |
| /** | |
| * Cache-first fetch, so a reload doesn't re-download hundreds of MB. | |
| * | |
| * The body is streamed to completion first and only then handed to the cache. | |
| * Doing it the other way round -- `await cache.put(response.clone())` before | |
| * reading -- makes the tee buffer the entire 312 MB while the cache write drains | |
| * it, so the progress callback fires nothing for minutes and the UI looks hung. | |
| * A failed cache write (quota, private mode) must not fail the load either: the | |
| * bytes are already in hand, and the only cost is re-downloading next time. | |
| */ | |
| async function cachedFetch(url, onProgress) { | |
| const cache = await caches.open(CACHE_NAME).catch(() => null); | |
| const hit = await cache?.match(url).catch(() => null); | |
| if (hit) { | |
| onProgress?.({ url, loaded: 1, total: 1, cached: true }); | |
| return hit.arrayBuffer(); | |
| } | |
| const response = await fetch(url); | |
| if (!response.ok) throw new Error(`${url}: HTTP ${response.status}`); | |
| const total = Number(response.headers.get('content-length')) || 0; | |
| let buffer; | |
| if (!response.body) { | |
| buffer = await response.arrayBuffer(); | |
| } else { | |
| const reader = response.body.getReader(); | |
| const chunks = []; | |
| let loaded = 0; | |
| for (;;) { | |
| const { done, value } = await reader.read(); | |
| if (done) break; | |
| chunks.push(value); | |
| loaded += value.length; | |
| onProgress?.({ url, loaded, total }); | |
| } | |
| const bytes = new Uint8Array(loaded); | |
| let at = 0; | |
| for (const c of chunks) { bytes.set(c, at); at += c.length; } | |
| buffer = bytes.buffer; | |
| } | |
| if (cache) { | |
| try { | |
| await cache.put(url, new Response(buffer.slice(0), { | |
| headers: { 'Content-Type': 'application/octet-stream', 'Content-Length': String(buffer.byteLength) }, | |
| })); | |
| } catch (e) { | |
| console.warn(`could not cache ${url} (will re-download next time):`, e); | |
| } | |
| } | |
| return buffer; | |
| } | |
| class CtcEngine { | |
| constructor({ session, frontend, decoder, config, device }) { | |
| this.session = session; | |
| this.frontend = frontend; | |
| this.decoder = decoder; | |
| this.config = config; | |
| this.device = device; | |
| } | |
| /** | |
| * Create the session and load the front-end assets. | |
| * | |
| * `device: 'webgpu'` falls back to wasm if the adapter or a kernel is | |
| * missing -- a CTC pass is still usable on wasm, just slower. | |
| */ | |
| static async load({ | |
| baseUrl = './onnx-ctc/', | |
| modelFile = 'ctc_conformer_q4f16-b16.onnx', | |
| device = 'webgpu', | |
| onProgress = null, | |
| allowFallback = true, | |
| } = {}) { | |
| const [frontend, decoder] = await Promise.all([ | |
| CtcFrontend.load(baseUrl), | |
| CtcDecoder.load(baseUrl), | |
| ]); | |
| const modelUrl = baseUrl + modelFile; | |
| // The exporter writes weights to `<model>.onnx.data` and records that | |
| // basename inside the graph, so the session's externalData path must | |
| // match it exactly. | |
| const dataName = `${modelFile}.data`; | |
| const [modelBuffer, dataBuffer] = await Promise.all([ | |
| cachedFetch(modelUrl, onProgress), | |
| cachedFetch(baseUrl + dataName, onProgress), | |
| ]); | |
| const options = { | |
| executionProviders: [device], | |
| externalData: [{ path: dataName, data: dataBuffer }], | |
| graphOptimizationLevel: 'all', | |
| }; | |
| let session; | |
| let used = device; | |
| let fallbackError = null; | |
| try { | |
| session = await ort.InferenceSession.create(modelBuffer, options); | |
| } catch (e) { | |
| if (device !== 'webgpu' || !allowFallback) throw e; | |
| console.warn('WebGPU session failed, falling back to wasm:', e); | |
| fallbackError = String(e?.message || e); | |
| used = 'wasm'; | |
| session = await ort.InferenceSession.create(modelBuffer, { | |
| ...options, | |
| executionProviders: ['wasm'], | |
| }); | |
| } | |
| const engine = new CtcEngine({ session, frontend, decoder, config: frontend.config, device: used }); | |
| // A WebGPU kernel can also fail on its *first run* rather than at session | |
| // build (ORT-web reports unsupported MatMulNBits shapes that way), so the | |
| // warmup pass is where a fallback is really decided. | |
| const warmupError = await engine.warmup(); | |
| if (warmupError && used === 'webgpu') { | |
| if (!allowFallback) throw new Error(`WebGPU warmup failed: ${warmupError}`); | |
| console.warn('WebGPU warmup failed, falling back to wasm:', warmupError); | |
| fallbackError = warmupError; | |
| engine.device = 'wasm'; | |
| engine.session = await ort.InferenceSession.create(modelBuffer, { | |
| ...options, | |
| executionProviders: ['wasm'], | |
| }); | |
| await engine.warmup(); | |
| } | |
| engine.fallbackError = fallbackError; | |
| return engine; | |
| } | |
| /** | |
| * Run one throwaway pass so shader compilation doesn't land on the user's | |
| * first spoken window (measured on WebGPU: 1271 ms first call vs 295 ms | |
| * steady state). Every window is padded to the same frame count, so a single | |
| * warmup covers the shapes the streaming loop will use. | |
| */ | |
| async warmup() { | |
| try { | |
| const frames = this.config.pad_multiple; | |
| await this.session.run({ | |
| input_features: new ort.Tensor('float32', new Float32Array(frames * this.config.input_dim), [1, frames, this.config.input_dim]), | |
| padding_mask: new ort.Tensor('float32', new Float32Array(frames), [1, frames]), | |
| }); | |
| return null; | |
| } catch (e) { | |
| return String(e?.message || e); | |
| } | |
| } | |
| /** Encoder output frames covering `frames` of real audio (both subsample blocks floor-halve). */ | |
| static realOutFrames(frames) { | |
| return Math.floor(Math.floor(frames / 2) / 2); | |
| } | |
| /** | |
| * Transcribe a waveform (Float32Array, 16 kHz mono). | |
| * | |
| * `applyAgc` mirrors the sidecar: the log-mel front-end normalizes against | |
| * the clip's global peak, so without it a quiet tail after a loud passage is | |
| * floored away and its words are silently dropped. | |
| */ | |
| async transcribe(wav, { applyAgc = true } = {}) { | |
| const t0 = performance.now(); | |
| const audio = applyAgc ? agc(wav, { sampleRate: this.config.sample_rate }) : wav; | |
| const { data, frames, dim } = this.frontend.compute(audio); | |
| if (frames === 0) return { text: '', words: [], avgLogprob: null, frames: 0, ms: 0 }; | |
| const multiple = this.config.pad_multiple; | |
| const padded = Math.ceil(frames / multiple) * multiple; | |
| const features = new Float32Array(padded * dim); | |
| features.set(data); | |
| const mask = new Float32Array(padded).fill(1); | |
| mask.fill(0, 0, frames); | |
| const outputs = await this.session.run({ | |
| input_features: new ort.Tensor('float32', features, [1, padded, dim]), | |
| padding_mask: new ort.Tensor('float32', mask, [1, padded]), | |
| }); | |
| // The graph emits padded/4 frames; only those covering real audio may | |
| // contribute tokens, so trim before collapsing. | |
| const nOut = CtcEngine.realOutFrames(frames); | |
| const ids = outputs.ids.data.subarray(0, nOut); | |
| const logp = outputs.top_logprob.data.subarray(0, nOut); | |
| const content = ctcCollapse(ids, { | |
| blankId: this.config.blank_id, | |
| numSpecialTokens: this.config.num_special_tokens, | |
| }); | |
| const text = this.decoder.decode(content).trim(); | |
| // Confidence: mean top log-probability over non-blank frames. The CTC | |
| // analogue of a mean per-token logprob -- clean speech sits near -0.03, | |
| // silence/noise confabulation near -1.85 and below. | |
| let sum = 0, n = 0; | |
| for (let i = 0; i < nOut; i++) { | |
| if (ids[i] !== this.config.blank_id) { sum += logp[i]; n++; } | |
| } | |
| return { | |
| text, | |
| words: text.length ? text.split(/\s+/) : [], | |
| avgLogprob: n ? sum / n : null, | |
| frames, | |
| paddedFrames: padded, | |
| ms: performance.now() - t0, | |
| }; | |
| } | |
| } | |
| Object.assign(__M, { CtcEngine }); | |
| })(); | |
| // ==== ctc-streaming.js ================================================== | |
| (() => { | |
| /** | |
| * Live dictation: growing window + LocalAgreement-2, ported from the macOS | |
| * GraniteLiveDictation app (SpeechGate / Reconciler / StreamingCoordinator). | |
| * | |
| * continuous 16 kHz capture -> rolling window + energy SpeechGate | |
| * every `step` while speech is in the window: | |
| * window = audio[segment start .. now] (grows; overlaps the last) | |
| * hypo = words(CTC transcription of window) | |
| * a word COMMITS once two consecutive windows agree on it (longest | |
| * common prefix); everything past that is TENTATIVE (shown grey) | |
| * on a pause: one clean full-window pass commits the segment and re-anchors | |
| * | |
| * The window *grows* rather than sliding because the model returns text with no | |
| * word timestamps -- there is no safe place to trim a fixed window without | |
| * risking a mid-word cut. Anchoring each window at the start of the current | |
| * speech segment makes reconciliation a plain prefix comparison and keeps the | |
| * window bounded (a phrase between pauses). | |
| * | |
| * The tuning constants are the ones the macOS app converged on against real | |
| * far-field recordings, not defaults -- see the comments on each. | |
| */ | |
| const SAMPLE_RATE = 16000; | |
| const INT16_SCALE = 32768; // gate thresholds are in Int16 RMS units (as tuned) | |
| /** | |
| * URL to hand `audioWorklet.addModule()`. | |
| * | |
| * addModule fetches its argument as a *module*, which is the one request type a | |
| * private Space's cross-site iframe fails to authenticate — measured there, the | |
| * same file returns 200 to a plain fetch and intermittently fails as a module | |
| * import. That would break dictation at the moment the user presses record, well | |
| * after everything else had loaded cleanly. So fetch the source (plain fetches | |
| * work) and hand addModule a blob: URL, which needs no network request at all. | |
| * Falls back to the direct URL, which is what works everywhere else. | |
| * | |
| * Resolved against this module rather than the document so it survives being | |
| * served from a subdirectory; build_bundle.py rewrites `document.baseURI` to | |
| * `document.baseURI` for the bundled build, where the two are the same directory. | |
| */ | |
| let cachedWorkletUrl = null; | |
| async function captureWorkletUrl() { | |
| if (cachedWorkletUrl) return cachedWorkletUrl; | |
| const direct = new URL('./ctc-capture-worklet.js', document.baseURI).href; | |
| try { | |
| const r = await fetch(direct); | |
| if (!r.ok) throw new Error(`HTTP ${r.status}`); | |
| const source = await r.text(); | |
| cachedWorkletUrl = URL.createObjectURL(new Blob([source], { type: 'text/javascript' })); | |
| } catch (e) { | |
| console.warn('could not preload the capture worklet; using its URL directly:', e); | |
| cachedWorkletUrl = direct; | |
| } | |
| return cachedWorkletUrl; | |
| } | |
| /** | |
| * Energy speech/silence gate with an adaptive noise floor. | |
| * | |
| * Not a neural VAD -- an RMS gate whose enter/exit thresholds float with the | |
| * measured background level. That matters for its one job: finding a *pause* so | |
| * the growing window can flush. With fixed thresholds, a room whose ambient RMS | |
| * sits above the silence cutoff never produces a silence frame, the endpoint | |
| * timer never fires, and every window runs to the maxWindow cap and cuts | |
| * mid-sentence. | |
| */ | |
| class SpeechGate { | |
| constructor(config = {}) { | |
| this.config = { | |
| endpointSilenceMs: 600, // contiguous below-exit audio that ends a segment | |
| frameMs: 30, | |
| // Biased toward inclusion: this CTC model doesn't confabulate on | |
| // silence the way an autoregressive decoder does (it decodes to empty | |
| // / low confidence, and the confidence gate backstops it), so | |
| // admitting soft audio costs nothing -- while excluding it dropped | |
| // real words (soft onsets never triggered, soft tails got endpointed | |
| // mid-phrase). | |
| onRatio: 1.5, | |
| minEnter: 80, | |
| // Exit stays near the original: dropping it toward ambient stops | |
| // pause detection altogether and every segment runs to maxWindow, | |
| // which wrecks paragraphing. | |
| offRatio: 1.4, | |
| minExit: 60, | |
| onFrames: 2, // debounce transient clicks | |
| // Seeded from the *mean* RMS over the startup window, not the min: the | |
| // min tracks the quietest valley of fluctuating noise and seeds the | |
| // floor far below the ambient body, which then lets noise read as | |
| // speech. | |
| calibrationMs: 360, | |
| floorAdapt: 0.05, // per-frame EMA; at 30 ms frames ~0.6 s to settle | |
| initialNoiseFloor: 60, | |
| maxNoiseFloor: 4000, | |
| ...config, | |
| }; | |
| this.frameLen = Math.max(1, (SAMPLE_RATE * this.config.frameMs) / 1000); | |
| this.calibrationFrames = Math.floor(this.config.calibrationMs / this.config.frameMs); | |
| this.onEvent = null; | |
| this.isSpeaking = false; | |
| this.noiseFloor = this.config.initialNoiseFloor; | |
| this.silenceMs = 0; | |
| this.onCount = 0; | |
| this.pending = []; | |
| this.pendingLen = 0; | |
| this.calibLeft = this.calibrationFrames; | |
| this.calibSum = 0; | |
| this.calibCount = 0; | |
| } | |
| get enterThreshold() { return Math.max(this.config.minEnter, this.noiseFloor * this.config.onRatio); } | |
| get exitThreshold() { return Math.max(this.config.minExit, this.noiseFloor * this.config.offRatio); } | |
| /** | |
| * Clear per-utterance state. The noise floor and calibration are deliberately | |
| * preserved: they describe the *room*, not the utterance, and this is called | |
| * on every commit (including maxWindow cuts taken mid-speech), where | |
| * re-seeding would calibrate the floor from speech energy and blind the gate. | |
| */ | |
| reset() { | |
| this.isSpeaking = false; | |
| this.silenceMs = 0; | |
| this.onCount = 0; | |
| this.pending = []; | |
| this.pendingLen = 0; | |
| } | |
| /** Re-arm calibration -- for a fresh capture session / new environment. */ | |
| recalibrate() { | |
| this.noiseFloor = this.config.initialNoiseFloor; | |
| this.calibLeft = this.calibrationFrames; | |
| this.calibSum = 0; | |
| this.calibCount = 0; | |
| } | |
| /** Feed a chunk of float samples (any length); re-framed into fixed blocks. */ | |
| feed(samples) { | |
| if (!samples.length) return; | |
| this.pending.push(samples); | |
| this.pendingLen += samples.length; | |
| while (this.pendingLen >= this.frameLen) { | |
| const frame = this.takeFrame(); | |
| // process() emits events synchronously and a handler may call reset() | |
| // re-entrantly (which drops `pending`), so the frame is removed first. | |
| this.process(frame); | |
| } | |
| } | |
| takeFrame() { | |
| const frame = new Float32Array(this.frameLen); | |
| let at = 0; | |
| while (at < this.frameLen) { | |
| const head = this.pending[0]; | |
| const take = Math.min(head.length, this.frameLen - at); | |
| frame.set(take === head.length ? head : head.subarray(0, take), at); | |
| at += take; | |
| if (take === head.length) this.pending.shift(); | |
| else this.pending[0] = head.subarray(take); | |
| } | |
| this.pendingLen -= this.frameLen; | |
| return frame; | |
| } | |
| process(frame) { | |
| const rms = rmsInt16(frame); | |
| if (this.calibLeft > 0) { | |
| this.calibLeft--; | |
| this.calibSum += rms; | |
| this.calibCount++; | |
| this.noiseFloor = Math.min(Math.max(this.calibSum / this.calibCount, 1), this.config.maxNoiseFloor); | |
| return; | |
| } | |
| this.adaptNoise(rms); | |
| if (this.isSpeaking) { | |
| if (rms < this.exitThreshold) { | |
| this.silenceMs += this.config.frameMs; | |
| if (this.silenceMs >= this.config.endpointSilenceMs) { | |
| this.isSpeaking = false; | |
| this.silenceMs = 0; | |
| this.onCount = 0; | |
| this.onEvent?.('speechEnded'); | |
| } | |
| } else { | |
| this.silenceMs = 0; // speech resumed; reset the endpoint timer | |
| } | |
| } else if (rms >= this.enterThreshold) { | |
| this.onCount++; | |
| if (this.onCount >= this.config.onFrames) { | |
| this.isSpeaking = true; | |
| this.silenceMs = 0; | |
| this.onCount = 0; | |
| this.onEvent?.('speechStarted'); | |
| } | |
| } else { | |
| this.onCount = 0; | |
| } | |
| } | |
| /** | |
| * Learn the floor from ambient frames only, tracking the *body* (mean) of the | |
| * ambient level. Updating solely when not in a segment AND below the enter | |
| * threshold keeps sustained speech (or a speech attack) from dragging the | |
| * floor up to speech level. | |
| */ | |
| adaptNoise(rms) { | |
| if (this.isSpeaking || rms >= this.enterThreshold) return; | |
| const a = this.config.floorAdapt; | |
| this.noiseFloor = Math.min(Math.max((1 - a) * this.noiseFloor + a * rms, 1), this.config.maxNoiseFloor); | |
| } | |
| } | |
| function rmsInt16(samples) { | |
| if (!samples.length) return 0; | |
| let acc = 0; | |
| for (let i = 0; i < samples.length; i++) acc += samples[i] * samples[i]; | |
| return Math.sqrt(acc / samples.length) * INT16_SCALE; | |
| } | |
| /** | |
| * LocalAgreement-2 reconciliation. | |
| * | |
| * Each step transcribes the growing window and yields a fresh hypothesis. Because | |
| * the window overlaps the previous one, the two should agree on the stable part | |
| * and differ only at the unfinished tail. A word is committed once two | |
| * *consecutive* hypotheses agree on it (their longest common prefix, compared | |
| * case-insensitively and ignoring edge punctuation); everything past that is | |
| * tentative and may still change. | |
| */ | |
| class Reconciler { | |
| constructor() { | |
| this.confirmed = []; | |
| this.prevHypo = []; | |
| } | |
| update(hypothesis) { | |
| const agreed = commonPrefixCount(this.prevHypo, hypothesis); | |
| if (agreed > this.confirmed.length) { | |
| this.confirmed.push(...hypothesis.slice(this.confirmed.length, agreed)); | |
| } | |
| this.prevHypo = hypothesis; | |
| const tentative = hypothesis.length > this.confirmed.length | |
| ? hypothesis.slice(this.confirmed.length) | |
| : []; | |
| return { confirmed: this.confirmed, tentative }; | |
| } | |
| /** End of segment: a clean full-window hypothesis is authoritative. */ | |
| finalize(words) { | |
| this.confirmed = words; | |
| this.prevHypo = words; | |
| return this.confirmed; | |
| } | |
| reset() { | |
| this.confirmed = []; | |
| this.prevHypo = []; | |
| } | |
| } | |
| /** Lowercase and strip edge punctuation so "Cat," and "cat" match. */ | |
| function normalizeWord(word) { | |
| return word.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, '').toLowerCase(); | |
| } | |
| function commonPrefixCount(a, b) { | |
| const n = Math.min(a.length, b.length); | |
| let i = 0; | |
| while (i < n && normalizeWord(a[i]) === normalizeWord(b[i])) i++; | |
| return i; | |
| } | |
| /** | |
| * Stock phrases the model confidently emits on ambient/silence -- the classic ASR | |
| * hallucination. Confidence can't catch these (they score *higher* than genuine | |
| * soft speech), so they're matched against the whole commit and dropped only when | |
| * word density is low: a silence-padded hallucination is long and slow, a real | |
| * crisp "thank you" is short and fast. | |
| */ | |
| const CANNED_HALLUCINATIONS = new Set([ | |
| 'thank you', 'thank you very much', 'thanks for watching', | |
| 'thank you for watching', 'thanks for watching everyone', | |
| ]); | |
| class StreamingCoordinator { | |
| /** | |
| * @param {object} opts | |
| * @param {import('./ctc-engine.js').CtcEngine} opts.engine | |
| * @param {(state: {committed: string, tentative: string}) => void} opts.onTranscript | |
| * @param {(level: number) => void} [opts.onLevel] 0..1 input level | |
| * @param {(info: object) => void} [opts.onStep] per-window diagnostics | |
| * @param {(text: string) => Promise<string>} [opts.punctuate] committed-text post-pass | |
| */ | |
| constructor({ engine, onTranscript, onLevel = null, onStep = null, punctuate = null, tuning = {} }) { | |
| this.engine = engine; | |
| this.onTranscript = onTranscript; | |
| this.onLevel = onLevel; | |
| this.onStep = onStep; | |
| this.punctuate = punctuate; | |
| this.tuning = { | |
| stepMs: 400, // re-transcribe cadence. A window costs one padded | |
| // forward pass (>= 10.24 s of frames, see | |
| // ctc-engine), so this is coarser than the macOS | |
| // app's 100 ms; an in-flight tick is dropped, so | |
| // it self-limits either way. | |
| minWindowSec: 0.4, // don't transcribe windows shorter than this | |
| maxWindowSec: 10.0, // force-commit a window that grows this long with no pause | |
| newlineGapMs: 1200, // a pause >= this starts a new paragraph | |
| hardPauseMs: 1500, | |
| prerollSec: 0.5, // audio kept before speech onset, so the first word isn't clipped | |
| // Calibrated on real far-field logs: clean speech ~ -0.03, but genuine | |
| // soft utterances scored -0.98..-1.34 and were wrongly rejected at a | |
| // -0.6 gate, dropping whole clips; silence confabulation sits at | |
| // -1.85 and below, so -1.5 lands in the gap. | |
| minCommitAvgLogprob: -1.5, | |
| cannedHallucinationMaxWps: 2.0, | |
| ...tuning, | |
| }; | |
| this.gate = new SpeechGate(); | |
| this.gate.onEvent = (e) => this.handleGate(e); | |
| this.reconciler = new Reconciler(); | |
| this.window = []; // Float32Array chunks since the last commit | |
| this.windowLen = 0; | |
| this.windowHasSpeech = false; | |
| this.transcribing = false; | |
| this.windowGen = 0; | |
| this.paragraphs = []; // committed text, one entry per paragraph | |
| this.lastVoiceAt = 0; | |
| this.segmentGapMs = Infinity; | |
| this.active = false; | |
| this.audioContext = null; | |
| this.mediaStream = null; | |
| this.stepTimer = null; | |
| // One ONNX session can't have two concurrent run() calls, and a commit | |
| // fired by the gate can land while a step transcription is still in | |
| // flight, so every engine call goes through this chain. | |
| this.engineLock = Promise.resolve(); | |
| // Commits are started from the gate callback (inside onSamples), which | |
| // can't await them; stop() drains these so the last words aren't lost. | |
| this.pending = new Set(); | |
| } | |
| /** Serialize engine access; returns the callback's result. */ | |
| async withEngine(fn) { | |
| const prior = this.engineLock; | |
| let release; | |
| this.engineLock = new Promise((r) => { release = r; }); | |
| try { | |
| await prior; | |
| return await fn(); | |
| } finally { | |
| release(); | |
| } | |
| } | |
| /** Track a fire-and-forget async task so stop() can wait for it. */ | |
| track(promise) { | |
| this.pending.add(promise); | |
| promise.catch(() => {}).finally(() => this.pending.delete(promise)); | |
| return promise; | |
| } | |
| /** Resolve once no commit or step is outstanding. */ | |
| async idle() { | |
| while (this.pending.size) await Promise.allSettled([...this.pending]); | |
| await this.engineLock; | |
| } | |
| async start() { | |
| if (this.active) return; | |
| this.mediaStream = await navigator.mediaDevices.getUserMedia({ | |
| audio: { | |
| channelCount: 1, | |
| echoCancellation: true, | |
| noiseSuppression: false, // the gate wants the true ambient level | |
| autoGainControl: false, // the engine's own AGC handles levels | |
| }, | |
| }); | |
| this.audioContext = new AudioContext({ sampleRate: SAMPLE_RATE }); | |
| await this.audioContext.audioWorklet.addModule(await captureWorkletUrl()); | |
| const source = this.audioContext.createMediaStreamSource(this.mediaStream); | |
| const node = new AudioWorkletNode(this.audioContext, 'ctc-capture'); | |
| node.port.onmessage = (e) => this.onSamples(e.data); | |
| source.connect(node); | |
| // Keep the graph pulling without routing mic audio to the speakers. | |
| const sink = this.audioContext.createGain(); | |
| sink.gain.value = 0; | |
| node.connect(sink).connect(this.audioContext.destination); | |
| this.resetWindow(); | |
| this.gate.recalibrate(); | |
| this.paragraphs = []; | |
| this.lastVoiceAt = 0; | |
| this.segmentGapMs = Infinity; | |
| this.active = true; | |
| this.publish([], []); | |
| this.stepTimer = setInterval(() => this.tick(), this.tuning.stepMs); | |
| } | |
| async stop() { | |
| if (!this.active) return; | |
| this.active = false; | |
| clearInterval(this.stepTimer); | |
| this.stepTimer = null; | |
| this.mediaStream?.getTracks().forEach((t) => t.stop()); | |
| await this.audioContext?.close(); | |
| this.audioContext = null; | |
| this.onLevel?.(0); | |
| // Let any commit the gate started finish before deciding whether the | |
| // window still holds uncommitted speech. | |
| await this.idle(); | |
| if (this.windowHasSpeech) await this.commit('stop'); | |
| this.resetWindow(); | |
| this.publish([], []); | |
| } | |
| // ---------------------------------------------------------------- capture | |
| onSamples(chunk) { | |
| this.window.push(chunk); | |
| this.windowLen += chunk.length; | |
| // Before speech is detected, keep only a short pre-roll. The window is | |
| // otherwise cleared only on commit, so a long ambient stretch would | |
| // accumulate unbounded and the first real utterance would commit an | |
| // enormous mostly-noise window. | |
| if (!this.windowHasSpeech) { | |
| const maxPreroll = Math.floor(SAMPLE_RATE * this.tuning.prerollSec); | |
| while (this.windowLen > maxPreroll && this.window.length > 1) { | |
| this.windowLen -= this.window.shift().length; | |
| } | |
| } | |
| const now = performance.now(); | |
| this.onLevel?.(meterLevel(rmsInt16(chunk))); | |
| this.gate.feed(chunk); // may fire speechEnded -> commit | |
| // Drive the window's speech state from the gate's debounced, | |
| // calibration-aware decision rather than a parallel per-chunk RMS test: | |
| // such a test has no debounce and runs during calibration, so one noise | |
| // chunk latches the window open and the model transcribes ambient. | |
| if (this.gate.isSpeaking) { | |
| if (!this.windowHasSpeech) { | |
| this.segmentGapMs = this.lastVoiceAt ? now - this.lastVoiceAt : Infinity; | |
| this.windowHasSpeech = true; | |
| } | |
| this.lastVoiceAt = now; | |
| } | |
| } | |
| handleGate(event) { | |
| if (event !== 'speechEnded') return; // capture is continuous; onset needs nothing | |
| // Called synchronously from onSamples, so the commit can only be tracked, | |
| // not awaited (see idle()). | |
| if (this.windowHasSpeech) this.track(this.commit('pause')); | |
| else this.resetWindow(); // pure-silence window: drop it | |
| } | |
| // -------------------------------------------------------------- step loop | |
| async tick() { | |
| if (!this.active || !this.windowHasSpeech || this.transcribing) return; | |
| if (this.windowLen / SAMPLE_RATE >= this.tuning.maxWindowSec) { | |
| await this.commit('maxWindow'); | |
| return; | |
| } | |
| const speech = this.trimmedSpeech(); | |
| if (speech.length / SAMPLE_RATE < this.tuning.minWindowSec) return; | |
| const gen = this.windowGen; | |
| this.transcribing = true; | |
| try { | |
| const result = await this.withEngine(() => this.engine.transcribe(speech)); | |
| if (gen !== this.windowGen) return; // window committed meanwhile; discard | |
| const { confirmed, tentative } = this.reconciler.update(result.words); | |
| this.publish(confirmed, tentative); | |
| this.onStep?.({ kind: 'step', windowSec: speech.length / SAMPLE_RATE, ...result }); | |
| } catch (e) { | |
| console.error('step transcription failed:', e); | |
| } finally { | |
| this.transcribing = false; | |
| } | |
| } | |
| /** | |
| * Commit the current window: snapshot it, immediately clear and re-anchor (so | |
| * capture continues uninterrupted and any in-flight partial is discarded), | |
| * then transcribe the snapshot once cleanly and append it. | |
| */ | |
| async commit(reason) { | |
| const speech = this.trimmedSpeech(); | |
| const windowSec = speech.length / SAMPLE_RATE; | |
| if (windowSec < this.tuning.minWindowSec) { | |
| this.resetWindow(); | |
| return; | |
| } | |
| const gapMs = this.segmentGapMs; | |
| this.resetWindow(); | |
| this.publish([], []); | |
| let result; | |
| try { | |
| result = await this.withEngine(() => this.engine.transcribe(speech)); | |
| } catch (e) { | |
| console.error('commit transcription failed:', e); | |
| return; | |
| } | |
| const words = result.words; | |
| const wps = windowSec > 0 ? words.length / windowSec : 0; | |
| // Confidence gate: a low mean logprob means the model wasn't sure of its | |
| // own tokens -- the signature of confabulation on noise/silence. A null | |
| // confidence fails open. | |
| if (result.avgLogprob !== null && result.avgLogprob < this.tuning.minCommitAvgLogprob) { | |
| this.onStep?.({ kind: 'reject', why: 'confidence', reason, windowSec, wps, ...result }); | |
| this.publish([], []); | |
| return; | |
| } | |
| const normalized = words.join(' ').toLowerCase(); | |
| if (CANNED_HALLUCINATIONS.has(normalized) && wps < this.tuning.cannedHallucinationMaxWps) { | |
| this.onStep?.({ kind: 'reject', why: 'canned', reason, windowSec, wps, ...result }); | |
| this.publish([], []); | |
| return; | |
| } | |
| if (!words.length) { | |
| this.publish([], []); | |
| return; | |
| } | |
| await this.appendSegment(words.join(' '), gapMs, reason); | |
| this.onStep?.({ kind: 'commit', reason, windowSec, wps, gapMs, ...result }); | |
| this.publish([], []); | |
| } | |
| /** | |
| * Append a committed segment, opening a new paragraph after a real pause. | |
| * | |
| * Unlike the macOS app, sentence punctuation and capitalization are NOT | |
| * synthesized here -- the model emits raw lowercase text and the punctuator | |
| * restores both. A maxWindow cut is mid-utterance, so it extends the current | |
| * paragraph and is re-punctuated together with what follows. | |
| */ | |
| async appendSegment(text, gapMs, reason) { | |
| const startsParagraph = this.paragraphs.length === 0 | |
| || gapMs >= this.tuning.hardPauseMs | |
| || (gapMs >= this.tuning.newlineGapMs && reason !== 'maxWindow'); | |
| if (startsParagraph) this.paragraphs.push(text); | |
| else this.paragraphs[this.paragraphs.length - 1] += ` ${text}`; | |
| if (this.punctuate) { | |
| const i = this.paragraphs.length - 1; | |
| const raw = this.paragraphs[i]; | |
| try { | |
| this.paragraphs[i] = await this.punctuate(raw); | |
| } catch (e) { | |
| console.warn('punctuation failed; keeping raw text:', e); | |
| } | |
| } | |
| } | |
| // ---------------------------------------------------------------- helpers | |
| /** | |
| * The speech-bearing span of the window, with leading/trailing silence | |
| * removed. Trailing trim matters: the front-end normalizes against the clip's | |
| * global peak, so an untrimmed quiet tail gets amplified and decoded into | |
| * spurious words. | |
| */ | |
| trimmedSpeech() { | |
| const audio = this.flattenWindow(); | |
| const frame = 480; // 30 ms | |
| const floor = this.gate.exitThreshold; | |
| let first = -1, last = -1; | |
| for (let i = 0; i < audio.length; i += frame) { | |
| const end = Math.min(i + frame, audio.length); | |
| if (rmsInt16(audio.subarray(i, end)) >= floor) { | |
| if (first < 0) first = i; | |
| last = end; | |
| } | |
| } | |
| if (first < 0) return new Float32Array(0); | |
| const margin = frame * 3; // ~90 ms of context each side | |
| return audio.subarray(Math.max(0, first - margin), Math.min(audio.length, last + margin)); | |
| } | |
| flattenWindow() { | |
| if (this.window.length === 1) return this.window[0]; | |
| const out = new Float32Array(this.windowLen); | |
| let at = 0; | |
| for (const c of this.window) { out.set(c, at); at += c.length; } | |
| this.window = [out]; // collapse so repeated ticks don't re-copy | |
| return out; | |
| } | |
| resetWindow() { | |
| this.window = []; | |
| this.windowLen = 0; | |
| this.windowHasSpeech = false; | |
| this.reconciler.reset(); | |
| this.gate.reset(); | |
| this.windowGen++; | |
| } | |
| publish(confirmed, tentative) { | |
| const live = confirmed.join(' '); | |
| const committed = live | |
| ? [...this.paragraphs, live].join('\n\n') | |
| : this.paragraphs.join('\n\n'); | |
| this.onTranscript({ committed, tentative: tentative.join(' ') }); | |
| } | |
| /** Plain text of everything committed so far. */ | |
| get transcript() { | |
| return this.paragraphs.join('\n\n'); | |
| } | |
| } | |
| /** Int16 RMS -> 0..1 meter level on a dBFS scale (-60 dBFS -> 0, 0 dBFS -> 1). */ | |
| function meterLevel(rms) { | |
| if (rms <= 0) return 0; | |
| const dbfs = 20 * Math.log10(rms / 32767); | |
| return Math.max(0, Math.min(1, (dbfs + 60) / 60)); | |
| } | |
| Object.assign(__M, { SpeechGate, Reconciler, StreamingCoordinator }); | |
| })(); | |
| // ==== app.js ============================================================ | |
| (() => { | |
| const { CtcEngine, SpeechGate, StreamingCoordinator } = __M; | |
| /** | |
| * Granite Live Dictation (WebGPU) | |
| * | |
| * Real-time in-browser dictation with ibm-granite/granite-speech-4.2-470m-turboctc, | |
| * a 473M-parameter FastConformer CTC model. Unlike the previous | |
| * granite-speech-4.1-2b build, this is not a transformers.js pipeline: | |
| * `ctc_conformer` is a custom architecture, so the ONNX session is driven directly | |
| * (see ctc-engine.js) -- one forward pass and a greedy collapse per window, no | |
| * autoregressive decode. That is what makes it fast enough to re-transcribe a | |
| * growing window several times a second. | |
| * | |
| * The model emits raw lowercase, unpunctuated English. Punctuation and | |
| * capitalization are restored by the separate punctuator model (punctuator.js), | |
| * which is loaded on demand because it is a 209 MB download of its own. | |
| * | |
| * `ort` is the global from the onnxruntime-web script tag in index.html, shared | |
| * with punctuator.js so both sessions use one runtime. | |
| */ | |
| const MODEL_DIR = './onnx-ctc/'; | |
| // 4-bit weights, block 32, on an fp16 graph (312 MB from a 1.9 GB fp32 export). | |
| // This is the only shape WebGPU can actually run: ORT-web's WebGPU EP enforces | |
| // `nbits == 4 || nbits == 2` on MatMulNBits, so an 8-bit build silently falls | |
| // back to wasm (1320 ms per window against 305 ms) and buys nothing -- on 8 | |
| // minutes of LibriSpeech the 4-bit weights cost 0.09% WER against the fp32 | |
| // export, and what residue remains in the browser comes from fp16 activations, | |
| // which more weight bits would not fix. | |
| const MODEL_FILE = 'ctc_conformer_q4f16.onnx'; | |
| const SAMPLE_RATE = 16000; | |
| // One padded forward pass covers 10.24 s of frames, so file segments are capped | |
| // near that: longer segments cost proportionally more and buy nothing. | |
| const MAX_SEGMENT_SEC = 20; | |
| let engine = null; | |
| let coordinator = null; | |
| let punctuatorReady = false; | |
| let punctuatorLoad = null; | |
| let isLoading = false; | |
| let currentAudioData = null; | |
| const statusDot = document.getElementById('statusDot'); | |
| const statusText = document.getElementById('statusText'); | |
| const dictateBtn = document.getElementById('dictateBtn'); | |
| const levelFill = document.getElementById('levelFill'); | |
| const audioFile = document.getElementById('audioFile'); | |
| const fileTile = document.querySelector('.file-label'); | |
| const inputCard = document.querySelector('.input-card'); | |
| const punctCheckbox = document.getElementById('punctCheckbox'); | |
| const transcriptCard = document.getElementById('transcriptCard'); | |
| const committedEl = document.getElementById('committedText'); | |
| const tentativeEl = document.getElementById('tentativeText'); | |
| const copyBtn = document.getElementById('copyBtn'); | |
| const downloadBtn = document.getElementById('downloadBtn'); | |
| const clearBtn = document.getElementById('clearBtn'); | |
| const progressFill = document.getElementById('progressFill'); | |
| const progressSection = document.getElementById('progressSection'); | |
| const progressText = document.getElementById('progressText'); | |
| const gpuInfo = document.getElementById('gpuInfo'); | |
| const statsEl = document.getElementById('stats'); | |
| function setStatus(state, message) { | |
| statusDot.className = `status-dot ${state}`; | |
| statusText.textContent = message; | |
| } | |
| function showProgress(show, text = '') { | |
| progressSection.style.display = show ? 'block' : 'none'; | |
| if (text) progressText.textContent = text; | |
| } | |
| // ---------------------------------------------------------------- model load | |
| /** | |
| * Hand ORT its wasm assets directly, fetched with a plain `fetch()`. | |
| * | |
| * ORT locates its backend by dynamically `import()`ing | |
| * ort-wasm-simd-threaded.jsep.mjs. In a private Space's cross-site iframe that | |
| * module fetch is not authenticated -- the same failure that stopped app.js's own | |
| * imports and forced the bundle -- and it surfaces from session creation as | |
| * "no available backend found. ERR: [wasm] previous call to 'initWasm()' failed". | |
| * | |
| * Plain fetches *are* authenticated in that context, so the glue is fetched here | |
| * and handed over as a blob: URL, which `import()` reads with no network request | |
| * at all, and the wasm bytes go across as `wasmBinary` so nothing else is looked | |
| * up either. Falls back to letting ORT fetch them itself, which is what works | |
| * everywhere other than that iframe. | |
| */ | |
| async function configureOrtWasm(baseUrl = './ort/') { | |
| const dir = new URL(baseUrl, document.baseURI).href; | |
| const get = async (name, as) => { | |
| const r = await fetch(dir + name); | |
| if (!r.ok) throw new Error(`${name}: HTTP ${r.status}`); | |
| return as === 'text' ? r.text() : r.arrayBuffer(); | |
| }; | |
| try { | |
| const [glue, binary] = await Promise.all([ | |
| get('ort-wasm-simd-threaded.jsep.mjs', 'text'), | |
| get('ort-wasm-simd-threaded.jsep.wasm'), | |
| ]); | |
| ort.env.wasm.wasmBinary = binary; | |
| ort.env.wasm.wasmPaths = { | |
| mjs: URL.createObjectURL(new Blob([glue], { type: 'text/javascript' })), | |
| }; | |
| } catch (e) { | |
| console.warn('could not preload the ORT wasm assets; letting ORT fetch them:', e); | |
| // Absolute, not './ort/': ORT resolves a relative wasmPaths against the | |
| // location of ort.all.min.js (already in ort/), which would look for | |
| // ort/ort/ort-wasm-simd-threaded.jsep.mjs and find no backend. | |
| ort.env.wasm.wasmPaths = dir; | |
| } | |
| } | |
| async function initEngine() { | |
| if (engine || isLoading) return engine; | |
| isLoading = true; | |
| try { | |
| // The runtime is a blocking classic script in index.html, so reaching here | |
| // without it means it was blocked (Edge Tracking Prevention, an extension, | |
| // a CSP). Say that plainly instead of throwing a bare ReferenceError. | |
| if (typeof ort === 'undefined') { | |
| throw new Error('the ONNX runtime failed to load (ort/ort.all.min.js) — ' | |
| + 'check for a blocked request in the network tab'); | |
| } | |
| if (!navigator.gpu) { | |
| gpuInfo.textContent = 'WebGPU not available — running on WASM (slower)'; | |
| } | |
| await configureOrtWasm(); | |
| // Threaded wasm needs cross-origin isolation (COOP *and* COEP). Hugging | |
| // Face Spaces send COOP but not COEP, so asking for threads there only | |
| // produces a console warning and a wasted init before ORT falls back. | |
| ort.env.wasm.numThreads = self.crossOriginIsolated ? (navigator.hardwareConcurrency || 4) : 1; | |
| setStatus('loading', 'Downloading model...'); | |
| showProgress(true, 'Downloading model...'); | |
| const progress = {}; | |
| engine = await CtcEngine.load({ | |
| baseUrl: MODEL_DIR, | |
| modelFile: MODEL_FILE, | |
| device: navigator.gpu ? 'webgpu' : 'wasm', | |
| onProgress: ({ url, loaded, total, cached }) => { | |
| progress[url] = { loaded, total, cached }; | |
| let l = 0, t = 0, allCached = true; | |
| for (const p of Object.values(progress)) { | |
| l += p.loaded; | |
| t += p.total; | |
| if (!p.cached) allCached = false; | |
| } | |
| if (allCached) { | |
| showProgress(true, 'Loading model from cache...'); | |
| return; | |
| } | |
| const pct = t > 0 ? (l / t) * 100 : 0; | |
| progressFill.style.width = `${pct}%`; | |
| showProgress(true, `Downloading model... ${(l / 1e6).toFixed(0)} / ${(t / 1e6).toFixed(0)} MB`); | |
| }, | |
| }); | |
| progressFill.style.width = '0%'; | |
| showProgress(false); | |
| // A wasm fallback is ~4x slower per window (1325 ms vs 305 ms for a 10 s | |
| // window), which is the difference between live and not, so say so | |
| // rather than letting it look like the model is just slow. | |
| gpuInfo.textContent = engine.device === 'webgpu' | |
| ? 'Backend: WebGPU' | |
| : 'Backend: WASM — WebGPU unavailable, so dictation will lag well behind speech'; | |
| setStatus('ready', 'Ready — start dictation or upload audio'); | |
| dictateBtn.disabled = false; | |
| audioFile.disabled = false; | |
| if (punctCheckbox.checked) startPunctuatorLoad(); | |
| return engine; | |
| } catch (e) { | |
| console.error('Model loading failed:', e); | |
| setStatus('error', `Error: ${e.message || e}`); | |
| showProgress(false); | |
| throw e; | |
| } finally { | |
| isLoading = false; | |
| } | |
| } | |
| /** | |
| * Load the punctuator (209 MB of its own). | |
| * | |
| * Kicked off in the background as soon as the CTC engine is ready, so it is | |
| * usually done before the user presses record -- loading it on the click would | |
| * put a ~30 s download between "start dictation" and hearing anything. Calls | |
| * share one promise, and only a caller that actually has to wait shows progress. | |
| */ | |
| function startPunctuatorLoad() { | |
| if (!punctuatorLoad && typeof window.loadPunctuator === 'function') { | |
| punctuatorLoad = window.loadPunctuator() | |
| .then(() => { punctuatorReady = true; }) | |
| .catch((e) => { | |
| console.warn('punctuator failed to load; transcript stays lowercase:', e); | |
| punctuatorReady = false; | |
| }); | |
| } | |
| return punctuatorLoad; | |
| } | |
| async function ensurePunctuator() { | |
| if (punctuatorReady || !punctCheckbox.checked) return punctuatorReady; | |
| const load = startPunctuatorLoad(); | |
| if (!load) return false; | |
| const stillLoading = !punctuatorReady; | |
| if (stillLoading) showProgress(true, 'Loading punctuation model...'); | |
| await load; | |
| if (stillLoading) showProgress(false); | |
| return punctuatorReady; | |
| } | |
| /** Punctuate committed text, or pass it through if the model is off/unavailable. */ | |
| async function punctuate(text) { | |
| if (!punctCheckbox.checked || !punctuatorReady) return text; | |
| return window.applyPunctuation(text); | |
| } | |
| // ------------------------------------------------------------ live dictation | |
| function renderTranscript({ committed, tentative }) { | |
| transcriptCard.style.display = 'block'; | |
| committedEl.textContent = committed; | |
| tentativeEl.textContent = tentative ? (committed ? ` ${tentative}` : tentative) : ''; | |
| const out = committedEl.parentElement; | |
| out.scrollTop = out.scrollHeight; | |
| const hasText = Boolean(committed || tentative); | |
| copyBtn.disabled = !hasText; | |
| downloadBtn.disabled = !hasText; | |
| } | |
| async function toggleDictation() { | |
| if (coordinator) { | |
| dictateBtn.disabled = true; | |
| setStatus('processing', 'Finishing…'); | |
| try { | |
| await coordinator.stop(); | |
| } finally { | |
| coordinator = null; | |
| dictateBtn.classList.remove('recording'); | |
| dictateBtn.querySelector('span').textContent = 'Start dictation'; | |
| dictateBtn.disabled = false; | |
| levelFill.style.width = '0%'; | |
| setStatus('ready', 'Ready — start dictation or upload audio'); | |
| } | |
| return; | |
| } | |
| await initEngine(); | |
| await ensurePunctuator(); | |
| coordinator = new StreamingCoordinator({ | |
| engine, | |
| onTranscript: renderTranscript, | |
| onLevel: (level) => { levelFill.style.width = `${(level * 100).toFixed(0)}%`; }, | |
| onStep: (info) => { | |
| if (info.kind === 'step') { | |
| statsEl.textContent = `${info.windowSec.toFixed(1)}s window · ${info.ms.toFixed(0)} ms · conf ${info.avgLogprob?.toFixed(2) ?? 'n/a'}`; | |
| } else if (info.kind === 'reject') { | |
| console.log(`[reject:${info.why}] "${info.text}" conf=${info.avgLogprob?.toFixed(2)} wps=${info.wps.toFixed(2)}`); | |
| } else { | |
| console.log(`[commit:${info.reason}] ${info.windowSec.toFixed(1)}s ${info.ms.toFixed(0)}ms conf=${info.avgLogprob?.toFixed(2)} "${info.text}"`); | |
| } | |
| }, | |
| punctuate, | |
| }); | |
| try { | |
| await coordinator.start(); | |
| dictateBtn.classList.add('recording'); | |
| dictateBtn.querySelector('span').textContent = 'Stop dictation'; | |
| setStatus('recording', 'Listening — speak naturally, pause between sentences'); | |
| } catch (e) { | |
| console.error('dictation failed to start:', e); | |
| coordinator = null; | |
| setStatus('error', e.name === 'NotAllowedError' ? 'Microphone access denied' : `Error: ${e.message || e}`); | |
| } | |
| } | |
| // --------------------------------------------------------- file transcription | |
| /** | |
| * Split a file into speech segments with the same energy gate the live path uses. | |
| * | |
| * This replaces the old Silero VAD download: the gate is already here, it is what | |
| * the streaming path is tuned against, and long audio has to be split anyway -- | |
| * a single pass over a 5-minute file would allocate a ~250 MB logits tensor | |
| * inside the graph. | |
| */ | |
| function segmentAudio(audio) { | |
| const gate = new SpeechGate(); | |
| const frame = 480; | |
| const segments = []; | |
| let start = null; | |
| let consumed = 0; | |
| gate.onEvent = (event) => { | |
| const t = consumed / SAMPLE_RATE; | |
| if (event === 'speechStarted') start = t; | |
| else if (start !== null) { | |
| segments.push({ start, end: t }); | |
| start = null; | |
| } | |
| }; | |
| for (let i = 0; i < audio.length; i += frame) { | |
| const chunk = audio.subarray(i, Math.min(i + frame, audio.length)); | |
| consumed = i + chunk.length; | |
| gate.feed(chunk); | |
| } | |
| if (start !== null) segments.push({ start, end: audio.length / SAMPLE_RATE }); | |
| if (!segments.length) return [{ start: 0, end: audio.length / SAMPLE_RATE }]; | |
| // Pad each segment slightly (the gate's onset debounce trims the first | |
| // phoneme) and cap the length. | |
| const out = []; | |
| for (const seg of segments) { | |
| let from = Math.max(0, seg.start - 0.2); | |
| const to = Math.min(audio.length / SAMPLE_RATE, seg.end + 0.2); | |
| while (to - from > MAX_SEGMENT_SEC) { | |
| out.push({ start: from, end: from + MAX_SEGMENT_SEC }); | |
| from += MAX_SEGMENT_SEC; | |
| } | |
| out.push({ start: from, end: to }); | |
| } | |
| return out; | |
| } | |
| async function transcribeFile() { | |
| if (!currentAudioData) return; | |
| await initEngine(); | |
| await ensurePunctuator(); | |
| setStatus('processing', 'Transcribing…'); | |
| transcriptCard.style.display = 'block'; | |
| const segments = segmentAudio(currentAudioData); | |
| const parts = []; | |
| let totalMs = 0; | |
| for (let i = 0; i < segments.length; i++) { | |
| const seg = segments[i]; | |
| showProgress(true, `Segment ${i + 1} / ${segments.length}`); | |
| progressFill.style.width = `${((i + 1) / segments.length) * 100}%`; | |
| const slice = currentAudioData.subarray( | |
| Math.floor(seg.start * SAMPLE_RATE), | |
| Math.floor(seg.end * SAMPLE_RATE), | |
| ); | |
| const result = await engine.transcribe(slice); | |
| totalMs += result.ms; | |
| if (result.text) { | |
| parts.push(await punctuate(result.text)); | |
| renderTranscript({ committed: parts.join('\n\n'), tentative: '' }); | |
| } | |
| } | |
| showProgress(false); | |
| progressFill.style.width = '0%'; | |
| const audioSec = currentAudioData.length / SAMPLE_RATE; | |
| statsEl.textContent = `${audioSec.toFixed(1)}s audio · ${segments.length} segment(s) · ${totalMs.toFixed(0)} ms · ${(audioSec / (totalMs / 1000)).toFixed(0)}x realtime`; | |
| if (!parts.length) renderTranscript({ committed: '(no speech detected)', tentative: '' }); | |
| setStatus('ready', 'Transcription complete'); | |
| } | |
| /** Decode any audio file to 16 kHz mono float32. */ | |
| async function loadAudioFile(file) { | |
| setStatus('processing', 'Decoding audio…'); | |
| try { | |
| const audioCtx = new AudioContext({ sampleRate: SAMPLE_RATE }); | |
| const buffer = await audioCtx.decodeAudioData(await file.arrayBuffer()); | |
| let audio = buffer.getChannelData(0); | |
| if (buffer.numberOfChannels > 1) { | |
| const mixed = new Float32Array(audio.length); | |
| for (let c = 0; c < buffer.numberOfChannels; c++) { | |
| const ch = buffer.getChannelData(c); | |
| for (let i = 0; i < mixed.length; i++) mixed[i] += ch[i] / buffer.numberOfChannels; | |
| } | |
| audio = mixed; | |
| } | |
| // decodeAudioData already resampled to the context rate. | |
| await audioCtx.close(); | |
| currentAudioData = audio; | |
| await transcribeFile(); | |
| } catch (e) { | |
| console.error('audio decode failed:', e); | |
| setStatus('error', `Could not read that file: ${e.message || e}`); | |
| } | |
| } | |
| // -------------------------------------------------------------- transcript IO | |
| function transcriptText() { | |
| return committedEl.textContent + (tentativeEl.textContent || ''); | |
| } | |
| async function copyTranscript() { | |
| await navigator.clipboard.writeText(transcriptText()); | |
| const original = copyBtn.title; | |
| copyBtn.title = 'Copied'; | |
| setTimeout(() => { copyBtn.title = original; }, 1200); | |
| } | |
| function downloadTranscript() { | |
| const blob = new Blob([transcriptText()], { type: 'text/plain' }); | |
| const a = document.createElement('a'); | |
| a.href = URL.createObjectURL(blob); | |
| a.download = `transcript-${new Date().toISOString().slice(0, 19).replace(/[:T]/g, '-')}.txt`; | |
| a.click(); | |
| URL.revokeObjectURL(a.href); | |
| } | |
| function clearTranscript() { | |
| renderTranscript({ committed: '', tentative: '' }); | |
| transcriptCard.style.display = 'none'; | |
| statsEl.textContent = ''; | |
| if (coordinator) coordinator.paragraphs = []; | |
| } | |
| // ---------------------------------------------------------------------- wiring | |
| dictateBtn.addEventListener('click', toggleDictation); | |
| audioFile.addEventListener('change', (e) => { | |
| const file = e.target.files[0]; | |
| if (file) loadAudioFile(file); | |
| e.target.value = ''; | |
| }); | |
| copyBtn.addEventListener('click', copyTranscript); | |
| downloadBtn.addEventListener('click', downloadTranscript); | |
| clearBtn.addEventListener('click', clearTranscript); | |
| // Drag and drop onto the input card | |
| for (const [event, handler] of [ | |
| ['dragover', (e) => { e.preventDefault(); inputCard.classList.add('drag-over'); }], | |
| ['dragleave', () => inputCard.classList.remove('drag-over')], | |
| ['drop', (e) => { | |
| e.preventDefault(); | |
| inputCard.classList.remove('drag-over'); | |
| const file = e.dataTransfer.files[0]; | |
| if (file?.type.startsWith('audio/')) loadAudioFile(file); | |
| else setStatus('error', 'Please drop an audio file'); | |
| }], | |
| ]) { | |
| inputCard.addEventListener(event, handler); | |
| fileTile.addEventListener(event, (e) => e.stopPropagation(), { capture: false }); | |
| } | |
| punctCheckbox.addEventListener('change', () => { | |
| if (punctCheckbox.checked) ensurePunctuator(); | |
| }); | |
| // Space bar toggles dictation, unless a control has focus. | |
| document.addEventListener('keydown', (e) => { | |
| if (e.code !== 'Space' || e.repeat) return; | |
| if (['INPUT', 'TEXTAREA', 'SELECT', 'BUTTON'].includes(document.activeElement?.tagName)) return; | |
| e.preventDefault(); | |
| toggleDictation(); | |
| }); | |
| setStatus('loading', 'Loading…'); | |
| initEngine().catch(() => {}); | |
| })(); | |
| })(); | |