/* * GENERATED by build_bundle.py -- do not edit. * * The app's ES modules concatenated into one classic script. A private * Space's cross-site iframe does not authenticate the module fetches issued * while the page is parsing (they 401, and the module map then caches that * failure), whereas classic scripts always carry the cookie. Edit the * sources listed below and re-run build_bundle.py. * * Sources, in dependency order: ctc-frontend.js, ctc-tokenizer.js, ctc-engine.js, ctc-streaming.js, app.js */ (() => { 'use strict'; // Stands in for the module namespace: each module destructures what it // imports off this and assigns its exports back onto it. const __M = {}; // ==== ctc-frontend.js =================================================== (() => { /** * Audio front-end for the FastConformer CTC model. * * A port of the model's `CtcConformerProcessor._frontend` (log-mel + deltas + * frame stacking) plus the sidecar's `_agc`. The mel filterbank and the STFT * window are NOT recomputed here -- they are loaded from `mel_filters.bin` / * `stft_window.bin`, dumped by export_ctc_onnx.py straight out of the torchaudio * objects the Python path uses. Reimplementing torchaudio's htk mel scale * (norm=None) and torch.stft's window centring in JS is the most likely place for * a silent mismatch, and a front-end that is subtly wrong degrades WER without * ever raising an error. * * Verified bit-for-bit against the Python features by test-frontend.mjs. */ const EPS = 1e-10; /** In-place iterative radix-2 complex FFT (Cooley-Tukey, decimation in time). */ class FFT { constructor(size) { this.size = size; this.levels = Math.log2(size) | 0; if (2 ** this.levels !== size) throw new Error(`FFT size ${size} is not a power of two`); // Bit-reversal permutation and twiddle tables, precomputed once: the // front-end runs this transform ~100 times per second of audio. this.rev = new Uint32Array(size); for (let i = 0; i < size; i++) { let r = 0; for (let b = 0; b < this.levels; b++) r |= ((i >> b) & 1) << (this.levels - 1 - b); this.rev[i] = r; } this.cos = new Float32Array(size / 2); this.sin = new Float32Array(size / 2); for (let i = 0; i < size / 2; i++) { this.cos[i] = Math.cos((-2 * Math.PI * i) / size); this.sin[i] = Math.sin((-2 * Math.PI * i) / size); } } /** Transform `re`/`im` in place (both length `size`). */ run(re, im) { const { size, rev, cos, sin } = this; for (let i = 0; i < size; i++) { const j = rev[i]; if (j > i) { let t = re[i]; re[i] = re[j]; re[j] = t; t = im[i]; im[i] = im[j]; im[j] = t; } } for (let len = 2; len <= size; len <<= 1) { const half = len >> 1; const step = size / len; for (let i = 0; i < size; i += len) { for (let j = 0, k = 0; j < half; j++, k += step) { const c = cos[k], s = sin[k]; const a = i + j, b = a + half; const tre = re[b] * c - im[b] * s; const tim = re[b] * s + im[b] * c; re[b] = re[a] - tre; im[b] = im[a] - tim; re[a] += tre; im[a] += tim; } } } } } /** * Centred moving average, cumsum-based (O(n)). * * Mirrors the sidecar's `_moving_avg`, including its edge handling: the valid * region is padded by replicating the first/last valid value, with the extra * sample going on the right when the window is even. */ function movingAvg(v, win) { const n = v.length; win = Math.max(1, win | 0); if (win <= 1 || n <= 1) return v; win = Math.min(win, n); // float64 accumulator: a float32 cumsum over ~10 s of samples loses enough // precision to shift the gain envelope (the Python side uses float64 too). const csum = new Float64Array(n + 1); for (let i = 0; i < n; i++) csum[i + 1] = csum[i] + v[i]; const valid = n - win + 1; const padL = (win - 1) >> 1; const out = new Float32Array(n); const first = (csum[win] - csum[0]) / win; const last = (csum[n] - csum[n - win]) / win; for (let i = 0; i < n; i++) { const k = i - padL; out[i] = k < 0 ? first : k >= valid ? last : (csum[k + win] - csum[k]) / win; } return out; } /** * Automatic gain control -- port of the sidecar's `_agc`. * * Not cosmetic: the CTC front-end normalizes log-mel against the clip's *global * peak*, so a quiet tail 20-40 dB below a loud head gets floored and silently * dropped. A smoothed moving-RMS gain toward `target` equalizes the levels. */ function agc(x, { sampleRate = 16000, target = 0.12, winMs = 150, maxGain = 20 } = {}) { const n = x.length; if (n === 0) return x; const win = Math.max(1, Math.round((sampleRate * winMs) / 1000)); const sq = new Float32Array(n); for (let i = 0; i < n; i++) sq[i] = x[i] * x[i]; const ms = movingAvg(sq, win); const g = new Float32Array(n); for (let i = 0; i < n; i++) { const env = Math.sqrt(ms[i] + 1e-9); g[i] = Math.min(Math.max(target / Math.max(env, 1e-4), 0), maxGain); } const gs = movingAvg(g, win); // smooth the gain to avoid pumping const y = new Float32Array(n); let peak = 0; for (let i = 0; i < n; i++) { y[i] = x[i] * gs[i]; const a = Math.abs(y[i]); if (a > peak) peak = a; } if (peak > 0.97) { const k = 0.97 / peak; for (let i = 0; i < n; i++) y[i] *= k; } return y; } class CtcFrontend { constructor(config, melFilters, window) { this.config = config; this.melFilters = melFilters; // [n_mels][n_freqs], mel-major this.window = window; // n_fft, already centred by the exporter this.fft = new FFT(config.n_fft); this.nFreqs = config.n_fft / 2 + 1; } static async load(baseUrl = './onnx-ctc/') { const get = async (name, as) => { const r = await fetch(baseUrl + name); if (!r.ok) throw new Error(`front-end asset ${name}: HTTP ${r.status}`); return as === 'json' ? r.json() : new Float32Array(await r.arrayBuffer()); }; const [config, melFilters, window] = await Promise.all([ get('frontend_config.json', 'json'), get('mel_filters.bin'), get('stft_window.bin'), ]); return new CtcFrontend(config, melFilters, window); } /** Feature frames a clip of `nSamples` produces (before padding to pad_multiple). */ frameCount(nSamples) { const s = this.config.stack_factor; const melFrames = Math.floor(nSamples / this.config.hop_length); return s * Math.ceil(melFrames / s); } /** * waveform (Float32Array, 16 kHz mono) -> stacked log-mel(+delta) features. * * Returns `{ data, frames, dim }` where `data` is [frames, dim] row-major, * ready to become the model's `input_features` tensor. */ compute(wav) { const { hop_length: hop, n_fft: nFft, n_mels: nMels, stack_factor: stack, logmel_floor_db: floorDb, deltas: useDeltas } = this.config; const nFrames = this.frameCount(wav.length); if (nFrames === 0) return { data: new Float32Array(0), frames: 0, dim: this.config.input_dim }; // The processor right-pads the waveform so the trailing partial stack // group is filled rather than dropped, then slices to exactly nFrames. const need = (nFrames - 1) * hop + 1; let x = wav; if (x.length < need) { x = new Float32Array(need); x.set(wav); } // torch.stft(center=True, pad_mode='reflect'): reflect-pad by n_fft/2 so // frame i is centred on sample i*hop. const half = nFft >> 1; const padded = new Float32Array(x.length + nFft); padded.set(x, half); for (let i = 0; i < half; i++) { padded[half - 1 - i] = x[Math.min(i + 1, x.length - 1)]; padded[half + x.length + i] = x[Math.max(x.length - 2 - i, 0)]; } // Mel spectrogram, mel-major ([nMels][nFrames]) because deltas run along time. const mel = new Float32Array(nMels * nFrames); const re = new Float32Array(nFft); const im = new Float32Array(nFft); const power = new Float32Array(this.nFreqs); for (let t = 0; t < nFrames; t++) { const off = t * hop; for (let i = 0; i < nFft; i++) { re[i] = padded[off + i] * this.window[i]; im[i] = 0; } this.fft.run(re, im); for (let k = 0; k < this.nFreqs; k++) power[k] = re[k] * re[k] + im[k] * im[k]; for (let m = 0; m < nMels; m++) { const fb = m * this.nFreqs; let acc = 0; for (let k = 0; k < this.nFreqs; k++) acc += this.melFilters[fb + k] * power[k]; mel[m * nFrames + t] = acc; } } // log10 with a floor relative to the clip's global peak, then the /4 +1 // scaling the model was trained with. let mx = -Infinity; for (let i = 0; i < mel.length; i++) { const v = Math.log10(Math.max(mel[i], EPS)); mel[i] = v; if (v > mx) mx = v; } const floor = mx - floorDb; for (let i = 0; i < mel.length; i++) mel[i] = Math.max(mel[i], floor) / 4 + 1; // Deltas: torchaudio.functional.compute_deltas(win_length=3) is // (x[t+1] - x[t-1]) / 2 with replicate edge padding. const nCh = useDeltas ? nMels * 2 : nMels; const chans = new Float32Array(nCh * nFrames); chans.set(mel); if (useDeltas) { for (let m = 0; m < nMels; m++) { const src = m * nFrames; const dst = (nMels + m) * nFrames; for (let t = 0; t < nFrames; t++) { const prev = mel[src + Math.max(t - 1, 0)]; const next = mel[src + Math.min(t + 1, nFrames - 1)]; chans[dst + t] = (next - prev) / 2; } } } // Stack: (nCh, nFrames) -> (nFrames/stack, stack*nCh), i.e. row j is the // channels of frame stack*j followed by those of frame stack*j+1. const outFrames = nFrames / stack; const dim = stack * nCh; const data = new Float32Array(outFrames * dim); for (let j = 0; j < outFrames; j++) { for (let k = 0; k < stack; k++) { const t = j * stack + k; const base = j * dim + k * nCh; for (let c = 0; c < nCh; c++) data[base + c] = chans[c * nFrames + t]; } } return { data, frames: outFrames, dim }; } } Object.assign(__M, { agc, CtcFrontend }); })(); // ==== ctc-tokenizer.js ================================================== (() => { /** * Decoder for the model's 16384-entry ByteLevel BPE vocabulary. * * Only decoding is needed (the model never encodes text), so export_ctc_onnx.py * dumps the id->piece slice that can reach the decoder as vocab.json (~170 KB) * instead of shipping the whole tokenizer.json. * * granite-speech-4.2-470m-turboctc uses a GPT-2 style **ByteLevel** tokenizer: * every byte 0-255 is represented by one printable codepoint (space is `Ġ`), so a * piece is a string of byte-stand-ins rather than text, and one UTF-8 character * may be split across several pieces ("é" arrives as "Ã" + "©"). Decoding is * therefore: map each character back to its byte, concatenate, then UTF-8 decode * once at the end -- never per piece, or multi-byte characters break. * * (The earlier C-T26.18 build used a SentencePiece tokenizer whose decoder was * Replace(U+2581 -> space) / ByteFallback / Fuse / Strip. That is a different * algorithm and would silently produce mojibake here, so the exporter asserts the * tokenizer's decoder type matches this implementation.) */ /** * GPT-2's byte <-> codepoint table, inverted to codepoint -> byte. * * The 188 bytes that are already printable map to themselves; the remaining 68 * are moved up to U+0100 and beyond so no piece ever contains a control * character or a bare space. */ function byteLevelCharToByte() { const bytes = []; for (let b = 0x21; b <= 0x7e; b++) bytes.push(b); // !..~ for (let b = 0xa1; b <= 0xac; b++) bytes.push(b); // ¡..¬ for (let b = 0xae; b <= 0xff; b++) bytes.push(b); // ®..ÿ const codepoints = bytes.slice(); const printable = new Set(bytes); let next = 0; for (let b = 0; b < 256; b++) { if (!printable.has(b)) { bytes.push(b); codepoints.push(256 + next); next++; } } const map = new Map(); for (let i = 0; i < bytes.length; i++) map.set(codepoints[i], bytes[i]); return map; } class CtcDecoder { constructor(pieces, numSpecialTokens) { this.numSpecialTokens = numSpecialTokens; const charToByte = byteLevelCharToByte(); // Pre-resolve every piece to its bytes, so decoding a frame of ids is a // few array copies rather than per-call string work. this.encoded = new Array(pieces.length); for (let i = 0; i < pieces.length; i++) { const piece = pieces[i]; const bytes = new Uint8Array(piece.length); let n = 0; for (const ch of piece) { const b = charToByte.get(ch.codePointAt(0)); // A codepoint outside the table means this vocab was not produced // by a ByteLevel tokenizer; skip rather than emit a wrong byte. if (b !== undefined) bytes[n++] = b; } this.encoded[i] = bytes.subarray(0, n); } this.utf8Decoder = new TextDecoder('utf-8'); } static async load(baseUrl = './onnx-ctc/') { const [vocab, config] = await Promise.all([ fetch(baseUrl + 'vocab.json').then((r) => { if (!r.ok) throw new Error(`vocab.json: HTTP ${r.status}`); return r.json(); }), fetch(baseUrl + 'frontend_config.json').then((r) => { if (!r.ok) throw new Error(`frontend_config.json: HTTP ${r.status}`); return r.json(); }), ]); return new CtcDecoder(vocab, config.num_special_tokens); } /** Decode content ids (all >= numSpecialTokens) to text. */ decode(ids) { let total = 0; for (const id of ids) { const e = this.encoded[id - this.numSpecialTokens]; if (e) total += e.length; } const buf = new Uint8Array(total); let at = 0; for (const id of ids) { const e = this.encoded[id - this.numSpecialTokens]; if (!e) continue; // out-of-range id: skip rather than poison the string buf.set(e, at); at += e.length; } return this.utf8Decoder.decode(buf); } } /** * Greedy CTC collapse: drop repeats, then the blank, then any remaining * below-content ids. * * Order matters and matches the Python path: repeats are collapsed against the * *raw* previous id (blanks included), which is what lets a blank between two * identical tokens preserve both. For this model `numSpecialTokens` is 1 -- the * CTC blank at id 0 is the only non-text id, and everything above it is a * tokenizer id as-is. */ function ctcCollapse(ids, { blankId = 0, numSpecialTokens = 1 } = {}) { const out = []; let prev = -1; for (let i = 0; i < ids.length; i++) { const id = ids[i]; if (id !== prev && id !== blankId && id >= numSpecialTokens) out.push(id); prev = id; } return out; } Object.assign(__M, { CtcDecoder, ctcCollapse }); })(); // ==== ctc-engine.js ===================================================== (() => { const { CtcFrontend, agc, CtcDecoder, ctcCollapse } = __M; /** * FastConformer CTC inference engine (onnxruntime-web, WebGPU). * * The model is a custom `ctc_conformer` architecture with no transformers.js * support, so this drives an ONNX session directly: features -> one forward pass * -> greedy CTC collapse. There is no autoregressive loop, no KV cache and no * chat template; a window is one `session.run`. * * Two contracts come from export_ctc_onnx.py and must hold here: * * - **Frames are padded to `pad_multiple`.** The encoder chunks attention into * context_size (128) blocks at every layer and subsamples time by 4, so a * frame count with no partial block anywhere is a multiple of 512 (10.24 s). * The exported graph therefore has no tail-block path, and short windows are * padded up with `padding_mask` marking the pad. Measured cost of the masked * padding on granite-speech-4.2-470m-turboctc: none -- all 14 fixture clips * decode identically to the unpadded path, with zero per-frame argmax flips. * (Its encoder masks inside the conv module specifically so a padded frame * cannot leak through the stride-2 kernel into the last valid frames.) * * - **The graph returns ids, not logits.** `[1, T/4, 16384]` float32 logits * would be an ~8 MB device-to-host copy per window; argmax and the * log-softmax max happen inside the graph instead, so a window costs ~1 KB of * output. `top_logprob` is the confidence signal the streaming gate uses. */ const CACHE_NAME = 'granite-ctc-models-v1'; /** * Cache-first fetch, so a reload doesn't re-download hundreds of MB. * * The body is streamed to completion first and only then handed to the cache. * Doing it the other way round -- `await cache.put(response.clone())` before * reading -- makes the tee buffer the entire 312 MB while the cache write drains * it, so the progress callback fires nothing for minutes and the UI looks hung. * A failed cache write (quota, private mode) must not fail the load either: the * bytes are already in hand, and the only cost is re-downloading next time. */ async function cachedFetch(url, onProgress) { const cache = await caches.open(CACHE_NAME).catch(() => null); const hit = await cache?.match(url).catch(() => null); if (hit) { onProgress?.({ url, loaded: 1, total: 1, cached: true }); return hit.arrayBuffer(); } const response = await fetch(url); if (!response.ok) throw new Error(`${url}: HTTP ${response.status}`); const total = Number(response.headers.get('content-length')) || 0; let buffer; if (!response.body) { buffer = await response.arrayBuffer(); } else { const reader = response.body.getReader(); const chunks = []; let loaded = 0; for (;;) { const { done, value } = await reader.read(); if (done) break; chunks.push(value); loaded += value.length; onProgress?.({ url, loaded, total }); } const bytes = new Uint8Array(loaded); let at = 0; for (const c of chunks) { bytes.set(c, at); at += c.length; } buffer = bytes.buffer; } if (cache) { try { await cache.put(url, new Response(buffer.slice(0), { headers: { 'Content-Type': 'application/octet-stream', 'Content-Length': String(buffer.byteLength) }, })); } catch (e) { console.warn(`could not cache ${url} (will re-download next time):`, e); } } return buffer; } class CtcEngine { constructor({ session, frontend, decoder, config, device }) { this.session = session; this.frontend = frontend; this.decoder = decoder; this.config = config; this.device = device; } /** * Create the session and load the front-end assets. * * `device: 'webgpu'` falls back to wasm if the adapter or a kernel is * missing -- a CTC pass is still usable on wasm, just slower. */ static async load({ baseUrl = './onnx-ctc/', modelFile = 'ctc_conformer_q4f16-b16.onnx', device = 'webgpu', onProgress = null, allowFallback = true, } = {}) { const [frontend, decoder] = await Promise.all([ CtcFrontend.load(baseUrl), CtcDecoder.load(baseUrl), ]); const modelUrl = baseUrl + modelFile; // The exporter writes weights to `.onnx.data` and records that // basename inside the graph, so the session's externalData path must // match it exactly. const dataName = `${modelFile}.data`; const [modelBuffer, dataBuffer] = await Promise.all([ cachedFetch(modelUrl, onProgress), cachedFetch(baseUrl + dataName, onProgress), ]); const options = { executionProviders: [device], externalData: [{ path: dataName, data: dataBuffer }], graphOptimizationLevel: 'all', }; let session; let used = device; let fallbackError = null; try { session = await ort.InferenceSession.create(modelBuffer, options); } catch (e) { if (device !== 'webgpu' || !allowFallback) throw e; console.warn('WebGPU session failed, falling back to wasm:', e); fallbackError = String(e?.message || e); used = 'wasm'; session = await ort.InferenceSession.create(modelBuffer, { ...options, executionProviders: ['wasm'], }); } const engine = new CtcEngine({ session, frontend, decoder, config: frontend.config, device: used }); // A WebGPU kernel can also fail on its *first run* rather than at session // build (ORT-web reports unsupported MatMulNBits shapes that way), so the // warmup pass is where a fallback is really decided. const warmupError = await engine.warmup(); if (warmupError && used === 'webgpu') { if (!allowFallback) throw new Error(`WebGPU warmup failed: ${warmupError}`); console.warn('WebGPU warmup failed, falling back to wasm:', warmupError); fallbackError = warmupError; engine.device = 'wasm'; engine.session = await ort.InferenceSession.create(modelBuffer, { ...options, executionProviders: ['wasm'], }); await engine.warmup(); } engine.fallbackError = fallbackError; return engine; } /** * Run one throwaway pass so shader compilation doesn't land on the user's * first spoken window (measured on WebGPU: 1271 ms first call vs 295 ms * steady state). Every window is padded to the same frame count, so a single * warmup covers the shapes the streaming loop will use. */ async warmup() { try { const frames = this.config.pad_multiple; await this.session.run({ input_features: new ort.Tensor('float32', new Float32Array(frames * this.config.input_dim), [1, frames, this.config.input_dim]), padding_mask: new ort.Tensor('float32', new Float32Array(frames), [1, frames]), }); return null; } catch (e) { return String(e?.message || e); } } /** Encoder output frames covering `frames` of real audio (both subsample blocks floor-halve). */ static realOutFrames(frames) { return Math.floor(Math.floor(frames / 2) / 2); } /** * Transcribe a waveform (Float32Array, 16 kHz mono). * * `applyAgc` mirrors the sidecar: the log-mel front-end normalizes against * the clip's global peak, so without it a quiet tail after a loud passage is * floored away and its words are silently dropped. */ async transcribe(wav, { applyAgc = true } = {}) { const t0 = performance.now(); const audio = applyAgc ? agc(wav, { sampleRate: this.config.sample_rate }) : wav; const { data, frames, dim } = this.frontend.compute(audio); if (frames === 0) return { text: '', words: [], avgLogprob: null, frames: 0, ms: 0 }; const multiple = this.config.pad_multiple; const padded = Math.ceil(frames / multiple) * multiple; const features = new Float32Array(padded * dim); features.set(data); const mask = new Float32Array(padded).fill(1); mask.fill(0, 0, frames); const outputs = await this.session.run({ input_features: new ort.Tensor('float32', features, [1, padded, dim]), padding_mask: new ort.Tensor('float32', mask, [1, padded]), }); // The graph emits padded/4 frames; only those covering real audio may // contribute tokens, so trim before collapsing. const nOut = CtcEngine.realOutFrames(frames); const ids = outputs.ids.data.subarray(0, nOut); const logp = outputs.top_logprob.data.subarray(0, nOut); const content = ctcCollapse(ids, { blankId: this.config.blank_id, numSpecialTokens: this.config.num_special_tokens, }); const text = this.decoder.decode(content).trim(); // Confidence: mean top log-probability over non-blank frames. The CTC // analogue of a mean per-token logprob -- clean speech sits near -0.03, // silence/noise confabulation near -1.85 and below. let sum = 0, n = 0; for (let i = 0; i < nOut; i++) { if (ids[i] !== this.config.blank_id) { sum += logp[i]; n++; } } return { text, words: text.length ? text.split(/\s+/) : [], avgLogprob: n ? sum / n : null, frames, paddedFrames: padded, ms: performance.now() - t0, }; } } Object.assign(__M, { CtcEngine }); })(); // ==== ctc-streaming.js ================================================== (() => { /** * Live dictation: growing window + LocalAgreement-2, ported from the macOS * GraniteLiveDictation app (SpeechGate / Reconciler / StreamingCoordinator). * * continuous 16 kHz capture -> rolling window + energy SpeechGate * every `step` while speech is in the window: * window = audio[segment start .. now] (grows; overlaps the last) * hypo = words(CTC transcription of window) * a word COMMITS once two consecutive windows agree on it (longest * common prefix); everything past that is TENTATIVE (shown grey) * on a pause: one clean full-window pass commits the segment and re-anchors * * The window *grows* rather than sliding because the model returns text with no * word timestamps -- there is no safe place to trim a fixed window without * risking a mid-word cut. Anchoring each window at the start of the current * speech segment makes reconciliation a plain prefix comparison and keeps the * window bounded (a phrase between pauses). * * The tuning constants are the ones the macOS app converged on against real * far-field recordings, not defaults -- see the comments on each. */ const SAMPLE_RATE = 16000; const INT16_SCALE = 32768; // gate thresholds are in Int16 RMS units (as tuned) /** * URL to hand `audioWorklet.addModule()`. * * addModule fetches its argument as a *module*, which is the one request type a * private Space's cross-site iframe fails to authenticate — measured there, the * same file returns 200 to a plain fetch and intermittently fails as a module * import. That would break dictation at the moment the user presses record, well * after everything else had loaded cleanly. So fetch the source (plain fetches * work) and hand addModule a blob: URL, which needs no network request at all. * Falls back to the direct URL, which is what works everywhere else. * * Resolved against this module rather than the document so it survives being * served from a subdirectory; build_bundle.py rewrites `document.baseURI` to * `document.baseURI` for the bundled build, where the two are the same directory. */ let cachedWorkletUrl = null; async function captureWorkletUrl() { if (cachedWorkletUrl) return cachedWorkletUrl; const direct = new URL('./ctc-capture-worklet.js', document.baseURI).href; try { const r = await fetch(direct); if (!r.ok) throw new Error(`HTTP ${r.status}`); const source = await r.text(); cachedWorkletUrl = URL.createObjectURL(new Blob([source], { type: 'text/javascript' })); } catch (e) { console.warn('could not preload the capture worklet; using its URL directly:', e); cachedWorkletUrl = direct; } return cachedWorkletUrl; } /** * Energy speech/silence gate with an adaptive noise floor. * * Not a neural VAD -- an RMS gate whose enter/exit thresholds float with the * measured background level. That matters for its one job: finding a *pause* so * the growing window can flush. With fixed thresholds, a room whose ambient RMS * sits above the silence cutoff never produces a silence frame, the endpoint * timer never fires, and every window runs to the maxWindow cap and cuts * mid-sentence. */ class SpeechGate { constructor(config = {}) { this.config = { endpointSilenceMs: 600, // contiguous below-exit audio that ends a segment frameMs: 30, // Biased toward inclusion: this CTC model doesn't confabulate on // silence the way an autoregressive decoder does (it decodes to empty // / low confidence, and the confidence gate backstops it), so // admitting soft audio costs nothing -- while excluding it dropped // real words (soft onsets never triggered, soft tails got endpointed // mid-phrase). onRatio: 1.5, minEnter: 80, // Exit stays near the original: dropping it toward ambient stops // pause detection altogether and every segment runs to maxWindow, // which wrecks paragraphing. offRatio: 1.4, minExit: 60, onFrames: 2, // debounce transient clicks // Seeded from the *mean* RMS over the startup window, not the min: the // min tracks the quietest valley of fluctuating noise and seeds the // floor far below the ambient body, which then lets noise read as // speech. calibrationMs: 360, floorAdapt: 0.05, // per-frame EMA; at 30 ms frames ~0.6 s to settle initialNoiseFloor: 60, maxNoiseFloor: 4000, ...config, }; this.frameLen = Math.max(1, (SAMPLE_RATE * this.config.frameMs) / 1000); this.calibrationFrames = Math.floor(this.config.calibrationMs / this.config.frameMs); this.onEvent = null; this.isSpeaking = false; this.noiseFloor = this.config.initialNoiseFloor; this.silenceMs = 0; this.onCount = 0; this.pending = []; this.pendingLen = 0; this.calibLeft = this.calibrationFrames; this.calibSum = 0; this.calibCount = 0; } get enterThreshold() { return Math.max(this.config.minEnter, this.noiseFloor * this.config.onRatio); } get exitThreshold() { return Math.max(this.config.minExit, this.noiseFloor * this.config.offRatio); } /** * Clear per-utterance state. The noise floor and calibration are deliberately * preserved: they describe the *room*, not the utterance, and this is called * on every commit (including maxWindow cuts taken mid-speech), where * re-seeding would calibrate the floor from speech energy and blind the gate. */ reset() { this.isSpeaking = false; this.silenceMs = 0; this.onCount = 0; this.pending = []; this.pendingLen = 0; } /** Re-arm calibration -- for a fresh capture session / new environment. */ recalibrate() { this.noiseFloor = this.config.initialNoiseFloor; this.calibLeft = this.calibrationFrames; this.calibSum = 0; this.calibCount = 0; } /** Feed a chunk of float samples (any length); re-framed into fixed blocks. */ feed(samples) { if (!samples.length) return; this.pending.push(samples); this.pendingLen += samples.length; while (this.pendingLen >= this.frameLen) { const frame = this.takeFrame(); // process() emits events synchronously and a handler may call reset() // re-entrantly (which drops `pending`), so the frame is removed first. this.process(frame); } } takeFrame() { const frame = new Float32Array(this.frameLen); let at = 0; while (at < this.frameLen) { const head = this.pending[0]; const take = Math.min(head.length, this.frameLen - at); frame.set(take === head.length ? head : head.subarray(0, take), at); at += take; if (take === head.length) this.pending.shift(); else this.pending[0] = head.subarray(take); } this.pendingLen -= this.frameLen; return frame; } process(frame) { const rms = rmsInt16(frame); if (this.calibLeft > 0) { this.calibLeft--; this.calibSum += rms; this.calibCount++; this.noiseFloor = Math.min(Math.max(this.calibSum / this.calibCount, 1), this.config.maxNoiseFloor); return; } this.adaptNoise(rms); if (this.isSpeaking) { if (rms < this.exitThreshold) { this.silenceMs += this.config.frameMs; if (this.silenceMs >= this.config.endpointSilenceMs) { this.isSpeaking = false; this.silenceMs = 0; this.onCount = 0; this.onEvent?.('speechEnded'); } } else { this.silenceMs = 0; // speech resumed; reset the endpoint timer } } else if (rms >= this.enterThreshold) { this.onCount++; if (this.onCount >= this.config.onFrames) { this.isSpeaking = true; this.silenceMs = 0; this.onCount = 0; this.onEvent?.('speechStarted'); } } else { this.onCount = 0; } } /** * Learn the floor from ambient frames only, tracking the *body* (mean) of the * ambient level. Updating solely when not in a segment AND below the enter * threshold keeps sustained speech (or a speech attack) from dragging the * floor up to speech level. */ adaptNoise(rms) { if (this.isSpeaking || rms >= this.enterThreshold) return; const a = this.config.floorAdapt; this.noiseFloor = Math.min(Math.max((1 - a) * this.noiseFloor + a * rms, 1), this.config.maxNoiseFloor); } } function rmsInt16(samples) { if (!samples.length) return 0; let acc = 0; for (let i = 0; i < samples.length; i++) acc += samples[i] * samples[i]; return Math.sqrt(acc / samples.length) * INT16_SCALE; } /** * LocalAgreement-2 reconciliation. * * Each step transcribes the growing window and yields a fresh hypothesis. Because * the window overlaps the previous one, the two should agree on the stable part * and differ only at the unfinished tail. A word is committed once two * *consecutive* hypotheses agree on it (their longest common prefix, compared * case-insensitively and ignoring edge punctuation); everything past that is * tentative and may still change. */ class Reconciler { constructor() { this.confirmed = []; this.prevHypo = []; } update(hypothesis) { const agreed = commonPrefixCount(this.prevHypo, hypothesis); if (agreed > this.confirmed.length) { this.confirmed.push(...hypothesis.slice(this.confirmed.length, agreed)); } this.prevHypo = hypothesis; const tentative = hypothesis.length > this.confirmed.length ? hypothesis.slice(this.confirmed.length) : []; return { confirmed: this.confirmed, tentative }; } /** End of segment: a clean full-window hypothesis is authoritative. */ finalize(words) { this.confirmed = words; this.prevHypo = words; return this.confirmed; } reset() { this.confirmed = []; this.prevHypo = []; } } /** Lowercase and strip edge punctuation so "Cat," and "cat" match. */ function normalizeWord(word) { return word.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, '').toLowerCase(); } function commonPrefixCount(a, b) { const n = Math.min(a.length, b.length); let i = 0; while (i < n && normalizeWord(a[i]) === normalizeWord(b[i])) i++; return i; } /** * Stock phrases the model confidently emits on ambient/silence -- the classic ASR * hallucination. Confidence can't catch these (they score *higher* than genuine * soft speech), so they're matched against the whole commit and dropped only when * word density is low: a silence-padded hallucination is long and slow, a real * crisp "thank you" is short and fast. */ const CANNED_HALLUCINATIONS = new Set([ 'thank you', 'thank you very much', 'thanks for watching', 'thank you for watching', 'thanks for watching everyone', ]); class StreamingCoordinator { /** * @param {object} opts * @param {import('./ctc-engine.js').CtcEngine} opts.engine * @param {(state: {committed: string, tentative: string}) => void} opts.onTranscript * @param {(level: number) => void} [opts.onLevel] 0..1 input level * @param {(info: object) => void} [opts.onStep] per-window diagnostics * @param {(text: string) => Promise} [opts.punctuate] committed-text post-pass */ constructor({ engine, onTranscript, onLevel = null, onStep = null, punctuate = null, tuning = {} }) { this.engine = engine; this.onTranscript = onTranscript; this.onLevel = onLevel; this.onStep = onStep; this.punctuate = punctuate; this.tuning = { stepMs: 400, // re-transcribe cadence. A window costs one padded // forward pass (>= 10.24 s of frames, see // ctc-engine), so this is coarser than the macOS // app's 100 ms; an in-flight tick is dropped, so // it self-limits either way. minWindowSec: 0.4, // don't transcribe windows shorter than this maxWindowSec: 10.0, // force-commit a window that grows this long with no pause newlineGapMs: 1200, // a pause >= this starts a new paragraph hardPauseMs: 1500, prerollSec: 0.5, // audio kept before speech onset, so the first word isn't clipped // Calibrated on real far-field logs: clean speech ~ -0.03, but genuine // soft utterances scored -0.98..-1.34 and were wrongly rejected at a // -0.6 gate, dropping whole clips; silence confabulation sits at // -1.85 and below, so -1.5 lands in the gap. minCommitAvgLogprob: -1.5, cannedHallucinationMaxWps: 2.0, ...tuning, }; this.gate = new SpeechGate(); this.gate.onEvent = (e) => this.handleGate(e); this.reconciler = new Reconciler(); this.window = []; // Float32Array chunks since the last commit this.windowLen = 0; this.windowHasSpeech = false; this.transcribing = false; this.windowGen = 0; this.paragraphs = []; // committed text, one entry per paragraph this.lastVoiceAt = 0; this.segmentGapMs = Infinity; this.active = false; this.audioContext = null; this.mediaStream = null; this.stepTimer = null; // One ONNX session can't have two concurrent run() calls, and a commit // fired by the gate can land while a step transcription is still in // flight, so every engine call goes through this chain. this.engineLock = Promise.resolve(); // Commits are started from the gate callback (inside onSamples), which // can't await them; stop() drains these so the last words aren't lost. this.pending = new Set(); } /** Serialize engine access; returns the callback's result. */ async withEngine(fn) { const prior = this.engineLock; let release; this.engineLock = new Promise((r) => { release = r; }); try { await prior; return await fn(); } finally { release(); } } /** Track a fire-and-forget async task so stop() can wait for it. */ track(promise) { this.pending.add(promise); promise.catch(() => {}).finally(() => this.pending.delete(promise)); return promise; } /** Resolve once no commit or step is outstanding. */ async idle() { while (this.pending.size) await Promise.allSettled([...this.pending]); await this.engineLock; } async start() { if (this.active) return; this.mediaStream = await navigator.mediaDevices.getUserMedia({ audio: { channelCount: 1, echoCancellation: true, noiseSuppression: false, // the gate wants the true ambient level autoGainControl: false, // the engine's own AGC handles levels }, }); this.audioContext = new AudioContext({ sampleRate: SAMPLE_RATE }); await this.audioContext.audioWorklet.addModule(await captureWorkletUrl()); const source = this.audioContext.createMediaStreamSource(this.mediaStream); const node = new AudioWorkletNode(this.audioContext, 'ctc-capture'); node.port.onmessage = (e) => this.onSamples(e.data); source.connect(node); // Keep the graph pulling without routing mic audio to the speakers. const sink = this.audioContext.createGain(); sink.gain.value = 0; node.connect(sink).connect(this.audioContext.destination); this.resetWindow(); this.gate.recalibrate(); this.paragraphs = []; this.lastVoiceAt = 0; this.segmentGapMs = Infinity; this.active = true; this.publish([], []); this.stepTimer = setInterval(() => this.tick(), this.tuning.stepMs); } async stop() { if (!this.active) return; this.active = false; clearInterval(this.stepTimer); this.stepTimer = null; this.mediaStream?.getTracks().forEach((t) => t.stop()); await this.audioContext?.close(); this.audioContext = null; this.onLevel?.(0); // Let any commit the gate started finish before deciding whether the // window still holds uncommitted speech. await this.idle(); if (this.windowHasSpeech) await this.commit('stop'); this.resetWindow(); this.publish([], []); } // ---------------------------------------------------------------- capture onSamples(chunk) { this.window.push(chunk); this.windowLen += chunk.length; // Before speech is detected, keep only a short pre-roll. The window is // otherwise cleared only on commit, so a long ambient stretch would // accumulate unbounded and the first real utterance would commit an // enormous mostly-noise window. if (!this.windowHasSpeech) { const maxPreroll = Math.floor(SAMPLE_RATE * this.tuning.prerollSec); while (this.windowLen > maxPreroll && this.window.length > 1) { this.windowLen -= this.window.shift().length; } } const now = performance.now(); this.onLevel?.(meterLevel(rmsInt16(chunk))); this.gate.feed(chunk); // may fire speechEnded -> commit // Drive the window's speech state from the gate's debounced, // calibration-aware decision rather than a parallel per-chunk RMS test: // such a test has no debounce and runs during calibration, so one noise // chunk latches the window open and the model transcribes ambient. if (this.gate.isSpeaking) { if (!this.windowHasSpeech) { this.segmentGapMs = this.lastVoiceAt ? now - this.lastVoiceAt : Infinity; this.windowHasSpeech = true; } this.lastVoiceAt = now; } } handleGate(event) { if (event !== 'speechEnded') return; // capture is continuous; onset needs nothing // Called synchronously from onSamples, so the commit can only be tracked, // not awaited (see idle()). if (this.windowHasSpeech) this.track(this.commit('pause')); else this.resetWindow(); // pure-silence window: drop it } // -------------------------------------------------------------- step loop async tick() { if (!this.active || !this.windowHasSpeech || this.transcribing) return; if (this.windowLen / SAMPLE_RATE >= this.tuning.maxWindowSec) { await this.commit('maxWindow'); return; } const speech = this.trimmedSpeech(); if (speech.length / SAMPLE_RATE < this.tuning.minWindowSec) return; const gen = this.windowGen; this.transcribing = true; try { const result = await this.withEngine(() => this.engine.transcribe(speech)); if (gen !== this.windowGen) return; // window committed meanwhile; discard const { confirmed, tentative } = this.reconciler.update(result.words); this.publish(confirmed, tentative); this.onStep?.({ kind: 'step', windowSec: speech.length / SAMPLE_RATE, ...result }); } catch (e) { console.error('step transcription failed:', e); } finally { this.transcribing = false; } } /** * Commit the current window: snapshot it, immediately clear and re-anchor (so * capture continues uninterrupted and any in-flight partial is discarded), * then transcribe the snapshot once cleanly and append it. */ async commit(reason) { const speech = this.trimmedSpeech(); const windowSec = speech.length / SAMPLE_RATE; if (windowSec < this.tuning.minWindowSec) { this.resetWindow(); return; } const gapMs = this.segmentGapMs; this.resetWindow(); this.publish([], []); let result; try { result = await this.withEngine(() => this.engine.transcribe(speech)); } catch (e) { console.error('commit transcription failed:', e); return; } const words = result.words; const wps = windowSec > 0 ? words.length / windowSec : 0; // Confidence gate: a low mean logprob means the model wasn't sure of its // own tokens -- the signature of confabulation on noise/silence. A null // confidence fails open. if (result.avgLogprob !== null && result.avgLogprob < this.tuning.minCommitAvgLogprob) { this.onStep?.({ kind: 'reject', why: 'confidence', reason, windowSec, wps, ...result }); this.publish([], []); return; } const normalized = words.join(' ').toLowerCase(); if (CANNED_HALLUCINATIONS.has(normalized) && wps < this.tuning.cannedHallucinationMaxWps) { this.onStep?.({ kind: 'reject', why: 'canned', reason, windowSec, wps, ...result }); this.publish([], []); return; } if (!words.length) { this.publish([], []); return; } await this.appendSegment(words.join(' '), gapMs, reason); this.onStep?.({ kind: 'commit', reason, windowSec, wps, gapMs, ...result }); this.publish([], []); } /** * Append a committed segment, opening a new paragraph after a real pause. * * Unlike the macOS app, sentence punctuation and capitalization are NOT * synthesized here -- the model emits raw lowercase text and the punctuator * restores both. A maxWindow cut is mid-utterance, so it extends the current * paragraph and is re-punctuated together with what follows. */ async appendSegment(text, gapMs, reason) { const startsParagraph = this.paragraphs.length === 0 || gapMs >= this.tuning.hardPauseMs || (gapMs >= this.tuning.newlineGapMs && reason !== 'maxWindow'); if (startsParagraph) this.paragraphs.push(text); else this.paragraphs[this.paragraphs.length - 1] += ` ${text}`; if (this.punctuate) { const i = this.paragraphs.length - 1; const raw = this.paragraphs[i]; try { this.paragraphs[i] = await this.punctuate(raw); } catch (e) { console.warn('punctuation failed; keeping raw text:', e); } } } // ---------------------------------------------------------------- helpers /** * The speech-bearing span of the window, with leading/trailing silence * removed. Trailing trim matters: the front-end normalizes against the clip's * global peak, so an untrimmed quiet tail gets amplified and decoded into * spurious words. */ trimmedSpeech() { const audio = this.flattenWindow(); const frame = 480; // 30 ms const floor = this.gate.exitThreshold; let first = -1, last = -1; for (let i = 0; i < audio.length; i += frame) { const end = Math.min(i + frame, audio.length); if (rmsInt16(audio.subarray(i, end)) >= floor) { if (first < 0) first = i; last = end; } } if (first < 0) return new Float32Array(0); const margin = frame * 3; // ~90 ms of context each side return audio.subarray(Math.max(0, first - margin), Math.min(audio.length, last + margin)); } flattenWindow() { if (this.window.length === 1) return this.window[0]; const out = new Float32Array(this.windowLen); let at = 0; for (const c of this.window) { out.set(c, at); at += c.length; } this.window = [out]; // collapse so repeated ticks don't re-copy return out; } resetWindow() { this.window = []; this.windowLen = 0; this.windowHasSpeech = false; this.reconciler.reset(); this.gate.reset(); this.windowGen++; } publish(confirmed, tentative) { const live = confirmed.join(' '); const committed = live ? [...this.paragraphs, live].join('\n\n') : this.paragraphs.join('\n\n'); this.onTranscript({ committed, tentative: tentative.join(' ') }); } /** Plain text of everything committed so far. */ get transcript() { return this.paragraphs.join('\n\n'); } } /** Int16 RMS -> 0..1 meter level on a dBFS scale (-60 dBFS -> 0, 0 dBFS -> 1). */ function meterLevel(rms) { if (rms <= 0) return 0; const dbfs = 20 * Math.log10(rms / 32767); return Math.max(0, Math.min(1, (dbfs + 60) / 60)); } Object.assign(__M, { SpeechGate, Reconciler, StreamingCoordinator }); })(); // ==== app.js ============================================================ (() => { const { CtcEngine, SpeechGate, StreamingCoordinator } = __M; /** * Granite Live Dictation (WebGPU) * * Real-time in-browser dictation with ibm-granite/granite-speech-4.2-470m-turboctc, * a 473M-parameter FastConformer CTC model. Unlike the previous * granite-speech-4.1-2b build, this is not a transformers.js pipeline: * `ctc_conformer` is a custom architecture, so the ONNX session is driven directly * (see ctc-engine.js) -- one forward pass and a greedy collapse per window, no * autoregressive decode. That is what makes it fast enough to re-transcribe a * growing window several times a second. * * The model emits raw lowercase, unpunctuated English. Punctuation and * capitalization are restored by the separate punctuator model (punctuator.js), * which is loaded on demand because it is a 209 MB download of its own. * * `ort` is the global from the onnxruntime-web script tag in index.html, shared * with punctuator.js so both sessions use one runtime. */ const MODEL_DIR = './onnx-ctc/'; // 4-bit weights, block 32, on an fp16 graph (312 MB from a 1.9 GB fp32 export). // This is the only shape WebGPU can actually run: ORT-web's WebGPU EP enforces // `nbits == 4 || nbits == 2` on MatMulNBits, so an 8-bit build silently falls // back to wasm (1320 ms per window against 305 ms) and buys nothing -- on 8 // minutes of LibriSpeech the 4-bit weights cost 0.09% WER against the fp32 // export, and what residue remains in the browser comes from fp16 activations, // which more weight bits would not fix. const MODEL_FILE = 'ctc_conformer_q4f16.onnx'; const SAMPLE_RATE = 16000; // One padded forward pass covers 10.24 s of frames, so file segments are capped // near that: longer segments cost proportionally more and buy nothing. const MAX_SEGMENT_SEC = 20; let engine = null; let coordinator = null; let punctuatorReady = false; let punctuatorLoad = null; let isLoading = false; let currentAudioData = null; const statusDot = document.getElementById('statusDot'); const statusText = document.getElementById('statusText'); const dictateBtn = document.getElementById('dictateBtn'); const levelFill = document.getElementById('levelFill'); const audioFile = document.getElementById('audioFile'); const fileTile = document.querySelector('.file-label'); const inputCard = document.querySelector('.input-card'); const punctCheckbox = document.getElementById('punctCheckbox'); const transcriptCard = document.getElementById('transcriptCard'); const committedEl = document.getElementById('committedText'); const tentativeEl = document.getElementById('tentativeText'); const copyBtn = document.getElementById('copyBtn'); const downloadBtn = document.getElementById('downloadBtn'); const clearBtn = document.getElementById('clearBtn'); const progressFill = document.getElementById('progressFill'); const progressSection = document.getElementById('progressSection'); const progressText = document.getElementById('progressText'); const gpuInfo = document.getElementById('gpuInfo'); const statsEl = document.getElementById('stats'); function setStatus(state, message) { statusDot.className = `status-dot ${state}`; statusText.textContent = message; } function showProgress(show, text = '') { progressSection.style.display = show ? 'block' : 'none'; if (text) progressText.textContent = text; } // ---------------------------------------------------------------- model load /** * Hand ORT its wasm assets directly, fetched with a plain `fetch()`. * * ORT locates its backend by dynamically `import()`ing * ort-wasm-simd-threaded.jsep.mjs. In a private Space's cross-site iframe that * module fetch is not authenticated -- the same failure that stopped app.js's own * imports and forced the bundle -- and it surfaces from session creation as * "no available backend found. ERR: [wasm] previous call to 'initWasm()' failed". * * Plain fetches *are* authenticated in that context, so the glue is fetched here * and handed over as a blob: URL, which `import()` reads with no network request * at all, and the wasm bytes go across as `wasmBinary` so nothing else is looked * up either. Falls back to letting ORT fetch them itself, which is what works * everywhere other than that iframe. */ async function configureOrtWasm(baseUrl = './ort/') { const dir = new URL(baseUrl, document.baseURI).href; const get = async (name, as) => { const r = await fetch(dir + name); if (!r.ok) throw new Error(`${name}: HTTP ${r.status}`); return as === 'text' ? r.text() : r.arrayBuffer(); }; try { const [glue, binary] = await Promise.all([ get('ort-wasm-simd-threaded.jsep.mjs', 'text'), get('ort-wasm-simd-threaded.jsep.wasm'), ]); ort.env.wasm.wasmBinary = binary; ort.env.wasm.wasmPaths = { mjs: URL.createObjectURL(new Blob([glue], { type: 'text/javascript' })), }; } catch (e) { console.warn('could not preload the ORT wasm assets; letting ORT fetch them:', e); // Absolute, not './ort/': ORT resolves a relative wasmPaths against the // location of ort.all.min.js (already in ort/), which would look for // ort/ort/ort-wasm-simd-threaded.jsep.mjs and find no backend. ort.env.wasm.wasmPaths = dir; } } async function initEngine() { if (engine || isLoading) return engine; isLoading = true; try { // The runtime is a blocking classic script in index.html, so reaching here // without it means it was blocked (Edge Tracking Prevention, an extension, // a CSP). Say that plainly instead of throwing a bare ReferenceError. if (typeof ort === 'undefined') { throw new Error('the ONNX runtime failed to load (ort/ort.all.min.js) — ' + 'check for a blocked request in the network tab'); } if (!navigator.gpu) { gpuInfo.textContent = 'WebGPU not available — running on WASM (slower)'; } await configureOrtWasm(); // Threaded wasm needs cross-origin isolation (COOP *and* COEP). Hugging // Face Spaces send COOP but not COEP, so asking for threads there only // produces a console warning and a wasted init before ORT falls back. ort.env.wasm.numThreads = self.crossOriginIsolated ? (navigator.hardwareConcurrency || 4) : 1; setStatus('loading', 'Downloading model...'); showProgress(true, 'Downloading model...'); const progress = {}; engine = await CtcEngine.load({ baseUrl: MODEL_DIR, modelFile: MODEL_FILE, device: navigator.gpu ? 'webgpu' : 'wasm', onProgress: ({ url, loaded, total, cached }) => { progress[url] = { loaded, total, cached }; let l = 0, t = 0, allCached = true; for (const p of Object.values(progress)) { l += p.loaded; t += p.total; if (!p.cached) allCached = false; } if (allCached) { showProgress(true, 'Loading model from cache...'); return; } const pct = t > 0 ? (l / t) * 100 : 0; progressFill.style.width = `${pct}%`; showProgress(true, `Downloading model... ${(l / 1e6).toFixed(0)} / ${(t / 1e6).toFixed(0)} MB`); }, }); progressFill.style.width = '0%'; showProgress(false); // A wasm fallback is ~4x slower per window (1325 ms vs 305 ms for a 10 s // window), which is the difference between live and not, so say so // rather than letting it look like the model is just slow. gpuInfo.textContent = engine.device === 'webgpu' ? 'Backend: WebGPU' : 'Backend: WASM — WebGPU unavailable, so dictation will lag well behind speech'; setStatus('ready', 'Ready — start dictation or upload audio'); dictateBtn.disabled = false; audioFile.disabled = false; if (punctCheckbox.checked) startPunctuatorLoad(); return engine; } catch (e) { console.error('Model loading failed:', e); setStatus('error', `Error: ${e.message || e}`); showProgress(false); throw e; } finally { isLoading = false; } } /** * Load the punctuator (209 MB of its own). * * Kicked off in the background as soon as the CTC engine is ready, so it is * usually done before the user presses record -- loading it on the click would * put a ~30 s download between "start dictation" and hearing anything. Calls * share one promise, and only a caller that actually has to wait shows progress. */ function startPunctuatorLoad() { if (!punctuatorLoad && typeof window.loadPunctuator === 'function') { punctuatorLoad = window.loadPunctuator() .then(() => { punctuatorReady = true; }) .catch((e) => { console.warn('punctuator failed to load; transcript stays lowercase:', e); punctuatorReady = false; }); } return punctuatorLoad; } async function ensurePunctuator() { if (punctuatorReady || !punctCheckbox.checked) return punctuatorReady; const load = startPunctuatorLoad(); if (!load) return false; const stillLoading = !punctuatorReady; if (stillLoading) showProgress(true, 'Loading punctuation model...'); await load; if (stillLoading) showProgress(false); return punctuatorReady; } /** Punctuate committed text, or pass it through if the model is off/unavailable. */ async function punctuate(text) { if (!punctCheckbox.checked || !punctuatorReady) return text; return window.applyPunctuation(text); } // ------------------------------------------------------------ live dictation function renderTranscript({ committed, tentative }) { transcriptCard.style.display = 'block'; committedEl.textContent = committed; tentativeEl.textContent = tentative ? (committed ? ` ${tentative}` : tentative) : ''; const out = committedEl.parentElement; out.scrollTop = out.scrollHeight; const hasText = Boolean(committed || tentative); copyBtn.disabled = !hasText; downloadBtn.disabled = !hasText; } async function toggleDictation() { if (coordinator) { dictateBtn.disabled = true; setStatus('processing', 'Finishing…'); try { await coordinator.stop(); } finally { coordinator = null; dictateBtn.classList.remove('recording'); dictateBtn.querySelector('span').textContent = 'Start dictation'; dictateBtn.disabled = false; levelFill.style.width = '0%'; setStatus('ready', 'Ready — start dictation or upload audio'); } return; } await initEngine(); await ensurePunctuator(); coordinator = new StreamingCoordinator({ engine, onTranscript: renderTranscript, onLevel: (level) => { levelFill.style.width = `${(level * 100).toFixed(0)}%`; }, onStep: (info) => { if (info.kind === 'step') { statsEl.textContent = `${info.windowSec.toFixed(1)}s window · ${info.ms.toFixed(0)} ms · conf ${info.avgLogprob?.toFixed(2) ?? 'n/a'}`; } else if (info.kind === 'reject') { console.log(`[reject:${info.why}] "${info.text}" conf=${info.avgLogprob?.toFixed(2)} wps=${info.wps.toFixed(2)}`); } else { console.log(`[commit:${info.reason}] ${info.windowSec.toFixed(1)}s ${info.ms.toFixed(0)}ms conf=${info.avgLogprob?.toFixed(2)} "${info.text}"`); } }, punctuate, }); try { await coordinator.start(); dictateBtn.classList.add('recording'); dictateBtn.querySelector('span').textContent = 'Stop dictation'; setStatus('recording', 'Listening — speak naturally, pause between sentences'); } catch (e) { console.error('dictation failed to start:', e); coordinator = null; setStatus('error', e.name === 'NotAllowedError' ? 'Microphone access denied' : `Error: ${e.message || e}`); } } // --------------------------------------------------------- file transcription /** * Split a file into speech segments with the same energy gate the live path uses. * * This replaces the old Silero VAD download: the gate is already here, it is what * the streaming path is tuned against, and long audio has to be split anyway -- * a single pass over a 5-minute file would allocate a ~250 MB logits tensor * inside the graph. */ function segmentAudio(audio) { const gate = new SpeechGate(); const frame = 480; const segments = []; let start = null; let consumed = 0; gate.onEvent = (event) => { const t = consumed / SAMPLE_RATE; if (event === 'speechStarted') start = t; else if (start !== null) { segments.push({ start, end: t }); start = null; } }; for (let i = 0; i < audio.length; i += frame) { const chunk = audio.subarray(i, Math.min(i + frame, audio.length)); consumed = i + chunk.length; gate.feed(chunk); } if (start !== null) segments.push({ start, end: audio.length / SAMPLE_RATE }); if (!segments.length) return [{ start: 0, end: audio.length / SAMPLE_RATE }]; // Pad each segment slightly (the gate's onset debounce trims the first // phoneme) and cap the length. const out = []; for (const seg of segments) { let from = Math.max(0, seg.start - 0.2); const to = Math.min(audio.length / SAMPLE_RATE, seg.end + 0.2); while (to - from > MAX_SEGMENT_SEC) { out.push({ start: from, end: from + MAX_SEGMENT_SEC }); from += MAX_SEGMENT_SEC; } out.push({ start: from, end: to }); } return out; } async function transcribeFile() { if (!currentAudioData) return; await initEngine(); await ensurePunctuator(); setStatus('processing', 'Transcribing…'); transcriptCard.style.display = 'block'; const segments = segmentAudio(currentAudioData); const parts = []; let totalMs = 0; for (let i = 0; i < segments.length; i++) { const seg = segments[i]; showProgress(true, `Segment ${i + 1} / ${segments.length}`); progressFill.style.width = `${((i + 1) / segments.length) * 100}%`; const slice = currentAudioData.subarray( Math.floor(seg.start * SAMPLE_RATE), Math.floor(seg.end * SAMPLE_RATE), ); const result = await engine.transcribe(slice); totalMs += result.ms; if (result.text) { parts.push(await punctuate(result.text)); renderTranscript({ committed: parts.join('\n\n'), tentative: '' }); } } showProgress(false); progressFill.style.width = '0%'; const audioSec = currentAudioData.length / SAMPLE_RATE; statsEl.textContent = `${audioSec.toFixed(1)}s audio · ${segments.length} segment(s) · ${totalMs.toFixed(0)} ms · ${(audioSec / (totalMs / 1000)).toFixed(0)}x realtime`; if (!parts.length) renderTranscript({ committed: '(no speech detected)', tentative: '' }); setStatus('ready', 'Transcription complete'); } /** Decode any audio file to 16 kHz mono float32. */ async function loadAudioFile(file) { setStatus('processing', 'Decoding audio…'); try { const audioCtx = new AudioContext({ sampleRate: SAMPLE_RATE }); const buffer = await audioCtx.decodeAudioData(await file.arrayBuffer()); let audio = buffer.getChannelData(0); if (buffer.numberOfChannels > 1) { const mixed = new Float32Array(audio.length); for (let c = 0; c < buffer.numberOfChannels; c++) { const ch = buffer.getChannelData(c); for (let i = 0; i < mixed.length; i++) mixed[i] += ch[i] / buffer.numberOfChannels; } audio = mixed; } // decodeAudioData already resampled to the context rate. await audioCtx.close(); currentAudioData = audio; await transcribeFile(); } catch (e) { console.error('audio decode failed:', e); setStatus('error', `Could not read that file: ${e.message || e}`); } } // -------------------------------------------------------------- transcript IO function transcriptText() { return committedEl.textContent + (tentativeEl.textContent || ''); } async function copyTranscript() { await navigator.clipboard.writeText(transcriptText()); const original = copyBtn.title; copyBtn.title = 'Copied'; setTimeout(() => { copyBtn.title = original; }, 1200); } function downloadTranscript() { const blob = new Blob([transcriptText()], { type: 'text/plain' }); const a = document.createElement('a'); a.href = URL.createObjectURL(blob); a.download = `transcript-${new Date().toISOString().slice(0, 19).replace(/[:T]/g, '-')}.txt`; a.click(); URL.revokeObjectURL(a.href); } function clearTranscript() { renderTranscript({ committed: '', tentative: '' }); transcriptCard.style.display = 'none'; statsEl.textContent = ''; if (coordinator) coordinator.paragraphs = []; } // ---------------------------------------------------------------------- wiring dictateBtn.addEventListener('click', toggleDictation); audioFile.addEventListener('change', (e) => { const file = e.target.files[0]; if (file) loadAudioFile(file); e.target.value = ''; }); copyBtn.addEventListener('click', copyTranscript); downloadBtn.addEventListener('click', downloadTranscript); clearBtn.addEventListener('click', clearTranscript); // Drag and drop onto the input card for (const [event, handler] of [ ['dragover', (e) => { e.preventDefault(); inputCard.classList.add('drag-over'); }], ['dragleave', () => inputCard.classList.remove('drag-over')], ['drop', (e) => { e.preventDefault(); inputCard.classList.remove('drag-over'); const file = e.dataTransfer.files[0]; if (file?.type.startsWith('audio/')) loadAudioFile(file); else setStatus('error', 'Please drop an audio file'); }], ]) { inputCard.addEventListener(event, handler); fileTile.addEventListener(event, (e) => e.stopPropagation(), { capture: false }); } punctCheckbox.addEventListener('change', () => { if (punctCheckbox.checked) ensurePunctuator(); }); // Space bar toggles dictation, unless a control has focus. document.addEventListener('keydown', (e) => { if (e.code !== 'Space' || e.repeat) return; if (['INPUT', 'TEXTAREA', 'SELECT', 'BUTTON'].includes(document.activeElement?.tagName)) return; e.preventDefault(); toggleDictation(); }); setStatus('loading', 'Loading…'); initEngine().catch(() => {}); })(); })();