gsaon's picture
Load the capture worklet from a blob URL too; stop probing module imports the app no longer uses
c9d8ae0 verified
Raw History Blame Contribute Delete
70.9 kB
/*
* GENERATED by build_bundle.py -- do not edit.
*
* The app's ES modules concatenated into one classic script. A private
* Space's cross-site iframe does not authenticate the module fetches issued
* while the page is parsing (they 401, and the module map then caches that
* failure), whereas classic scripts always carry the cookie. Edit the
* sources listed below and re-run build_bundle.py.
*
* Sources, in dependency order: ctc-frontend.js, ctc-tokenizer.js, ctc-engine.js, ctc-streaming.js, app.js
*/
(() => {
'use strict';
// Stands in for the module namespace: each module destructures what it
// imports off this and assigns its exports back onto it.
const __M = {};
// ==== ctc-frontend.js ===================================================
(() => {
/**
* Audio front-end for the FastConformer CTC model.
*
* A port of the model's `CtcConformerProcessor._frontend` (log-mel + deltas +
* frame stacking) plus the sidecar's `_agc`. The mel filterbank and the STFT
* window are NOT recomputed here -- they are loaded from `mel_filters.bin` /
* `stft_window.bin`, dumped by export_ctc_onnx.py straight out of the torchaudio
* objects the Python path uses. Reimplementing torchaudio's htk mel scale
* (norm=None) and torch.stft's window centring in JS is the most likely place for
* a silent mismatch, and a front-end that is subtly wrong degrades WER without
* ever raising an error.
*
* Verified bit-for-bit against the Python features by test-frontend.mjs.
*/
const EPS = 1e-10;
/** In-place iterative radix-2 complex FFT (Cooley-Tukey, decimation in time). */
class FFT {
constructor(size) {
this.size = size;
this.levels = Math.log2(size) | 0;
if (2 ** this.levels !== size) throw new Error(`FFT size ${size} is not a power of two`);
// Bit-reversal permutation and twiddle tables, precomputed once: the
// front-end runs this transform ~100 times per second of audio.
this.rev = new Uint32Array(size);
for (let i = 0; i < size; i++) {
let r = 0;
for (let b = 0; b < this.levels; b++) r |= ((i >> b) & 1) << (this.levels - 1 - b);
this.rev[i] = r;
}
this.cos = new Float32Array(size / 2);
this.sin = new Float32Array(size / 2);
for (let i = 0; i < size / 2; i++) {
this.cos[i] = Math.cos((-2 * Math.PI * i) / size);
this.sin[i] = Math.sin((-2 * Math.PI * i) / size);
}
}
/** Transform `re`/`im` in place (both length `size`). */
run(re, im) {
const { size, rev, cos, sin } = this;
for (let i = 0; i < size; i++) {
const j = rev[i];
if (j > i) {
let t = re[i]; re[i] = re[j]; re[j] = t;
t = im[i]; im[i] = im[j]; im[j] = t;
}
}
for (let len = 2; len <= size; len <<= 1) {
const half = len >> 1;
const step = size / len;
for (let i = 0; i < size; i += len) {
for (let j = 0, k = 0; j < half; j++, k += step) {
const c = cos[k], s = sin[k];
const a = i + j, b = a + half;
const tre = re[b] * c - im[b] * s;
const tim = re[b] * s + im[b] * c;
re[b] = re[a] - tre; im[b] = im[a] - tim;
re[a] += tre; im[a] += tim;
}
}
}
}
}
/**
* Centred moving average, cumsum-based (O(n)).
*
* Mirrors the sidecar's `_moving_avg`, including its edge handling: the valid
* region is padded by replicating the first/last valid value, with the extra
* sample going on the right when the window is even.
*/
function movingAvg(v, win) {
const n = v.length;
win = Math.max(1, win | 0);
if (win <= 1 || n <= 1) return v;
win = Math.min(win, n);
// float64 accumulator: a float32 cumsum over ~10 s of samples loses enough
// precision to shift the gain envelope (the Python side uses float64 too).
const csum = new Float64Array(n + 1);
for (let i = 0; i < n; i++) csum[i + 1] = csum[i] + v[i];
const valid = n - win + 1;
const padL = (win - 1) >> 1;
const out = new Float32Array(n);
const first = (csum[win] - csum[0]) / win;
const last = (csum[n] - csum[n - win]) / win;
for (let i = 0; i < n; i++) {
const k = i - padL;
out[i] = k < 0 ? first : k >= valid ? last : (csum[k + win] - csum[k]) / win;
}
return out;
}
/**
* Automatic gain control -- port of the sidecar's `_agc`.
*
* Not cosmetic: the CTC front-end normalizes log-mel against the clip's *global
* peak*, so a quiet tail 20-40 dB below a loud head gets floored and silently
* dropped. A smoothed moving-RMS gain toward `target` equalizes the levels.
*/
function agc(x, { sampleRate = 16000, target = 0.12, winMs = 150, maxGain = 20 } = {}) {
const n = x.length;
if (n === 0) return x;
const win = Math.max(1, Math.round((sampleRate * winMs) / 1000));
const sq = new Float32Array(n);
for (let i = 0; i < n; i++) sq[i] = x[i] * x[i];
const ms = movingAvg(sq, win);
const g = new Float32Array(n);
for (let i = 0; i < n; i++) {
const env = Math.sqrt(ms[i] + 1e-9);
g[i] = Math.min(Math.max(target / Math.max(env, 1e-4), 0), maxGain);
}
const gs = movingAvg(g, win); // smooth the gain to avoid pumping
const y = new Float32Array(n);
let peak = 0;
for (let i = 0; i < n; i++) {
y[i] = x[i] * gs[i];
const a = Math.abs(y[i]);
if (a > peak) peak = a;
}
if (peak > 0.97) {
const k = 0.97 / peak;
for (let i = 0; i < n; i++) y[i] *= k;
}
return y;
}
class CtcFrontend {
constructor(config, melFilters, window) {
this.config = config;
this.melFilters = melFilters; // [n_mels][n_freqs], mel-major
this.window = window; // n_fft, already centred by the exporter
this.fft = new FFT(config.n_fft);
this.nFreqs = config.n_fft / 2 + 1;
}
static async load(baseUrl = './onnx-ctc/') {
const get = async (name, as) => {
const r = await fetch(baseUrl + name);
if (!r.ok) throw new Error(`front-end asset ${name}: HTTP ${r.status}`);
return as === 'json' ? r.json() : new Float32Array(await r.arrayBuffer());
};
const [config, melFilters, window] = await Promise.all([
get('frontend_config.json', 'json'),
get('mel_filters.bin'),
get('stft_window.bin'),
]);
return new CtcFrontend(config, melFilters, window);
}
/** Feature frames a clip of `nSamples` produces (before padding to pad_multiple). */
frameCount(nSamples) {
const s = this.config.stack_factor;
const melFrames = Math.floor(nSamples / this.config.hop_length);
return s * Math.ceil(melFrames / s);
}
/**
* waveform (Float32Array, 16 kHz mono) -> stacked log-mel(+delta) features.
*
* Returns `{ data, frames, dim }` where `data` is [frames, dim] row-major,
* ready to become the model's `input_features` tensor.
*/
compute(wav) {
const { hop_length: hop, n_fft: nFft, n_mels: nMels, stack_factor: stack,
logmel_floor_db: floorDb, deltas: useDeltas } = this.config;
const nFrames = this.frameCount(wav.length);
if (nFrames === 0) return { data: new Float32Array(0), frames: 0, dim: this.config.input_dim };
// The processor right-pads the waveform so the trailing partial stack
// group is filled rather than dropped, then slices to exactly nFrames.
const need = (nFrames - 1) * hop + 1;
let x = wav;
if (x.length < need) {
x = new Float32Array(need);
x.set(wav);
}
// torch.stft(center=True, pad_mode='reflect'): reflect-pad by n_fft/2 so
// frame i is centred on sample i*hop.
const half = nFft >> 1;
const padded = new Float32Array(x.length + nFft);
padded.set(x, half);
for (let i = 0; i < half; i++) {
padded[half - 1 - i] = x[Math.min(i + 1, x.length - 1)];
padded[half + x.length + i] = x[Math.max(x.length - 2 - i, 0)];
}
// Mel spectrogram, mel-major ([nMels][nFrames]) because deltas run along time.
const mel = new Float32Array(nMels * nFrames);
const re = new Float32Array(nFft);
const im = new Float32Array(nFft);
const power = new Float32Array(this.nFreqs);
for (let t = 0; t < nFrames; t++) {
const off = t * hop;
for (let i = 0; i < nFft; i++) {
re[i] = padded[off + i] * this.window[i];
im[i] = 0;
}
this.fft.run(re, im);
for (let k = 0; k < this.nFreqs; k++) power[k] = re[k] * re[k] + im[k] * im[k];
for (let m = 0; m < nMels; m++) {
const fb = m * this.nFreqs;
let acc = 0;
for (let k = 0; k < this.nFreqs; k++) acc += this.melFilters[fb + k] * power[k];
mel[m * nFrames + t] = acc;
}
}
// log10 with a floor relative to the clip's global peak, then the /4 +1
// scaling the model was trained with.
let mx = -Infinity;
for (let i = 0; i < mel.length; i++) {
const v = Math.log10(Math.max(mel[i], EPS));
mel[i] = v;
if (v > mx) mx = v;
}
const floor = mx - floorDb;
for (let i = 0; i < mel.length; i++) mel[i] = Math.max(mel[i], floor) / 4 + 1;
// Deltas: torchaudio.functional.compute_deltas(win_length=3) is
// (x[t+1] - x[t-1]) / 2 with replicate edge padding.
const nCh = useDeltas ? nMels * 2 : nMels;
const chans = new Float32Array(nCh * nFrames);
chans.set(mel);
if (useDeltas) {
for (let m = 0; m < nMels; m++) {
const src = m * nFrames;
const dst = (nMels + m) * nFrames;
for (let t = 0; t < nFrames; t++) {
const prev = mel[src + Math.max(t - 1, 0)];
const next = mel[src + Math.min(t + 1, nFrames - 1)];
chans[dst + t] = (next - prev) / 2;
}
}
}
// Stack: (nCh, nFrames) -> (nFrames/stack, stack*nCh), i.e. row j is the
// channels of frame stack*j followed by those of frame stack*j+1.
const outFrames = nFrames / stack;
const dim = stack * nCh;
const data = new Float32Array(outFrames * dim);
for (let j = 0; j < outFrames; j++) {
for (let k = 0; k < stack; k++) {
const t = j * stack + k;
const base = j * dim + k * nCh;
for (let c = 0; c < nCh; c++) data[base + c] = chans[c * nFrames + t];
}
}
return { data, frames: outFrames, dim };
}
}
Object.assign(__M, { agc, CtcFrontend });
})();
// ==== ctc-tokenizer.js ==================================================
(() => {
/**
* Decoder for the model's 16384-entry ByteLevel BPE vocabulary.
*
* Only decoding is needed (the model never encodes text), so export_ctc_onnx.py
* dumps the id->piece slice that can reach the decoder as vocab.json (~170 KB)
* instead of shipping the whole tokenizer.json.
*
* granite-speech-4.2-470m-turboctc uses a GPT-2 style **ByteLevel** tokenizer:
* every byte 0-255 is represented by one printable codepoint (space is `Ġ`), so a
* piece is a string of byte-stand-ins rather than text, and one UTF-8 character
* may be split across several pieces ("é" arrives as "Ã" + "©"). Decoding is
* therefore: map each character back to its byte, concatenate, then UTF-8 decode
* once at the end -- never per piece, or multi-byte characters break.
*
* (The earlier C-T26.18 build used a SentencePiece tokenizer whose decoder was
* Replace(U+2581 -> space) / ByteFallback / Fuse / Strip. That is a different
* algorithm and would silently produce mojibake here, so the exporter asserts the
* tokenizer's decoder type matches this implementation.)
*/
/**
* GPT-2's byte <-> codepoint table, inverted to codepoint -> byte.
*
* The 188 bytes that are already printable map to themselves; the remaining 68
* are moved up to U+0100 and beyond so no piece ever contains a control
* character or a bare space.
*/
function byteLevelCharToByte() {
const bytes = [];
for (let b = 0x21; b <= 0x7e; b++) bytes.push(b); // !..~
for (let b = 0xa1; b <= 0xac; b++) bytes.push(b); // ¡..¬
for (let b = 0xae; b <= 0xff; b++) bytes.push(b); // ®..ÿ
const codepoints = bytes.slice();
const printable = new Set(bytes);
let next = 0;
for (let b = 0; b < 256; b++) {
if (!printable.has(b)) {
bytes.push(b);
codepoints.push(256 + next);
next++;
}
}
const map = new Map();
for (let i = 0; i < bytes.length; i++) map.set(codepoints[i], bytes[i]);
return map;
}
class CtcDecoder {
constructor(pieces, numSpecialTokens) {
this.numSpecialTokens = numSpecialTokens;
const charToByte = byteLevelCharToByte();
// Pre-resolve every piece to its bytes, so decoding a frame of ids is a
// few array copies rather than per-call string work.
this.encoded = new Array(pieces.length);
for (let i = 0; i < pieces.length; i++) {
const piece = pieces[i];
const bytes = new Uint8Array(piece.length);
let n = 0;
for (const ch of piece) {
const b = charToByte.get(ch.codePointAt(0));
// A codepoint outside the table means this vocab was not produced
// by a ByteLevel tokenizer; skip rather than emit a wrong byte.
if (b !== undefined) bytes[n++] = b;
}
this.encoded[i] = bytes.subarray(0, n);
}
this.utf8Decoder = new TextDecoder('utf-8');
}
static async load(baseUrl = './onnx-ctc/') {
const [vocab, config] = await Promise.all([
fetch(baseUrl + 'vocab.json').then((r) => {
if (!r.ok) throw new Error(`vocab.json: HTTP ${r.status}`);
return r.json();
}),
fetch(baseUrl + 'frontend_config.json').then((r) => {
if (!r.ok) throw new Error(`frontend_config.json: HTTP ${r.status}`);
return r.json();
}),
]);
return new CtcDecoder(vocab, config.num_special_tokens);
}
/** Decode content ids (all >= numSpecialTokens) to text. */
decode(ids) {
let total = 0;
for (const id of ids) {
const e = this.encoded[id - this.numSpecialTokens];
if (e) total += e.length;
}
const buf = new Uint8Array(total);
let at = 0;
for (const id of ids) {
const e = this.encoded[id - this.numSpecialTokens];
if (!e) continue; // out-of-range id: skip rather than poison the string
buf.set(e, at);
at += e.length;
}
return this.utf8Decoder.decode(buf);
}
}
/**
* Greedy CTC collapse: drop repeats, then the blank, then any remaining
* below-content ids.
*
* Order matters and matches the Python path: repeats are collapsed against the
* *raw* previous id (blanks included), which is what lets a blank between two
* identical tokens preserve both. For this model `numSpecialTokens` is 1 -- the
* CTC blank at id 0 is the only non-text id, and everything above it is a
* tokenizer id as-is.
*/
function ctcCollapse(ids, { blankId = 0, numSpecialTokens = 1 } = {}) {
const out = [];
let prev = -1;
for (let i = 0; i < ids.length; i++) {
const id = ids[i];
if (id !== prev && id !== blankId && id >= numSpecialTokens) out.push(id);
prev = id;
}
return out;
}
Object.assign(__M, { CtcDecoder, ctcCollapse });
})();
// ==== ctc-engine.js =====================================================
(() => {
const { CtcFrontend, agc, CtcDecoder, ctcCollapse } = __M;
/**
* FastConformer CTC inference engine (onnxruntime-web, WebGPU).
*
* The model is a custom `ctc_conformer` architecture with no transformers.js
* support, so this drives an ONNX session directly: features -> one forward pass
* -> greedy CTC collapse. There is no autoregressive loop, no KV cache and no
* chat template; a window is one `session.run`.
*
* Two contracts come from export_ctc_onnx.py and must hold here:
*
* - **Frames are padded to `pad_multiple`.** The encoder chunks attention into
* context_size (128) blocks at every layer and subsamples time by 4, so a
* frame count with no partial block anywhere is a multiple of 512 (10.24 s).
* The exported graph therefore has no tail-block path, and short windows are
* padded up with `padding_mask` marking the pad. Measured cost of the masked
* padding on granite-speech-4.2-470m-turboctc: none -- all 14 fixture clips
* decode identically to the unpadded path, with zero per-frame argmax flips.
* (Its encoder masks inside the conv module specifically so a padded frame
* cannot leak through the stride-2 kernel into the last valid frames.)
*
* - **The graph returns ids, not logits.** `[1, T/4, 16384]` float32 logits
* would be an ~8 MB device-to-host copy per window; argmax and the
* log-softmax max happen inside the graph instead, so a window costs ~1 KB of
* output. `top_logprob` is the confidence signal the streaming gate uses.
*/
const CACHE_NAME = 'granite-ctc-models-v1';
/**
* Cache-first fetch, so a reload doesn't re-download hundreds of MB.
*
* The body is streamed to completion first and only then handed to the cache.
* Doing it the other way round -- `await cache.put(response.clone())` before
* reading -- makes the tee buffer the entire 312 MB while the cache write drains
* it, so the progress callback fires nothing for minutes and the UI looks hung.
* A failed cache write (quota, private mode) must not fail the load either: the
* bytes are already in hand, and the only cost is re-downloading next time.
*/
async function cachedFetch(url, onProgress) {
const cache = await caches.open(CACHE_NAME).catch(() => null);
const hit = await cache?.match(url).catch(() => null);
if (hit) {
onProgress?.({ url, loaded: 1, total: 1, cached: true });
return hit.arrayBuffer();
}
const response = await fetch(url);
if (!response.ok) throw new Error(`${url}: HTTP ${response.status}`);
const total = Number(response.headers.get('content-length')) || 0;
let buffer;
if (!response.body) {
buffer = await response.arrayBuffer();
} else {
const reader = response.body.getReader();
const chunks = [];
let loaded = 0;
for (;;) {
const { done, value } = await reader.read();
if (done) break;
chunks.push(value);
loaded += value.length;
onProgress?.({ url, loaded, total });
}
const bytes = new Uint8Array(loaded);
let at = 0;
for (const c of chunks) { bytes.set(c, at); at += c.length; }
buffer = bytes.buffer;
}
if (cache) {
try {
await cache.put(url, new Response(buffer.slice(0), {
headers: { 'Content-Type': 'application/octet-stream', 'Content-Length': String(buffer.byteLength) },
}));
} catch (e) {
console.warn(`could not cache ${url} (will re-download next time):`, e);
}
}
return buffer;
}
class CtcEngine {
constructor({ session, frontend, decoder, config, device }) {
this.session = session;
this.frontend = frontend;
this.decoder = decoder;
this.config = config;
this.device = device;
}
/**
* Create the session and load the front-end assets.
*
* `device: 'webgpu'` falls back to wasm if the adapter or a kernel is
* missing -- a CTC pass is still usable on wasm, just slower.
*/
static async load({
baseUrl = './onnx-ctc/',
modelFile = 'ctc_conformer_q4f16-b16.onnx',
device = 'webgpu',
onProgress = null,
allowFallback = true,
} = {}) {
const [frontend, decoder] = await Promise.all([
CtcFrontend.load(baseUrl),
CtcDecoder.load(baseUrl),
]);
const modelUrl = baseUrl + modelFile;
// The exporter writes weights to `<model>.onnx.data` and records that
// basename inside the graph, so the session's externalData path must
// match it exactly.
const dataName = `${modelFile}.data`;
const [modelBuffer, dataBuffer] = await Promise.all([
cachedFetch(modelUrl, onProgress),
cachedFetch(baseUrl + dataName, onProgress),
]);
const options = {
executionProviders: [device],
externalData: [{ path: dataName, data: dataBuffer }],
graphOptimizationLevel: 'all',
};
let session;
let used = device;
let fallbackError = null;
try {
session = await ort.InferenceSession.create(modelBuffer, options);
} catch (e) {
if (device !== 'webgpu' || !allowFallback) throw e;
console.warn('WebGPU session failed, falling back to wasm:', e);
fallbackError = String(e?.message || e);
used = 'wasm';
session = await ort.InferenceSession.create(modelBuffer, {
...options,
executionProviders: ['wasm'],
});
}
const engine = new CtcEngine({ session, frontend, decoder, config: frontend.config, device: used });
// A WebGPU kernel can also fail on its *first run* rather than at session
// build (ORT-web reports unsupported MatMulNBits shapes that way), so the
// warmup pass is where a fallback is really decided.
const warmupError = await engine.warmup();
if (warmupError && used === 'webgpu') {
if (!allowFallback) throw new Error(`WebGPU warmup failed: ${warmupError}`);
console.warn('WebGPU warmup failed, falling back to wasm:', warmupError);
fallbackError = warmupError;
engine.device = 'wasm';
engine.session = await ort.InferenceSession.create(modelBuffer, {
...options,
executionProviders: ['wasm'],
});
await engine.warmup();
}
engine.fallbackError = fallbackError;
return engine;
}
/**
* Run one throwaway pass so shader compilation doesn't land on the user's
* first spoken window (measured on WebGPU: 1271 ms first call vs 295 ms
* steady state). Every window is padded to the same frame count, so a single
* warmup covers the shapes the streaming loop will use.
*/
async warmup() {
try {
const frames = this.config.pad_multiple;
await this.session.run({
input_features: new ort.Tensor('float32', new Float32Array(frames * this.config.input_dim), [1, frames, this.config.input_dim]),
padding_mask: new ort.Tensor('float32', new Float32Array(frames), [1, frames]),
});
return null;
} catch (e) {
return String(e?.message || e);
}
}
/** Encoder output frames covering `frames` of real audio (both subsample blocks floor-halve). */
static realOutFrames(frames) {
return Math.floor(Math.floor(frames / 2) / 2);
}
/**
* Transcribe a waveform (Float32Array, 16 kHz mono).
*
* `applyAgc` mirrors the sidecar: the log-mel front-end normalizes against
* the clip's global peak, so without it a quiet tail after a loud passage is
* floored away and its words are silently dropped.
*/
async transcribe(wav, { applyAgc = true } = {}) {
const t0 = performance.now();
const audio = applyAgc ? agc(wav, { sampleRate: this.config.sample_rate }) : wav;
const { data, frames, dim } = this.frontend.compute(audio);
if (frames === 0) return { text: '', words: [], avgLogprob: null, frames: 0, ms: 0 };
const multiple = this.config.pad_multiple;
const padded = Math.ceil(frames / multiple) * multiple;
const features = new Float32Array(padded * dim);
features.set(data);
const mask = new Float32Array(padded).fill(1);
mask.fill(0, 0, frames);
const outputs = await this.session.run({
input_features: new ort.Tensor('float32', features, [1, padded, dim]),
padding_mask: new ort.Tensor('float32', mask, [1, padded]),
});
// The graph emits padded/4 frames; only those covering real audio may
// contribute tokens, so trim before collapsing.
const nOut = CtcEngine.realOutFrames(frames);
const ids = outputs.ids.data.subarray(0, nOut);
const logp = outputs.top_logprob.data.subarray(0, nOut);
const content = ctcCollapse(ids, {
blankId: this.config.blank_id,
numSpecialTokens: this.config.num_special_tokens,
});
const text = this.decoder.decode(content).trim();
// Confidence: mean top log-probability over non-blank frames. The CTC
// analogue of a mean per-token logprob -- clean speech sits near -0.03,
// silence/noise confabulation near -1.85 and below.
let sum = 0, n = 0;
for (let i = 0; i < nOut; i++) {
if (ids[i] !== this.config.blank_id) { sum += logp[i]; n++; }
}
return {
text,
words: text.length ? text.split(/\s+/) : [],
avgLogprob: n ? sum / n : null,
frames,
paddedFrames: padded,
ms: performance.now() - t0,
};
}
}
Object.assign(__M, { CtcEngine });
})();
// ==== ctc-streaming.js ==================================================
(() => {
/**
* Live dictation: growing window + LocalAgreement-2, ported from the macOS
* GraniteLiveDictation app (SpeechGate / Reconciler / StreamingCoordinator).
*
* continuous 16 kHz capture -> rolling window + energy SpeechGate
* every `step` while speech is in the window:
* window = audio[segment start .. now] (grows; overlaps the last)
* hypo = words(CTC transcription of window)
* a word COMMITS once two consecutive windows agree on it (longest
* common prefix); everything past that is TENTATIVE (shown grey)
* on a pause: one clean full-window pass commits the segment and re-anchors
*
* The window *grows* rather than sliding because the model returns text with no
* word timestamps -- there is no safe place to trim a fixed window without
* risking a mid-word cut. Anchoring each window at the start of the current
* speech segment makes reconciliation a plain prefix comparison and keeps the
* window bounded (a phrase between pauses).
*
* The tuning constants are the ones the macOS app converged on against real
* far-field recordings, not defaults -- see the comments on each.
*/
const SAMPLE_RATE = 16000;
const INT16_SCALE = 32768; // gate thresholds are in Int16 RMS units (as tuned)
/**
* URL to hand `audioWorklet.addModule()`.
*
* addModule fetches its argument as a *module*, which is the one request type a
* private Space's cross-site iframe fails to authenticate — measured there, the
* same file returns 200 to a plain fetch and intermittently fails as a module
* import. That would break dictation at the moment the user presses record, well
* after everything else had loaded cleanly. So fetch the source (plain fetches
* work) and hand addModule a blob: URL, which needs no network request at all.
* Falls back to the direct URL, which is what works everywhere else.
*
* Resolved against this module rather than the document so it survives being
* served from a subdirectory; build_bundle.py rewrites `document.baseURI` to
* `document.baseURI` for the bundled build, where the two are the same directory.
*/
let cachedWorkletUrl = null;
async function captureWorkletUrl() {
if (cachedWorkletUrl) return cachedWorkletUrl;
const direct = new URL('./ctc-capture-worklet.js', document.baseURI).href;
try {
const r = await fetch(direct);
if (!r.ok) throw new Error(`HTTP ${r.status}`);
const source = await r.text();
cachedWorkletUrl = URL.createObjectURL(new Blob([source], { type: 'text/javascript' }));
} catch (e) {
console.warn('could not preload the capture worklet; using its URL directly:', e);
cachedWorkletUrl = direct;
}
return cachedWorkletUrl;
}
/**
* Energy speech/silence gate with an adaptive noise floor.
*
* Not a neural VAD -- an RMS gate whose enter/exit thresholds float with the
* measured background level. That matters for its one job: finding a *pause* so
* the growing window can flush. With fixed thresholds, a room whose ambient RMS
* sits above the silence cutoff never produces a silence frame, the endpoint
* timer never fires, and every window runs to the maxWindow cap and cuts
* mid-sentence.
*/
class SpeechGate {
constructor(config = {}) {
this.config = {
endpointSilenceMs: 600, // contiguous below-exit audio that ends a segment
frameMs: 30,
// Biased toward inclusion: this CTC model doesn't confabulate on
// silence the way an autoregressive decoder does (it decodes to empty
// / low confidence, and the confidence gate backstops it), so
// admitting soft audio costs nothing -- while excluding it dropped
// real words (soft onsets never triggered, soft tails got endpointed
// mid-phrase).
onRatio: 1.5,
minEnter: 80,
// Exit stays near the original: dropping it toward ambient stops
// pause detection altogether and every segment runs to maxWindow,
// which wrecks paragraphing.
offRatio: 1.4,
minExit: 60,
onFrames: 2, // debounce transient clicks
// Seeded from the *mean* RMS over the startup window, not the min: the
// min tracks the quietest valley of fluctuating noise and seeds the
// floor far below the ambient body, which then lets noise read as
// speech.
calibrationMs: 360,
floorAdapt: 0.05, // per-frame EMA; at 30 ms frames ~0.6 s to settle
initialNoiseFloor: 60,
maxNoiseFloor: 4000,
...config,
};
this.frameLen = Math.max(1, (SAMPLE_RATE * this.config.frameMs) / 1000);
this.calibrationFrames = Math.floor(this.config.calibrationMs / this.config.frameMs);
this.onEvent = null;
this.isSpeaking = false;
this.noiseFloor = this.config.initialNoiseFloor;
this.silenceMs = 0;
this.onCount = 0;
this.pending = [];
this.pendingLen = 0;
this.calibLeft = this.calibrationFrames;
this.calibSum = 0;
this.calibCount = 0;
}
get enterThreshold() { return Math.max(this.config.minEnter, this.noiseFloor * this.config.onRatio); }
get exitThreshold() { return Math.max(this.config.minExit, this.noiseFloor * this.config.offRatio); }
/**
* Clear per-utterance state. The noise floor and calibration are deliberately
* preserved: they describe the *room*, not the utterance, and this is called
* on every commit (including maxWindow cuts taken mid-speech), where
* re-seeding would calibrate the floor from speech energy and blind the gate.
*/
reset() {
this.isSpeaking = false;
this.silenceMs = 0;
this.onCount = 0;
this.pending = [];
this.pendingLen = 0;
}
/** Re-arm calibration -- for a fresh capture session / new environment. */
recalibrate() {
this.noiseFloor = this.config.initialNoiseFloor;
this.calibLeft = this.calibrationFrames;
this.calibSum = 0;
this.calibCount = 0;
}
/** Feed a chunk of float samples (any length); re-framed into fixed blocks. */
feed(samples) {
if (!samples.length) return;
this.pending.push(samples);
this.pendingLen += samples.length;
while (this.pendingLen >= this.frameLen) {
const frame = this.takeFrame();
// process() emits events synchronously and a handler may call reset()
// re-entrantly (which drops `pending`), so the frame is removed first.
this.process(frame);
}
}
takeFrame() {
const frame = new Float32Array(this.frameLen);
let at = 0;
while (at < this.frameLen) {
const head = this.pending[0];
const take = Math.min(head.length, this.frameLen - at);
frame.set(take === head.length ? head : head.subarray(0, take), at);
at += take;
if (take === head.length) this.pending.shift();
else this.pending[0] = head.subarray(take);
}
this.pendingLen -= this.frameLen;
return frame;
}
process(frame) {
const rms = rmsInt16(frame);
if (this.calibLeft > 0) {
this.calibLeft--;
this.calibSum += rms;
this.calibCount++;
this.noiseFloor = Math.min(Math.max(this.calibSum / this.calibCount, 1), this.config.maxNoiseFloor);
return;
}
this.adaptNoise(rms);
if (this.isSpeaking) {
if (rms < this.exitThreshold) {
this.silenceMs += this.config.frameMs;
if (this.silenceMs >= this.config.endpointSilenceMs) {
this.isSpeaking = false;
this.silenceMs = 0;
this.onCount = 0;
this.onEvent?.('speechEnded');
}
} else {
this.silenceMs = 0; // speech resumed; reset the endpoint timer
}
} else if (rms >= this.enterThreshold) {
this.onCount++;
if (this.onCount >= this.config.onFrames) {
this.isSpeaking = true;
this.silenceMs = 0;
this.onCount = 0;
this.onEvent?.('speechStarted');
}
} else {
this.onCount = 0;
}
}
/**
* Learn the floor from ambient frames only, tracking the *body* (mean) of the
* ambient level. Updating solely when not in a segment AND below the enter
* threshold keeps sustained speech (or a speech attack) from dragging the
* floor up to speech level.
*/
adaptNoise(rms) {
if (this.isSpeaking || rms >= this.enterThreshold) return;
const a = this.config.floorAdapt;
this.noiseFloor = Math.min(Math.max((1 - a) * this.noiseFloor + a * rms, 1), this.config.maxNoiseFloor);
}
}
function rmsInt16(samples) {
if (!samples.length) return 0;
let acc = 0;
for (let i = 0; i < samples.length; i++) acc += samples[i] * samples[i];
return Math.sqrt(acc / samples.length) * INT16_SCALE;
}
/**
* LocalAgreement-2 reconciliation.
*
* Each step transcribes the growing window and yields a fresh hypothesis. Because
* the window overlaps the previous one, the two should agree on the stable part
* and differ only at the unfinished tail. A word is committed once two
* *consecutive* hypotheses agree on it (their longest common prefix, compared
* case-insensitively and ignoring edge punctuation); everything past that is
* tentative and may still change.
*/
class Reconciler {
constructor() {
this.confirmed = [];
this.prevHypo = [];
}
update(hypothesis) {
const agreed = commonPrefixCount(this.prevHypo, hypothesis);
if (agreed > this.confirmed.length) {
this.confirmed.push(...hypothesis.slice(this.confirmed.length, agreed));
}
this.prevHypo = hypothesis;
const tentative = hypothesis.length > this.confirmed.length
? hypothesis.slice(this.confirmed.length)
: [];
return { confirmed: this.confirmed, tentative };
}
/** End of segment: a clean full-window hypothesis is authoritative. */
finalize(words) {
this.confirmed = words;
this.prevHypo = words;
return this.confirmed;
}
reset() {
this.confirmed = [];
this.prevHypo = [];
}
}
/** Lowercase and strip edge punctuation so "Cat," and "cat" match. */
function normalizeWord(word) {
return word.replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, '').toLowerCase();
}
function commonPrefixCount(a, b) {
const n = Math.min(a.length, b.length);
let i = 0;
while (i < n && normalizeWord(a[i]) === normalizeWord(b[i])) i++;
return i;
}
/**
* Stock phrases the model confidently emits on ambient/silence -- the classic ASR
* hallucination. Confidence can't catch these (they score *higher* than genuine
* soft speech), so they're matched against the whole commit and dropped only when
* word density is low: a silence-padded hallucination is long and slow, a real
* crisp "thank you" is short and fast.
*/
const CANNED_HALLUCINATIONS = new Set([
'thank you', 'thank you very much', 'thanks for watching',
'thank you for watching', 'thanks for watching everyone',
]);
class StreamingCoordinator {
/**
* @param {object} opts
* @param {import('./ctc-engine.js').CtcEngine} opts.engine
* @param {(state: {committed: string, tentative: string}) => void} opts.onTranscript
* @param {(level: number) => void} [opts.onLevel] 0..1 input level
* @param {(info: object) => void} [opts.onStep] per-window diagnostics
* @param {(text: string) => Promise<string>} [opts.punctuate] committed-text post-pass
*/
constructor({ engine, onTranscript, onLevel = null, onStep = null, punctuate = null, tuning = {} }) {
this.engine = engine;
this.onTranscript = onTranscript;
this.onLevel = onLevel;
this.onStep = onStep;
this.punctuate = punctuate;
this.tuning = {
stepMs: 400, // re-transcribe cadence. A window costs one padded
// forward pass (>= 10.24 s of frames, see
// ctc-engine), so this is coarser than the macOS
// app's 100 ms; an in-flight tick is dropped, so
// it self-limits either way.
minWindowSec: 0.4, // don't transcribe windows shorter than this
maxWindowSec: 10.0, // force-commit a window that grows this long with no pause
newlineGapMs: 1200, // a pause >= this starts a new paragraph
hardPauseMs: 1500,
prerollSec: 0.5, // audio kept before speech onset, so the first word isn't clipped
// Calibrated on real far-field logs: clean speech ~ -0.03, but genuine
// soft utterances scored -0.98..-1.34 and were wrongly rejected at a
// -0.6 gate, dropping whole clips; silence confabulation sits at
// -1.85 and below, so -1.5 lands in the gap.
minCommitAvgLogprob: -1.5,
cannedHallucinationMaxWps: 2.0,
...tuning,
};
this.gate = new SpeechGate();
this.gate.onEvent = (e) => this.handleGate(e);
this.reconciler = new Reconciler();
this.window = []; // Float32Array chunks since the last commit
this.windowLen = 0;
this.windowHasSpeech = false;
this.transcribing = false;
this.windowGen = 0;
this.paragraphs = []; // committed text, one entry per paragraph
this.lastVoiceAt = 0;
this.segmentGapMs = Infinity;
this.active = false;
this.audioContext = null;
this.mediaStream = null;
this.stepTimer = null;
// One ONNX session can't have two concurrent run() calls, and a commit
// fired by the gate can land while a step transcription is still in
// flight, so every engine call goes through this chain.
this.engineLock = Promise.resolve();
// Commits are started from the gate callback (inside onSamples), which
// can't await them; stop() drains these so the last words aren't lost.
this.pending = new Set();
}
/** Serialize engine access; returns the callback's result. */
async withEngine(fn) {
const prior = this.engineLock;
let release;
this.engineLock = new Promise((r) => { release = r; });
try {
await prior;
return await fn();
} finally {
release();
}
}
/** Track a fire-and-forget async task so stop() can wait for it. */
track(promise) {
this.pending.add(promise);
promise.catch(() => {}).finally(() => this.pending.delete(promise));
return promise;
}
/** Resolve once no commit or step is outstanding. */
async idle() {
while (this.pending.size) await Promise.allSettled([...this.pending]);
await this.engineLock;
}
async start() {
if (this.active) return;
this.mediaStream = await navigator.mediaDevices.getUserMedia({
audio: {
channelCount: 1,
echoCancellation: true,
noiseSuppression: false, // the gate wants the true ambient level
autoGainControl: false, // the engine's own AGC handles levels
},
});
this.audioContext = new AudioContext({ sampleRate: SAMPLE_RATE });
await this.audioContext.audioWorklet.addModule(await captureWorkletUrl());
const source = this.audioContext.createMediaStreamSource(this.mediaStream);
const node = new AudioWorkletNode(this.audioContext, 'ctc-capture');
node.port.onmessage = (e) => this.onSamples(e.data);
source.connect(node);
// Keep the graph pulling without routing mic audio to the speakers.
const sink = this.audioContext.createGain();
sink.gain.value = 0;
node.connect(sink).connect(this.audioContext.destination);
this.resetWindow();
this.gate.recalibrate();
this.paragraphs = [];
this.lastVoiceAt = 0;
this.segmentGapMs = Infinity;
this.active = true;
this.publish([], []);
this.stepTimer = setInterval(() => this.tick(), this.tuning.stepMs);
}
async stop() {
if (!this.active) return;
this.active = false;
clearInterval(this.stepTimer);
this.stepTimer = null;
this.mediaStream?.getTracks().forEach((t) => t.stop());
await this.audioContext?.close();
this.audioContext = null;
this.onLevel?.(0);
// Let any commit the gate started finish before deciding whether the
// window still holds uncommitted speech.
await this.idle();
if (this.windowHasSpeech) await this.commit('stop');
this.resetWindow();
this.publish([], []);
}
// ---------------------------------------------------------------- capture
onSamples(chunk) {
this.window.push(chunk);
this.windowLen += chunk.length;
// Before speech is detected, keep only a short pre-roll. The window is
// otherwise cleared only on commit, so a long ambient stretch would
// accumulate unbounded and the first real utterance would commit an
// enormous mostly-noise window.
if (!this.windowHasSpeech) {
const maxPreroll = Math.floor(SAMPLE_RATE * this.tuning.prerollSec);
while (this.windowLen > maxPreroll && this.window.length > 1) {
this.windowLen -= this.window.shift().length;
}
}
const now = performance.now();
this.onLevel?.(meterLevel(rmsInt16(chunk)));
this.gate.feed(chunk); // may fire speechEnded -> commit
// Drive the window's speech state from the gate's debounced,
// calibration-aware decision rather than a parallel per-chunk RMS test:
// such a test has no debounce and runs during calibration, so one noise
// chunk latches the window open and the model transcribes ambient.
if (this.gate.isSpeaking) {
if (!this.windowHasSpeech) {
this.segmentGapMs = this.lastVoiceAt ? now - this.lastVoiceAt : Infinity;
this.windowHasSpeech = true;
}
this.lastVoiceAt = now;
}
}
handleGate(event) {
if (event !== 'speechEnded') return; // capture is continuous; onset needs nothing
// Called synchronously from onSamples, so the commit can only be tracked,
// not awaited (see idle()).
if (this.windowHasSpeech) this.track(this.commit('pause'));
else this.resetWindow(); // pure-silence window: drop it
}
// -------------------------------------------------------------- step loop
async tick() {
if (!this.active || !this.windowHasSpeech || this.transcribing) return;
if (this.windowLen / SAMPLE_RATE >= this.tuning.maxWindowSec) {
await this.commit('maxWindow');
return;
}
const speech = this.trimmedSpeech();
if (speech.length / SAMPLE_RATE < this.tuning.minWindowSec) return;
const gen = this.windowGen;
this.transcribing = true;
try {
const result = await this.withEngine(() => this.engine.transcribe(speech));
if (gen !== this.windowGen) return; // window committed meanwhile; discard
const { confirmed, tentative } = this.reconciler.update(result.words);
this.publish(confirmed, tentative);
this.onStep?.({ kind: 'step', windowSec: speech.length / SAMPLE_RATE, ...result });
} catch (e) {
console.error('step transcription failed:', e);
} finally {
this.transcribing = false;
}
}
/**
* Commit the current window: snapshot it, immediately clear and re-anchor (so
* capture continues uninterrupted and any in-flight partial is discarded),
* then transcribe the snapshot once cleanly and append it.
*/
async commit(reason) {
const speech = this.trimmedSpeech();
const windowSec = speech.length / SAMPLE_RATE;
if (windowSec < this.tuning.minWindowSec) {
this.resetWindow();
return;
}
const gapMs = this.segmentGapMs;
this.resetWindow();
this.publish([], []);
let result;
try {
result = await this.withEngine(() => this.engine.transcribe(speech));
} catch (e) {
console.error('commit transcription failed:', e);
return;
}
const words = result.words;
const wps = windowSec > 0 ? words.length / windowSec : 0;
// Confidence gate: a low mean logprob means the model wasn't sure of its
// own tokens -- the signature of confabulation on noise/silence. A null
// confidence fails open.
if (result.avgLogprob !== null && result.avgLogprob < this.tuning.minCommitAvgLogprob) {
this.onStep?.({ kind: 'reject', why: 'confidence', reason, windowSec, wps, ...result });
this.publish([], []);
return;
}
const normalized = words.join(' ').toLowerCase();
if (CANNED_HALLUCINATIONS.has(normalized) && wps < this.tuning.cannedHallucinationMaxWps) {
this.onStep?.({ kind: 'reject', why: 'canned', reason, windowSec, wps, ...result });
this.publish([], []);
return;
}
if (!words.length) {
this.publish([], []);
return;
}
await this.appendSegment(words.join(' '), gapMs, reason);
this.onStep?.({ kind: 'commit', reason, windowSec, wps, gapMs, ...result });
this.publish([], []);
}
/**
* Append a committed segment, opening a new paragraph after a real pause.
*
* Unlike the macOS app, sentence punctuation and capitalization are NOT
* synthesized here -- the model emits raw lowercase text and the punctuator
* restores both. A maxWindow cut is mid-utterance, so it extends the current
* paragraph and is re-punctuated together with what follows.
*/
async appendSegment(text, gapMs, reason) {
const startsParagraph = this.paragraphs.length === 0
|| gapMs >= this.tuning.hardPauseMs
|| (gapMs >= this.tuning.newlineGapMs && reason !== 'maxWindow');
if (startsParagraph) this.paragraphs.push(text);
else this.paragraphs[this.paragraphs.length - 1] += ` ${text}`;
if (this.punctuate) {
const i = this.paragraphs.length - 1;
const raw = this.paragraphs[i];
try {
this.paragraphs[i] = await this.punctuate(raw);
} catch (e) {
console.warn('punctuation failed; keeping raw text:', e);
}
}
}
// ---------------------------------------------------------------- helpers
/**
* The speech-bearing span of the window, with leading/trailing silence
* removed. Trailing trim matters: the front-end normalizes against the clip's
* global peak, so an untrimmed quiet tail gets amplified and decoded into
* spurious words.
*/
trimmedSpeech() {
const audio = this.flattenWindow();
const frame = 480; // 30 ms
const floor = this.gate.exitThreshold;
let first = -1, last = -1;
for (let i = 0; i < audio.length; i += frame) {
const end = Math.min(i + frame, audio.length);
if (rmsInt16(audio.subarray(i, end)) >= floor) {
if (first < 0) first = i;
last = end;
}
}
if (first < 0) return new Float32Array(0);
const margin = frame * 3; // ~90 ms of context each side
return audio.subarray(Math.max(0, first - margin), Math.min(audio.length, last + margin));
}
flattenWindow() {
if (this.window.length === 1) return this.window[0];
const out = new Float32Array(this.windowLen);
let at = 0;
for (const c of this.window) { out.set(c, at); at += c.length; }
this.window = [out]; // collapse so repeated ticks don't re-copy
return out;
}
resetWindow() {
this.window = [];
this.windowLen = 0;
this.windowHasSpeech = false;
this.reconciler.reset();
this.gate.reset();
this.windowGen++;
}
publish(confirmed, tentative) {
const live = confirmed.join(' ');
const committed = live
? [...this.paragraphs, live].join('\n\n')
: this.paragraphs.join('\n\n');
this.onTranscript({ committed, tentative: tentative.join(' ') });
}
/** Plain text of everything committed so far. */
get transcript() {
return this.paragraphs.join('\n\n');
}
}
/** Int16 RMS -> 0..1 meter level on a dBFS scale (-60 dBFS -> 0, 0 dBFS -> 1). */
function meterLevel(rms) {
if (rms <= 0) return 0;
const dbfs = 20 * Math.log10(rms / 32767);
return Math.max(0, Math.min(1, (dbfs + 60) / 60));
}
Object.assign(__M, { SpeechGate, Reconciler, StreamingCoordinator });
})();
// ==== app.js ============================================================
(() => {
const { CtcEngine, SpeechGate, StreamingCoordinator } = __M;
/**
* Granite Live Dictation (WebGPU)
*
* Real-time in-browser dictation with ibm-granite/granite-speech-4.2-470m-turboctc,
* a 473M-parameter FastConformer CTC model. Unlike the previous
* granite-speech-4.1-2b build, this is not a transformers.js pipeline:
* `ctc_conformer` is a custom architecture, so the ONNX session is driven directly
* (see ctc-engine.js) -- one forward pass and a greedy collapse per window, no
* autoregressive decode. That is what makes it fast enough to re-transcribe a
* growing window several times a second.
*
* The model emits raw lowercase, unpunctuated English. Punctuation and
* capitalization are restored by the separate punctuator model (punctuator.js),
* which is loaded on demand because it is a 209 MB download of its own.
*
* `ort` is the global from the onnxruntime-web script tag in index.html, shared
* with punctuator.js so both sessions use one runtime.
*/
const MODEL_DIR = './onnx-ctc/';
// 4-bit weights, block 32, on an fp16 graph (312 MB from a 1.9 GB fp32 export).
// This is the only shape WebGPU can actually run: ORT-web's WebGPU EP enforces
// `nbits == 4 || nbits == 2` on MatMulNBits, so an 8-bit build silently falls
// back to wasm (1320 ms per window against 305 ms) and buys nothing -- on 8
// minutes of LibriSpeech the 4-bit weights cost 0.09% WER against the fp32
// export, and what residue remains in the browser comes from fp16 activations,
// which more weight bits would not fix.
const MODEL_FILE = 'ctc_conformer_q4f16.onnx';
const SAMPLE_RATE = 16000;
// One padded forward pass covers 10.24 s of frames, so file segments are capped
// near that: longer segments cost proportionally more and buy nothing.
const MAX_SEGMENT_SEC = 20;
let engine = null;
let coordinator = null;
let punctuatorReady = false;
let punctuatorLoad = null;
let isLoading = false;
let currentAudioData = null;
const statusDot = document.getElementById('statusDot');
const statusText = document.getElementById('statusText');
const dictateBtn = document.getElementById('dictateBtn');
const levelFill = document.getElementById('levelFill');
const audioFile = document.getElementById('audioFile');
const fileTile = document.querySelector('.file-label');
const inputCard = document.querySelector('.input-card');
const punctCheckbox = document.getElementById('punctCheckbox');
const transcriptCard = document.getElementById('transcriptCard');
const committedEl = document.getElementById('committedText');
const tentativeEl = document.getElementById('tentativeText');
const copyBtn = document.getElementById('copyBtn');
const downloadBtn = document.getElementById('downloadBtn');
const clearBtn = document.getElementById('clearBtn');
const progressFill = document.getElementById('progressFill');
const progressSection = document.getElementById('progressSection');
const progressText = document.getElementById('progressText');
const gpuInfo = document.getElementById('gpuInfo');
const statsEl = document.getElementById('stats');
function setStatus(state, message) {
statusDot.className = `status-dot ${state}`;
statusText.textContent = message;
}
function showProgress(show, text = '') {
progressSection.style.display = show ? 'block' : 'none';
if (text) progressText.textContent = text;
}
// ---------------------------------------------------------------- model load
/**
* Hand ORT its wasm assets directly, fetched with a plain `fetch()`.
*
* ORT locates its backend by dynamically `import()`ing
* ort-wasm-simd-threaded.jsep.mjs. In a private Space's cross-site iframe that
* module fetch is not authenticated -- the same failure that stopped app.js's own
* imports and forced the bundle -- and it surfaces from session creation as
* "no available backend found. ERR: [wasm] previous call to 'initWasm()' failed".
*
* Plain fetches *are* authenticated in that context, so the glue is fetched here
* and handed over as a blob: URL, which `import()` reads with no network request
* at all, and the wasm bytes go across as `wasmBinary` so nothing else is looked
* up either. Falls back to letting ORT fetch them itself, which is what works
* everywhere other than that iframe.
*/
async function configureOrtWasm(baseUrl = './ort/') {
const dir = new URL(baseUrl, document.baseURI).href;
const get = async (name, as) => {
const r = await fetch(dir + name);
if (!r.ok) throw new Error(`${name}: HTTP ${r.status}`);
return as === 'text' ? r.text() : r.arrayBuffer();
};
try {
const [glue, binary] = await Promise.all([
get('ort-wasm-simd-threaded.jsep.mjs', 'text'),
get('ort-wasm-simd-threaded.jsep.wasm'),
]);
ort.env.wasm.wasmBinary = binary;
ort.env.wasm.wasmPaths = {
mjs: URL.createObjectURL(new Blob([glue], { type: 'text/javascript' })),
};
} catch (e) {
console.warn('could not preload the ORT wasm assets; letting ORT fetch them:', e);
// Absolute, not './ort/': ORT resolves a relative wasmPaths against the
// location of ort.all.min.js (already in ort/), which would look for
// ort/ort/ort-wasm-simd-threaded.jsep.mjs and find no backend.
ort.env.wasm.wasmPaths = dir;
}
}
async function initEngine() {
if (engine || isLoading) return engine;
isLoading = true;
try {
// The runtime is a blocking classic script in index.html, so reaching here
// without it means it was blocked (Edge Tracking Prevention, an extension,
// a CSP). Say that plainly instead of throwing a bare ReferenceError.
if (typeof ort === 'undefined') {
throw new Error('the ONNX runtime failed to load (ort/ort.all.min.js) — '
+ 'check for a blocked request in the network tab');
}
if (!navigator.gpu) {
gpuInfo.textContent = 'WebGPU not available — running on WASM (slower)';
}
await configureOrtWasm();
// Threaded wasm needs cross-origin isolation (COOP *and* COEP). Hugging
// Face Spaces send COOP but not COEP, so asking for threads there only
// produces a console warning and a wasted init before ORT falls back.
ort.env.wasm.numThreads = self.crossOriginIsolated ? (navigator.hardwareConcurrency || 4) : 1;
setStatus('loading', 'Downloading model...');
showProgress(true, 'Downloading model...');
const progress = {};
engine = await CtcEngine.load({
baseUrl: MODEL_DIR,
modelFile: MODEL_FILE,
device: navigator.gpu ? 'webgpu' : 'wasm',
onProgress: ({ url, loaded, total, cached }) => {
progress[url] = { loaded, total, cached };
let l = 0, t = 0, allCached = true;
for (const p of Object.values(progress)) {
l += p.loaded;
t += p.total;
if (!p.cached) allCached = false;
}
if (allCached) {
showProgress(true, 'Loading model from cache...');
return;
}
const pct = t > 0 ? (l / t) * 100 : 0;
progressFill.style.width = `${pct}%`;
showProgress(true, `Downloading model... ${(l / 1e6).toFixed(0)} / ${(t / 1e6).toFixed(0)} MB`);
},
});
progressFill.style.width = '0%';
showProgress(false);
// A wasm fallback is ~4x slower per window (1325 ms vs 305 ms for a 10 s
// window), which is the difference between live and not, so say so
// rather than letting it look like the model is just slow.
gpuInfo.textContent = engine.device === 'webgpu'
? 'Backend: WebGPU'
: 'Backend: WASM — WebGPU unavailable, so dictation will lag well behind speech';
setStatus('ready', 'Ready — start dictation or upload audio');
dictateBtn.disabled = false;
audioFile.disabled = false;
if (punctCheckbox.checked) startPunctuatorLoad();
return engine;
} catch (e) {
console.error('Model loading failed:', e);
setStatus('error', `Error: ${e.message || e}`);
showProgress(false);
throw e;
} finally {
isLoading = false;
}
}
/**
* Load the punctuator (209 MB of its own).
*
* Kicked off in the background as soon as the CTC engine is ready, so it is
* usually done before the user presses record -- loading it on the click would
* put a ~30 s download between "start dictation" and hearing anything. Calls
* share one promise, and only a caller that actually has to wait shows progress.
*/
function startPunctuatorLoad() {
if (!punctuatorLoad && typeof window.loadPunctuator === 'function') {
punctuatorLoad = window.loadPunctuator()
.then(() => { punctuatorReady = true; })
.catch((e) => {
console.warn('punctuator failed to load; transcript stays lowercase:', e);
punctuatorReady = false;
});
}
return punctuatorLoad;
}
async function ensurePunctuator() {
if (punctuatorReady || !punctCheckbox.checked) return punctuatorReady;
const load = startPunctuatorLoad();
if (!load) return false;
const stillLoading = !punctuatorReady;
if (stillLoading) showProgress(true, 'Loading punctuation model...');
await load;
if (stillLoading) showProgress(false);
return punctuatorReady;
}
/** Punctuate committed text, or pass it through if the model is off/unavailable. */
async function punctuate(text) {
if (!punctCheckbox.checked || !punctuatorReady) return text;
return window.applyPunctuation(text);
}
// ------------------------------------------------------------ live dictation
function renderTranscript({ committed, tentative }) {
transcriptCard.style.display = 'block';
committedEl.textContent = committed;
tentativeEl.textContent = tentative ? (committed ? ` ${tentative}` : tentative) : '';
const out = committedEl.parentElement;
out.scrollTop = out.scrollHeight;
const hasText = Boolean(committed || tentative);
copyBtn.disabled = !hasText;
downloadBtn.disabled = !hasText;
}
async function toggleDictation() {
if (coordinator) {
dictateBtn.disabled = true;
setStatus('processing', 'Finishing…');
try {
await coordinator.stop();
} finally {
coordinator = null;
dictateBtn.classList.remove('recording');
dictateBtn.querySelector('span').textContent = 'Start dictation';
dictateBtn.disabled = false;
levelFill.style.width = '0%';
setStatus('ready', 'Ready — start dictation or upload audio');
}
return;
}
await initEngine();
await ensurePunctuator();
coordinator = new StreamingCoordinator({
engine,
onTranscript: renderTranscript,
onLevel: (level) => { levelFill.style.width = `${(level * 100).toFixed(0)}%`; },
onStep: (info) => {
if (info.kind === 'step') {
statsEl.textContent = `${info.windowSec.toFixed(1)}s window · ${info.ms.toFixed(0)} ms · conf ${info.avgLogprob?.toFixed(2) ?? 'n/a'}`;
} else if (info.kind === 'reject') {
console.log(`[reject:${info.why}] "${info.text}" conf=${info.avgLogprob?.toFixed(2)} wps=${info.wps.toFixed(2)}`);
} else {
console.log(`[commit:${info.reason}] ${info.windowSec.toFixed(1)}s ${info.ms.toFixed(0)}ms conf=${info.avgLogprob?.toFixed(2)} "${info.text}"`);
}
},
punctuate,
});
try {
await coordinator.start();
dictateBtn.classList.add('recording');
dictateBtn.querySelector('span').textContent = 'Stop dictation';
setStatus('recording', 'Listening — speak naturally, pause between sentences');
} catch (e) {
console.error('dictation failed to start:', e);
coordinator = null;
setStatus('error', e.name === 'NotAllowedError' ? 'Microphone access denied' : `Error: ${e.message || e}`);
}
}
// --------------------------------------------------------- file transcription
/**
* Split a file into speech segments with the same energy gate the live path uses.
*
* This replaces the old Silero VAD download: the gate is already here, it is what
* the streaming path is tuned against, and long audio has to be split anyway --
* a single pass over a 5-minute file would allocate a ~250 MB logits tensor
* inside the graph.
*/
function segmentAudio(audio) {
const gate = new SpeechGate();
const frame = 480;
const segments = [];
let start = null;
let consumed = 0;
gate.onEvent = (event) => {
const t = consumed / SAMPLE_RATE;
if (event === 'speechStarted') start = t;
else if (start !== null) {
segments.push({ start, end: t });
start = null;
}
};
for (let i = 0; i < audio.length; i += frame) {
const chunk = audio.subarray(i, Math.min(i + frame, audio.length));
consumed = i + chunk.length;
gate.feed(chunk);
}
if (start !== null) segments.push({ start, end: audio.length / SAMPLE_RATE });
if (!segments.length) return [{ start: 0, end: audio.length / SAMPLE_RATE }];
// Pad each segment slightly (the gate's onset debounce trims the first
// phoneme) and cap the length.
const out = [];
for (const seg of segments) {
let from = Math.max(0, seg.start - 0.2);
const to = Math.min(audio.length / SAMPLE_RATE, seg.end + 0.2);
while (to - from > MAX_SEGMENT_SEC) {
out.push({ start: from, end: from + MAX_SEGMENT_SEC });
from += MAX_SEGMENT_SEC;
}
out.push({ start: from, end: to });
}
return out;
}
async function transcribeFile() {
if (!currentAudioData) return;
await initEngine();
await ensurePunctuator();
setStatus('processing', 'Transcribing…');
transcriptCard.style.display = 'block';
const segments = segmentAudio(currentAudioData);
const parts = [];
let totalMs = 0;
for (let i = 0; i < segments.length; i++) {
const seg = segments[i];
showProgress(true, `Segment ${i + 1} / ${segments.length}`);
progressFill.style.width = `${((i + 1) / segments.length) * 100}%`;
const slice = currentAudioData.subarray(
Math.floor(seg.start * SAMPLE_RATE),
Math.floor(seg.end * SAMPLE_RATE),
);
const result = await engine.transcribe(slice);
totalMs += result.ms;
if (result.text) {
parts.push(await punctuate(result.text));
renderTranscript({ committed: parts.join('\n\n'), tentative: '' });
}
}
showProgress(false);
progressFill.style.width = '0%';
const audioSec = currentAudioData.length / SAMPLE_RATE;
statsEl.textContent = `${audioSec.toFixed(1)}s audio · ${segments.length} segment(s) · ${totalMs.toFixed(0)} ms · ${(audioSec / (totalMs / 1000)).toFixed(0)}x realtime`;
if (!parts.length) renderTranscript({ committed: '(no speech detected)', tentative: '' });
setStatus('ready', 'Transcription complete');
}
/** Decode any audio file to 16 kHz mono float32. */
async function loadAudioFile(file) {
setStatus('processing', 'Decoding audio…');
try {
const audioCtx = new AudioContext({ sampleRate: SAMPLE_RATE });
const buffer = await audioCtx.decodeAudioData(await file.arrayBuffer());
let audio = buffer.getChannelData(0);
if (buffer.numberOfChannels > 1) {
const mixed = new Float32Array(audio.length);
for (let c = 0; c < buffer.numberOfChannels; c++) {
const ch = buffer.getChannelData(c);
for (let i = 0; i < mixed.length; i++) mixed[i] += ch[i] / buffer.numberOfChannels;
}
audio = mixed;
}
// decodeAudioData already resampled to the context rate.
await audioCtx.close();
currentAudioData = audio;
await transcribeFile();
} catch (e) {
console.error('audio decode failed:', e);
setStatus('error', `Could not read that file: ${e.message || e}`);
}
}
// -------------------------------------------------------------- transcript IO
function transcriptText() {
return committedEl.textContent + (tentativeEl.textContent || '');
}
async function copyTranscript() {
await navigator.clipboard.writeText(transcriptText());
const original = copyBtn.title;
copyBtn.title = 'Copied';
setTimeout(() => { copyBtn.title = original; }, 1200);
}
function downloadTranscript() {
const blob = new Blob([transcriptText()], { type: 'text/plain' });
const a = document.createElement('a');
a.href = URL.createObjectURL(blob);
a.download = `transcript-${new Date().toISOString().slice(0, 19).replace(/[:T]/g, '-')}.txt`;
a.click();
URL.revokeObjectURL(a.href);
}
function clearTranscript() {
renderTranscript({ committed: '', tentative: '' });
transcriptCard.style.display = 'none';
statsEl.textContent = '';
if (coordinator) coordinator.paragraphs = [];
}
// ---------------------------------------------------------------------- wiring
dictateBtn.addEventListener('click', toggleDictation);
audioFile.addEventListener('change', (e) => {
const file = e.target.files[0];
if (file) loadAudioFile(file);
e.target.value = '';
});
copyBtn.addEventListener('click', copyTranscript);
downloadBtn.addEventListener('click', downloadTranscript);
clearBtn.addEventListener('click', clearTranscript);
// Drag and drop onto the input card
for (const [event, handler] of [
['dragover', (e) => { e.preventDefault(); inputCard.classList.add('drag-over'); }],
['dragleave', () => inputCard.classList.remove('drag-over')],
['drop', (e) => {
e.preventDefault();
inputCard.classList.remove('drag-over');
const file = e.dataTransfer.files[0];
if (file?.type.startsWith('audio/')) loadAudioFile(file);
else setStatus('error', 'Please drop an audio file');
}],
]) {
inputCard.addEventListener(event, handler);
fileTile.addEventListener(event, (e) => e.stopPropagation(), { capture: false });
}
punctCheckbox.addEventListener('change', () => {
if (punctCheckbox.checked) ensurePunctuator();
});
// Space bar toggles dictation, unless a control has focus.
document.addEventListener('keydown', (e) => {
if (e.code !== 'Space' || e.repeat) return;
if (['INPUT', 'TEXTAREA', 'SELECT', 'BUTTON'].includes(document.activeElement?.tagName)) return;
e.preventDefault();
toggleDictation();
});
setStatus('loading', 'Loading…');
initEngine().catch(() => {});
})();
})();