File size: 17,043 Bytes
6c3af4e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d84bd10
 
 
 
 
6c3af4e
d84bd10
 
 
6c3af4e
 
 
 
 
d84bd10
6c3af4e
 
 
d84bd10
 
 
 
 
6c3af4e
d84bd10
 
6c3af4e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d84bd10
6c3af4e
d84bd10
6c3af4e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
d84bd10
 
 
 
 
 
 
6c3af4e
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
import { DICTATION_MODELS_ROUTE_PREFIX } from "@shared/dictationModel";
import { addLogEntry } from "./logEntries";
import type { WorkerResponse } from "./speechToTextWorkerProtocol";

/**
 * Which path can transcribe speech in this browser. `"wasm"` runs the
 * Moonshine model locally in a worker; `"web-speech"` falls back to the
 * browser's own `SpeechRecognition`, which sends audio to the vendor.
 */
export type DictationEngine = "wasm" | "web-speech";

export type DictationErrorKind = "permission" | "unavailable" | "engine";

export class DictationError extends Error {
  kind: DictationErrorKind;

  constructor(kind: DictationErrorKind, message: string) {
    super(message);
    this.kind = kind;
  }
}

/**
 * The streaming English model (MIT-licensed), served from this instance at
 * `/dictation-models/<file>` instead of the Moonshine CDN, so the page makes
 * no third-party requests. The filenames are the canonical ones the
 * streaming architecture loads; the server hook maps them to the pinned
 * upstream files and caches them on disk.
 */
const DICTATION_MODEL_FILES: Record<string, string> = Object.fromEntries(
  [
    "frontend.ort",
    "encoder.ort",
    "adapter.ort",
    "cross_kv.ort",
    "decoder_kv.ort",
    "streaming_config.json",
    "tokenizer.bin",
  ].map((file) => [file, `${DICTATION_MODELS_ROUTE_PREFIX}${file}`]),
);

/** The minimum the app needs from `window.SpeechRecognition`. */
interface SpeechRecognitionLike {
  continuous: boolean;
  interimResults: boolean;
  lang: string;
  onresult:
    | ((event: {
        resultIndex: number;
        results: ArrayLike<{ isFinal: boolean; 0: { transcript: string } }>;
      }) => void)
    | null;
  onerror: ((event: { error: string }) => void) | null;
  onend: (() => void) | null;
  start: () => void;
  stop: () => void;
}

type SpeechRecognitionConstructor = new () => SpeechRecognitionLike;

function getSpeechRecognitionConstructor(): SpeechRecognitionConstructor | null {
  const candidate = window as unknown as {
    SpeechRecognition?: SpeechRecognitionConstructor;
    webkitSpeechRecognition?: SpeechRecognitionConstructor;
  };
  return (
    candidate.SpeechRecognition ?? candidate.webkitSpeechRecognition ?? null
  );
}

/**
 * The engine that would run a dictation right now, or null when neither the
 * WASM path nor `SpeechRecognition` is available and the button should hide.
 *
 * `preferLocalModel` reorders the two engines; it never empties the list.
 * A user who turns the on-device model off on a browser without
 * `SpeechRecognition` (Firefox, for one) must still get a working Dictate
 * button, so the preference cannot be an exclusion.
 */
export function getDictationEngine(
  preferLocalModel: boolean,
): DictationEngine | null {
  // Neither engine can open a microphone outside a secure context. Without
  // this, a plain-HTTP LAN deployment falls through to `SpeechRecognition`,
  // which then reports `not-allowed` and tells the user permission was denied
  // when the real cause is the origin.
  if (typeof isSecureContext !== "undefined" && !isSecureContext) return null;
  const wasmCapable =
    typeof Worker !== "undefined" &&
    typeof WebAssembly !== "undefined" &&
    typeof AudioContext !== "undefined" &&
    typeof navigator.mediaDevices?.getUserMedia === "function";
  const webSpeechAvailable = getSpeechRecognitionConstructor() !== null;
  if (preferLocalModel) {
    if (wasmCapable) return "wasm";
    return webSpeechAvailable ? "web-speech" : null;
  }
  if (webSpeechAvailable) return "web-speech";
  return wasmCapable ? "wasm" : null;
}

export interface DictationCallbacks {
  /** Called with the whole dictated text so far, as it grows. */
  onTranscript: (text: string) => void;
  /** Model download progress; `total` is undefined when sizes are unknown. */
  onProgress?: (loaded: number, total?: number) => void;
  /**
   * The local engine could not run and the browser's own recognizer took over.
   * That one sends audio to the browser vendor, so it is worth saying out loud.
   */
  onFallback?: () => void;
  /**
   * The engine stopped on its own, without failing. The browser's recognizer
   * ends after silence even with `continuous`, and the session is over once it
   * does.
   */
  onEnd?: () => void;
  /**
   * A failure after the engine loaded. The load itself rejects instead, so
   * this is the channel for a worker that dies mid-dictation, which would
   * otherwise leave the UI listening forever with the microphone open.
   */
  onError?: (error: DictationError) => void;
}

export interface DictationSession {
  stop: () => Promise<void>;
}

let workerFactory = () =>
  new Worker(new URL("./speechToTextWorker.ts", import.meta.url), {
    type: "module",
  });

/** Lets the tests supply a fake worker; the real one needs a bundler. */
export function setWorkerFactory(factory: () => Worker) {
  workerFactory = factory;
}

/**
 * Linear-interpolation resample to the 16 kHz mono PCM the model expects.
 * The capture rate depends on the sound device, so the conversion happens
 * here rather than assuming 16 kHz out of the AudioContext.
 */
export function resampleTo16k(
  input: Float32Array,
  inputRate: number,
): Float32Array {
  if (inputRate === 16000) return input;
  const ratio = inputRate / 16000;
  const outputLength = Math.max(1, Math.floor(input.length / ratio));
  const output = new Float32Array(outputLength);
  for (let i = 0; i < outputLength; i++) {
    const position = i * ratio;
    const index = Math.floor(position);
    const fraction = position - index;
    const current = input[index] ?? 0;
    const next = input[Math.min(index + 1, input.length - 1)] ?? 0;
    output[i] = current * (1 - fraction) + next * fraction;
  }
  return output;
}

function describeError(error: unknown): string {
  return error instanceof Error ? error.message : String(error);
}

async function startWasmDictation(
  callbacks: DictationCallbacks,
): Promise<DictationSession> {
  const worker = workerFactory();
  /**
   * Set the moment `stop()` is entered. Every callback is gated on it: the
   * engines both emit one last result after being told to stop, and the button
   * has already forgotten what it appended by then, so an ungated late
   * transcript is appended a second time in full.
   */
  let stopped = false;

  /**
   * An engine failure between `loaded` and the microphone being granted. The
   * load promise has already settled by then, so rejecting it again is a no-op;
   * the permission prompt can sit open for minutes, and the transcriber is
   * already running and can die in that window.
   */
  let failureAfterLoad: DictationError | null = null;
  let engineLoaded = false;

  const loadEngine = () =>
    new Promise<void>((resolve, reject) => {
      worker.onmessage = ({ data }: MessageEvent<WorkerResponse>) => {
        if (data.type === "loaded") {
          engineLoaded = true;
          resolve();
        } else if (data.type === "progress")
          callbacks.onProgress?.(data.loaded, data.total);
        else if (data.type === "error") {
          const failure = new DictationError("engine", data.message);
          if (engineLoaded) failureAfterLoad = failure;
          else reject(failure);
        }
      };
      worker.onerror = () => {
        const failure = new DictationError(
          "engine",
          "The dictation worker failed to start",
        );
        if (engineLoaded) failureAfterLoad = failure;
        else reject(failure);
      };
      worker.postMessage({ type: "load", modelFiles: DICTATION_MODEL_FILES });
    });

  try {
    await loadEngine();
  } catch (error) {
    worker.terminate();
    throw error;
  }

  // Only once the model is ready: opening it first would light the browser's
  // recording indicator for the whole of a first-run download.
  let mediaStream: MediaStream;
  try {
    mediaStream = await navigator.mediaDevices.getUserMedia({ audio: true });
  } catch (error) {
    worker.terminate();
    throw new DictationError(
      "permission",
      `The microphone could not be opened: ${describeError(error)}`,
    );
  }

  // Failed while the permission prompt was open. Thrown rather than reported
  // through `onError`, so it lands in `startDictation`'s catch and takes the
  // same fallback as any other engine failure.
  if (failureAfterLoad) {
    mediaStream.getTracks().forEach((track) => {
      track.stop();
    });
    worker.terminate();
    throw failureAfterLoad;
  }

  let acknowledgeStop: (() => void) | null = null;

  worker.onmessage = ({ data }: MessageEvent<WorkerResponse>) => {
    if (data.type === "stopped") {
      acknowledgeStop?.();
      return;
    }
    if (stopped) return;
    if (data.type === "transcript") callbacks.onTranscript(data.text);
    else if (data.type === "error")
      callbacks.onError?.(new DictationError("engine", data.message));
  };
  worker.onerror = () => {
    if (stopped) return;
    callbacks.onError?.(
      new DictationError("engine", "The dictation worker stopped unexpectedly"),
    );
  };

  // Asking for 16 kHz lets the browser resample properly. `resampleTo16k` is
  // the fallback for devices that refuse the rate: its linear interpolation
  // has no lowpass, so everything above 8 kHz aliases into the band the model
  // listens to.
  const releasePartialSession = (context?: AudioContext) => {
    mediaStream.getTracks().forEach((track) => {
      track.stop();
    });
    void context?.close();
    worker.terminate();
  };

  let audioContext: AudioContext | undefined;
  let source: MediaStreamAudioSourceNode | undefined;
  let processor: ScriptProcessorNode | undefined;
  try {
    const context = new AudioContext({ sampleRate: 16000 });
    audioContext = context;
    source = context.createMediaStreamSource(mediaStream);
    // A 1-channel ScriptProcessor downmixes stereo capture to mono, which
    // matters for devices whose microphone sits on the right channel only.
    processor = context.createScriptProcessor(4096, 1, 1);

    processor.onaudioprocess = (event) => {
      if (stopped) return;
      const input = event.inputBuffer.getChannelData(0);
      const resampled = resampleTo16k(input, context.sampleRate);
      // The capture buffer is reused and a transferred buffer cannot be read
      // again, so the untouched path needs a copy. `resampleTo16k` already
      // returned a fresh array nobody else holds.
      const copy = resampled === input ? new Float32Array(input) : resampled;
      worker.postMessage(
        { type: "audio", buffer: copy.buffer, sampleRate: 16000 },
        [copy.buffer],
      );
      // The processor must stay connected to the destination to be called;
      // zero the output so the microphone is not played back through the
      // speakers.
      event.outputBuffer.getChannelData(0).fill(0);
    };

    source.connect(processor);
    processor.connect(context.destination);
  } catch (error) {
    // `new AudioContext({ sampleRate })` throws when the rate is refused, and
    // Chrome throws once a page holds too many live contexts. Escaping here as
    // a plain Error would be caught as "the local engine cannot run" and hand
    // the microphone to the browser's cloud recognizer by accident.
    releasePartialSession(audioContext);
    throw new DictationError(
      "engine",
      `The audio pipeline could not be started: ${describeError(error)}`,
    );
  }

  return {
    stop: async () => {
      if (stopped) return;
      stopped = true;
      processor?.disconnect();
      source?.disconnect();
      mediaStream.getTracks().forEach((track) => {
        track.stop();
      });

      // Wait for the transcriber to drain its queue, but never hang the UI on
      // a worker that has stopped answering.
      const acknowledged = new Promise<void>((resolve) => {
        acknowledgeStop = resolve;
      });
      worker.postMessage({ type: "stop" });
      await Promise.race([
        acknowledged,
        new Promise<void>((resolve) => setTimeout(resolve, 2000)),
      ]);

      try {
        await audioContext?.close();
      } finally {
        // Never skipped: a rejecting close would otherwise leave the worker,
        // and its transcriber, running for the life of the page.
        worker.terminate();
      }
    },
  };
}

function startWebSpeechDictation(
  callbacks: DictationCallbacks,
): Promise<DictationSession> {
  const Recognition = getSpeechRecognitionConstructor();
  if (!Recognition) {
    return Promise.reject(
      new DictationError("unavailable", "No dictation engine is available"),
    );
  }

  const recognition = new Recognition();
  recognition.continuous = true;
  recognition.interimResults = true;
  recognition.lang = navigator.language;

  let finalText = "";
  /** The returned promise has resolved, so rejecting it would be a no-op. */
  let resolved = false;
  // `recognition.stop()` emits one last `result` by spec, and the caller has
  // already forgotten what it appended by then.
  let stopped = false;

  return new Promise<DictationSession>((resolve, reject) => {
    recognition.onresult = (event) => {
      if (stopped) return;
      let interim = "";
      for (let i = event.resultIndex; i < event.results.length; i++) {
        const result = event.results[i];
        if (result?.isFinal) finalText += result[0].transcript;
        else interim += result[0].transcript;
      }
      // Collapsed because the final text and the interim chunk each may or may
      // not carry their own spacing, and a search query wants one space.
      callbacks.onTranscript(
        `${finalText} ${interim}`.replace(/\s+/g, " ").trim(),
      );
    };

    recognition.onerror = (event) => {
      if (stopped) return;
      const denied =
        event.error === "not-allowed" || event.error === "service-not-allowed";
      const failure = denied
        ? new DictationError(
            "permission",
            "Microphone permission was denied for dictation",
          )
        : new DictationError("engine", `Dictation failed: ${event.error}`);
      // `start()` does not throw on a denial: the error arrives later, by which
      // point this promise has already resolved and rejecting it is a no-op.
      // Everything after the resolve therefore goes through `onError`.
      if (resolved) callbacks.onError?.(failure);
      else reject(failure);
    };

    recognition.onend = () => {
      if (stopped) return;
      // The recognizer ends on its own after silence even with `continuous`.
      // That is not a failure worth a notification, but the session is over, so
      // the button has to come back from Listening.
      if (resolved) {
        callbacks.onEnd?.();
        return;
      }
      reject(new DictationError("engine", "Dictation ended before any speech"));
    };

    try {
      recognition.start();
      resolved = true;
      resolve({
        stop: async () => {
          stopped = true;
          recognition.stop();
        },
      });
    } catch (error) {
      reject(
        new DictationError(
          "unavailable",
          `Dictation could not start: ${describeError(error)}`,
        ),
      );
    }
  });
}

/**
 * Starts dictating. Resolves once the microphone is open (and, on the WASM
 * path, the model is loaded); rejects with a `DictationError` whose `kind`
 * tells the caller whether the user denied permission or the engine failed.
 */
export async function startDictation(
  callbacks: DictationCallbacks,
  preferLocalModel: boolean,
): Promise<DictationSession> {
  const engine = getDictationEngine(preferLocalModel);
  if (engine === "wasm") {
    try {
      return await startWasmDictation(callbacks);
    } catch (error) {
      // A denied microphone is the user's answer, not an engine that cannot
      // run, so it is reported rather than retried on a path that would ask
      // again and, in Chrome, ship the audio to the browser vendor.
      if (error instanceof DictationError && error.kind === "permission")
        throw error;
      if (!getSpeechRecognitionConstructor()) throw error;
      addLogEntry(
        `Local dictation failed, falling back to the browser's recognizer: ${describeError(error)}`,
      );
      // The user pressed a button documented as on-device, so the switch to a
      // recognizer that ships audio to the browser vendor is announced.
      callbacks.onFallback?.();
      return startWebSpeechDictation(callbacks);
    }
  }
  if (engine === "web-speech") {
    // No reverse fallback to wasm when this recognizer fails. Falling through
    // would start a ~51 MB model download right after the user turned that
    // model off; a clear failure is the better answer, and it matches what
    // web-speech-only browsers already do.
    return startWebSpeechDictation(callbacks);
  }
  addLogEntry("Dictation is not available in this browser");
  throw new DictationError(
    "unavailable",
    "This browser cannot dictate a search query",
  );
}