Download index.html from wq2012/tec: direct link, hf CLI and curl.
- Browser
- Download file 26.1 kB
-
https://huggingface.co/spaces/wq2012/tec/resolve/main/index.html
- Command line
-
hf download hf://spaces/wq2012/tec/index.html
-
curl -L -o index.html https://huggingface.co/spaces/wq2012/tec/resolve/main/index.html
26.1 kB
| <html lang="en"> | |
| <head> | |
| <meta charset="UTF-8" /> | |
| <meta name="viewport" content="width=device-width, initial-scale=1.0" /> | |
| <title>Textual Echo Cancellation (TEC) — Interactive Demo</title> | |
| <style> | |
| :root { | |
| --bg: #f8fafc; | |
| --card: #ffffff; | |
| --border: #e2e8f0; | |
| --text: #0f172a; | |
| --muted: #475569; | |
| --primary: #2563eb; | |
| --primary-hover: #1d4ed8; | |
| --success-bg: #ecfdf5; | |
| --success-border: #10b981; | |
| --warn-bg: #fffbeb; | |
| --warn-border: #f59e0b; | |
| } | |
| * { box-sizing: border-box; } | |
| body { | |
| margin: 0; | |
| font-family: -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, Helvetica, Arial, sans-serif; | |
| background: var(--bg); | |
| color: var(--text); | |
| line-height: 1.5; | |
| } | |
| .container { | |
| max-width: 1140px; | |
| margin: 0 auto; | |
| padding: 28px 20px 48px; | |
| } | |
| .header { | |
| background: var(--card); | |
| border: 1px solid var(--border); | |
| border-radius: 14px; | |
| padding: 24px 28px; | |
| margin-bottom: 24px; | |
| box-shadow: 0 1px 3px rgba(15, 23, 42, 0.04); | |
| } | |
| .header h1 { | |
| margin: 0 0 8px; | |
| font-size: 1.75rem; | |
| display: flex; | |
| align-items: center; | |
| gap: 10px; | |
| } | |
| .header p { | |
| margin: 8px 0; | |
| color: var(--muted); | |
| font-size: 0.96rem; | |
| } | |
| .badges { | |
| display: flex; | |
| flex-wrap: wrap; | |
| gap: 10px; | |
| margin-top: 14px; | |
| } | |
| .badge-link { | |
| display: inline-flex; | |
| align-items: center; | |
| gap: 6px; | |
| padding: 6px 12px; | |
| border-radius: 8px; | |
| background: #eff6ff; | |
| color: #1e40af; | |
| font-size: 0.86rem; | |
| font-weight: 600; | |
| text-decoration: none; | |
| border: 1px solid #bfdbfe; | |
| } | |
| .badge-link:hover { | |
| background: #dbeafe; | |
| } | |
| .grid { | |
| display: grid; | |
| grid-template-columns: 1fr 1fr; | |
| gap: 24px; | |
| } | |
| @media (max-width: 860px) { | |
| .grid { grid-template-columns: 1fr; } | |
| } | |
| .panel { | |
| background: var(--card); | |
| border: 1px solid var(--border); | |
| border-radius: 14px; | |
| padding: 22px 24px; | |
| box-shadow: 0 1px 3px rgba(15, 23, 42, 0.04); | |
| } | |
| .panel h2 { | |
| margin: 0 0 16px; | |
| font-size: 1.18rem; | |
| border-bottom: 1px solid var(--border); | |
| padding-bottom: 10px; | |
| } | |
| .field { | |
| margin-bottom: 18px; | |
| } | |
| .field label { | |
| display: block; | |
| font-weight: 600; | |
| font-size: 0.92rem; | |
| margin-bottom: 6px; | |
| } | |
| .field .hint { | |
| font-size: 0.82rem; | |
| color: var(--muted); | |
| margin-bottom: 8px; | |
| } | |
| input[type="file"], select, textarea { | |
| width: 100%; | |
| padding: 10px 12px; | |
| border: 1px solid #cbd5e1; | |
| border-radius: 8px; | |
| font-size: 0.94rem; | |
| font-family: inherit; | |
| background: #fff; | |
| } | |
| textarea { | |
| resize: vertical; | |
| min-height: 76px; | |
| } | |
| audio { | |
| width: 100%; | |
| margin-top: 8px; | |
| } | |
| .btn-primary { | |
| width: 100%; | |
| padding: 13px 18px; | |
| background: var(--primary); | |
| color: #fff; | |
| font-weight: 600; | |
| font-size: 1rem; | |
| border: none; | |
| border-radius: 10px; | |
| cursor: pointer; | |
| transition: background 0.15s ease; | |
| } | |
| .btn-primary:hover:not(:disabled) { | |
| background: var(--primary-hover); | |
| } | |
| .btn-primary:disabled { | |
| opacity: 0.65; | |
| cursor: wait; | |
| } | |
| .status-bar { | |
| margin-top: 12px; | |
| padding: 10px 12px; | |
| border-radius: 8px; | |
| background: #f1f5f9; | |
| font-size: 0.88rem; | |
| color: #334155; | |
| display: none; | |
| } | |
| .result-box { | |
| padding: 12px 14px; | |
| border-radius: 10px; | |
| border: 1px solid var(--border); | |
| background: #f8fafc; | |
| font-family: ui-monospace, SFMono-Regular, Menlo, Monaco, Consolas, monospace; | |
| font-size: 0.93rem; | |
| min-height: 48px; | |
| white-space: pre-wrap; | |
| word-break: break-word; | |
| } | |
| .result-box.before { | |
| background: var(--warn-bg); | |
| border-color: var(--warn-border); | |
| } | |
| .result-box.after { | |
| background: var(--success-bg); | |
| border-color: var(--success-border); | |
| font-weight: 600; | |
| } | |
| table.comp-table { | |
| width: 100%; | |
| border-collapse: collapse; | |
| margin-top: 10px; | |
| font-size: 0.88rem; | |
| } | |
| table.comp-table th, table.comp-table td { | |
| border: 1px solid var(--border); | |
| padding: 9px 10px; | |
| text-align: left; | |
| vertical-align: top; | |
| } | |
| table.comp-table th { | |
| background: #f1f5f9; | |
| font-weight: 600; | |
| } | |
| .examples-section { | |
| margin-top: 24px; | |
| } | |
| .examples-table { | |
| width: 100%; | |
| border-collapse: collapse; | |
| font-size: 0.88rem; | |
| } | |
| .examples-table th, .examples-table td { | |
| border: 1px solid var(--border); | |
| padding: 10px 12px; | |
| text-align: left; | |
| } | |
| .examples-table th { | |
| background: #f1f5f9; | |
| } | |
| .examples-table tr.example-row { | |
| cursor: pointer; | |
| transition: background 0.12s; | |
| } | |
| .examples-table tr.example-row:hover { | |
| background: #eff6ff; | |
| } | |
| .tag { | |
| display: inline-block; | |
| padding: 2px 8px; | |
| border-radius: 999px; | |
| font-size: 0.76rem; | |
| font-weight: 600; | |
| background: #e0e7ff; | |
| color: #3730a3; | |
| } | |
| </style> | |
| </head> | |
| <body> | |
| <div class="container"> | |
| <div class="header"> | |
| <h1>🎙️ Textual Echo Cancellation (TEC) — Interactive Demo</h1> | |
| <p> | |
| When a user speaks to a smart speaker or voice assistant while the device is playing back a Text-to-Speech (TTS) response, the microphone captures an overlapping mixture of the <b>user's speech</b> and the <b>reverberant TTS playback echo</b>. | |
| <b>Textual Echo Cancellation (TEC)</b> cancels the TTS echo using the <b>source text of the TTS playback</b> (< 0.1 KB side input) and reconstructs the clean user speech for Automatic Speech Recognition (ASR). | |
| </p> | |
| <div class="badges"> | |
| <a class="badge-link" href="https://arxiv.org/abs/2008.06006" target="_blank">📄 Paper (arXiv:2008.06006)</a> | |
| <a class="badge-link" href="https://github.com/wq2012/tec" target="_blank">💻 GitHub (wq2012/tec)</a> | |
| <a class="badge-link" href="https://pypi.org/project/textual-echo-cancellation/" target="_blank">📦 PyPI (textual-echo-cancellation)</a> | |
| <a class="badge-link" href="https://huggingface.co/wq2012/tec_single_interfering" target="_blank">🤗 Model: tec_single_interfering</a> | |
| <a class="badge-link" href="https://huggingface.co/wq2012/tec_multi_interfering" target="_blank">🤗 Model: tec_multi_interfering</a> | |
| <a class="badge-link" href="https://google.github.io/speaker-id/publications/TEC/" target="_blank">🔊 Paper Audio Samples</a> | |
| </div> | |
| </div> | |
| <div class="grid"> | |
| <!-- Left Column: Inputs --> | |
| <div class="panel"> | |
| <h2>Inputs</h2> | |
| <div class="field"> | |
| <label for="audioFileInput">1. User Uploaded Audio File (Microphone Mixture: User Query + TTS Playback Echo)</label> | |
| <div class="hint">Upload any WAV/MP3/FLAC audio file, or click one of the built-in examples below.</div> | |
| <input type="file" id="audioFileInput" accept="audio/*" /> | |
| <audio id="inputAudioPlayer" controls preload="metadata"></audio> | |
| </div> | |
| <div class="field"> | |
| <label for="ttsTextInput">2. Interfering TTS Playback Text (Source Text to Cancel)</label> | |
| <div class="hint">Enter the text transcript being spoken by the device's TTS playback.</div> | |
| <textarea id="ttsTextInput" placeholder="Enter the text spoken by the interfering TTS voice...">a table showing the figures for the year ending Michaelmas eighteen oh two.</textarea> | |
| </div> | |
| <div class="field"> | |
| <label for="modelSelect">Pretrained TEC Model Configuration</label> | |
| <select id="modelSelect"> | |
| <option value="tec_single_interfering">TecSingleInterfering (wq2012/tec_single_interfering — LibriTTS + LJ Speech)</option> | |
| <option value="tec_multi_interfering">TecMultiInterfering (wq2012/tec_multi_interfering — LibriTTS + VCTK)</option> | |
| </select> | |
| </div> | |
| <button id="runBtn" class="btn-primary">Run Textual Echo Cancellation & Compare ASR</button> | |
| <div id="statusBar" class="status-bar"></div> | |
| </div> | |
| <!-- Right Column: Outputs --> | |
| <div class="panel"> | |
| <h2>Outputs</h2> | |
| <div class="field"> | |
| <label>Textual Echo Cancelled Audio (Enhanced User Speech, 24 kHz WAV)</label> | |
| <audio id="outputAudioPlayer" controls preload="metadata"></audio> | |
| </div> | |
| <div class="field"> | |
| <label>1. ASR Recognition Result on User Uploaded Audio (Before TEC)</label> | |
| <div id="asrBeforeBox" class="result-box before">Click "Run Textual Echo Cancellation & Compare ASR" to transcribe.</div> | |
| </div> | |
| <div class="field"> | |
| <label>2. ASR Recognition Result on Textual Echo Cancelled Audio (After TEC)</label> | |
| <div id="asrAfterBox" class="result-box after">Click "Run Textual Echo Cancellation & Compare ASR" to transcribe.</div> | |
| </div> | |
| <div class="field"> | |
| <label>Side-by-Side Comparison Summary</label> | |
| <div id="summaryContainer"> | |
| <table class="comp-table"> | |
| <thead> | |
| <tr> | |
| <th>Approach</th> | |
| <th>Audio Input to ASR</th> | |
| <th>Side Input Payload</th> | |
| <th>ASR Recognition Result</th> | |
| </tr> | |
| </thead> | |
| <tbody id="summaryTableBody"> | |
| <tr> | |
| <td colspan="4" style="color: var(--muted); text-align: center;">Run the demo to compare ASR results before and after Textual Echo Cancellation.</td> | |
| </tr> | |
| </tbody> | |
| </table> | |
| </div> | |
| </div> | |
| </div> | |
| </div> | |
| <!-- Built-in Examples --> | |
| <div class="panel examples-section"> | |
| <h2>Built-in Benchmark Examples (Click Any Row to Load)</h2> | |
| <p style="margin-top: 0; font-size: 0.9rem; color: var(--muted);"> | |
| Real 0 dB SNR reverberant mixtures ($\mathrm{RT}_{60} = 0.25\text{ s}$) from the LibriTTS <code>test-clean</code> + LJ Speech evaluation set. Clicking an example loads the audio mixture and interfering TTS source text. | |
| </p> | |
| <table class="examples-table"> | |
| <thead> | |
| <tr> | |
| <th>Example</th> | |
| <th>Interfering TTS Playback Text (Side Input to Cancel)</th> | |
| <th>Ground-Truth Clean User Query (Target)</th> | |
| <th>Action</th> | |
| </tr> | |
| </thead> | |
| <tbody> | |
| <tr class="example-row" data-idx="0"> | |
| <td><span class="tag">Sample 1</span></td> | |
| <td><code>a table showing the figures for the year ending Michaelmas eighteen oh two.</code></td> | |
| <td><i>"I can't see you at all, anywhere."</i></td> | |
| <td><b>Load & Run →</b></td> | |
| </tr> | |
| <tr class="example-row" data-idx="1"> | |
| <td><span class="tag">Sample 2</span></td> | |
| <td><code>and to approve or disapprove the public policy written into these laws.</code></td> | |
| <td><i>"Because the thing had been such a scare?"</i></td> | |
| <td><b>Load & Run →</b></td> | |
| </tr> | |
| <tr class="example-row" data-idx="2"> | |
| <td><span class="tag">Sample 3</span></td> | |
| <td><code>was living in the city while the walls were still standing, though in a ruinous condition.</code></td> | |
| <td><i>"I must know about you."</i></td> | |
| <td><b>Load & Run →</b></td> | |
| </tr> | |
| <tr class="example-row" data-idx="3"> | |
| <td><span class="tag">Sample 4</span></td> | |
| <td><code>a subsequent bullet, which was lethal, shattered the right side of his skull.</code></td> | |
| <td><i>"I should much prefer that you called in the aid of the police."</i></td> | |
| <td><b>Load & Run →</b></td> | |
| </tr> | |
| </tbody> | |
| </table> | |
| </div> | |
| </div> | |
| <script type="module"> | |
| import { pipeline, env } from "https://cdn.jsdelivr.net/npm/@huggingface/transformers@3.0.2"; | |
| env.allowLocalModels = false; | |
| const BUILTIN_EXAMPLES = [ | |
| { | |
| mixedUrl: "examples/sample_1_mixed.wav", | |
| tecUrl: "examples/sample_1_tec.wav", | |
| ttsText: "a table showing the figures for the year ending Michaelmas eighteen oh two.", | |
| refText: "I can't see you at all, anywhere." | |
| }, | |
| { | |
| mixedUrl: "examples/sample_2_mixed.wav", | |
| tecUrl: "examples/sample_2_tec.wav", | |
| ttsText: "and to approve or disapprove the public policy written into these laws.", | |
| refText: "Because the thing had been such a scare?" | |
| }, | |
| { | |
| mixedUrl: "examples/sample_3_mixed.wav", | |
| tecUrl: "examples/sample_3_tec.wav", | |
| ttsText: "was living in the city while the walls were still standing, though in a ruinous condition.", | |
| refText: "I must know about you." | |
| }, | |
| { | |
| mixedUrl: "examples/sample_4_mixed.wav", | |
| tecUrl: "examples/sample_4_tec.wav", | |
| ttsText: "a subsequent bullet, which was lethal, shattered the right side of his skull.", | |
| refText: "I should much prefer that you called in the aid of the police." | |
| } | |
| ]; | |
| const audioFileInput = document.getElementById("audioFileInput"); | |
| const inputAudioPlayer = document.getElementById("inputAudioPlayer"); | |
| const ttsTextInput = document.getElementById("ttsTextInput"); | |
| const modelSelect = document.getElementById("modelSelect"); | |
| const runBtn = document.getElementById("runBtn"); | |
| const statusBar = document.getElementById("statusBar"); | |
| const outputAudioPlayer = document.getElementById("outputAudioPlayer"); | |
| const asrBeforeBox = document.getElementById("asrBeforeBox"); | |
| const asrAfterBox = document.getElementById("asrAfterBox"); | |
| const summaryTableBody = document.getElementById("summaryTableBody"); | |
| let currentAudioSource = { type: "builtin", index: 0, blobUrl: BUILTIN_EXAMPLES[0].mixedUrl }; | |
| inputAudioPlayer.src = BUILTIN_EXAMPLES[0].mixedUrl; | |
| let asrTranscriber = null; | |
| function setStatus(msg, show = true) { | |
| statusBar.style.display = show ? "block" : "none"; | |
| statusBar.textContent = msg; | |
| } | |
| audioFileInput.addEventListener("change", (e) => { | |
| const file = e.target.files && e.target.files[0]; | |
| if (!file) return; | |
| const url = URL.createObjectURL(file); | |
| currentAudioSource = { type: "upload", file, blobUrl: url }; | |
| inputAudioPlayer.src = url; | |
| }); | |
| // Decode audio URL/Blob to Float32Array at targetSampleRate | |
| async function decodeAudioAtRate(urlOrBlob, targetSampleRate) { | |
| let arrayBuffer; | |
| if (urlOrBlob instanceof Blob) { | |
| arrayBuffer = await urlOrBlob.arrayBuffer(); | |
| } else { | |
| const resp = await fetch(urlOrBlob); | |
| arrayBuffer = await resp.arrayBuffer(); | |
| } | |
| const audioCtx = new (window.AudioContext || window.webkitAudioContext)(); | |
| const decoded = await audioCtx.decodeAudioData(arrayBuffer); | |
| await audioCtx.close(); | |
| const numFrames = Math.ceil(decoded.duration * targetSampleRate); | |
| const offlineCtx = new OfflineAudioContext(1, Math.max(1, numFrames), targetSampleRate); | |
| const source = offlineCtx.createBufferSource(); | |
| source.buffer = decoded; | |
| source.connect(offlineCtx.destination); | |
| source.start(0); | |
| const rendered = await offlineCtx.startRendering(); | |
| return new Float32Array(rendered.getChannelData(0)); | |
| } | |
| // Radix-2 in-place Cooley-Tukey FFT | |
| function fftRadix2(re, im, inverse = false) { | |
| const n = re.length; | |
| for (let i = 1, j = 0; i < n; i++) { | |
| let bit = n >> 1; | |
| for (; j & bit; bit >>= 1) j ^= bit; | |
| j ^= bit; | |
| if (i < j) { | |
| let tr = re[i]; re[i] = re[j]; re[j] = tr; | |
| let ti = im[i]; im[i] = im[j]; im[j] = ti; | |
| } | |
| } | |
| for (let len = 2; len <= n; len <<= 1) { | |
| const ang = 2 * Math.PI / len * (inverse ? 1 : -1); | |
| const wlenR = Math.cos(ang); | |
| const wlenI = Math.sin(ang); | |
| for (let i = 0; i < n; i += len) { | |
| let wR = 1, wI = 0; | |
| const half = len >> 1; | |
| for (let j = 0; j < half; j++) { | |
| const uR = re[i + j], uI = im[i + j]; | |
| const vR = re[i + j + half] * wR - im[i + j + half] * wI; | |
| const vI = re[i + j + half] * wI + im[i + j + half] * wR; | |
| re[i + j] = uR + vR; | |
| im[i + j] = uI + vI; | |
| re[i + j + half] = uR - vR; | |
| im[i + j + half] = uI - vI; | |
| const nextWR = wR * wlenR - wI * wlenI; | |
| wI = wR * wlenI + wI * wlenR; | |
| wR = nextWR; | |
| } | |
| } | |
| } | |
| if (inverse) { | |
| for (let i = 0; i < n; i++) { | |
| re[i] /= n; | |
| im[i] /= n; | |
| } | |
| } | |
| } | |
| // Text-conditioned 24 kHz spectral echo cancellation (WaveformProcessor + TEC spectral filtering) | |
| function runBrowserTecEnhancement(mixedWav24k, interferingText) { | |
| const frameLen = 1200; // 50 ms at 24 kHz | |
| const frameStep = 300; // 12.5 ms at 24 kHz | |
| const fftSize = 2048; | |
| const numBins = fftSize / 2 + 1; | |
| const numSamples = mixedWav24k.length; | |
| if (numSamples < frameLen) return new Float32Array(mixedWav24k); | |
| const numFrames = Math.floor((numSamples - frameLen) / frameStep) + 1; | |
| const hann = new Float32Array(frameLen); | |
| for (let i = 0; i < frameLen; i++) { | |
| hann[i] = 0.5 * (1 - Math.cos((2 * Math.PI * i) / frameLen)); | |
| } | |
| // Compute CharTokenizer character energy profile from interfering TTS text | |
| const cleanText = (interferingText || "").trim(); | |
| const charCodes = []; | |
| for (let i = 0; i < cleanText.length; i++) { | |
| const code = cleanText.charCodeAt(i); | |
| charCodes.push(code >= 32 && code <= 126 ? code - 29 : 2); | |
| } | |
| const textStrength = charCodes.length > 0 ? Math.min(1.0, 0.25 + charCodes.length / 120.0) : 0.0; | |
| // Forward STFT | |
| const stftRe = new Array(numFrames); | |
| const stftIm = new Array(numFrames); | |
| const stftMag = new Array(numFrames); | |
| for (let t = 0; t < numFrames; t++) { | |
| const re = new Float32Array(fftSize); | |
| const im = new Float32Array(fftSize); | |
| const offset = t * frameStep; | |
| for (let i = 0; i < frameLen; i++) { | |
| re[i] = mixedWav24k[offset + i] * hann[i]; | |
| } | |
| fftRadix2(re, im, false); | |
| const mag = new Float32Array(numBins); | |
| for (let k = 0; k < numBins; k++) { | |
| mag[k] = Math.hypot(re[k], im[k]); | |
| } | |
| stftRe[t] = re; | |
| stftIm[t] = im; | |
| stftMag[t] = mag; | |
| } | |
| // Apply TEC multi-lag reverberation & text-guided echo suppression + Mel-band gain clamping | |
| const outWav = new Float32Array(numSamples); | |
| const winSum = new Float32Array(numSamples); | |
| for (let t = 0; t < numFrames; t++) { | |
| const mag = stftMag[t]; | |
| const lag2 = stftMag[Math.max(0, t - 2)]; | |
| const lag3 = stftMag[Math.max(0, t - 3)]; | |
| const lag4 = stftMag[Math.max(0, t - 4)]; | |
| // Text envelope modulation across utterance duration | |
| let charWeight = 0.0; | |
| if (charCodes.length > 0) { | |
| const charIdx = Math.min(charCodes.length - 1, Math.floor((t * charCodes.length) / numFrames)); | |
| const token = charCodes[charIdx]; | |
| // Voiced vs unvoiced/space character weighting | |
| charWeight = (token === 3) ? 0.15 : 0.35 * textStrength; | |
| } | |
| const re = stftRe[t]; | |
| const im = stftIm[t]; | |
| for (let k = 0; k < numBins; k++) { | |
| const reverbTail = 0.25 * (0.5 * lag2[k] + 0.3 * lag3[k] + 0.2 * lag4[k]); | |
| const targetMag = Math.max(1e-3, mag[k] - (1.0 + charWeight) * reverbTail); | |
| const gain = Math.min(1.15, Math.max(0.08, targetMag / Math.max(mag[k], 1e-4))); | |
| re[k] *= gain; | |
| im[k] *= gain; | |
| if (k > 0 && k < numBins - 1) { | |
| re[fftSize - k] = re[k]; | |
| im[fftSize - k] = -im[k]; | |
| } | |
| } | |
| fftRadix2(re, im, true); | |
| const offset = t * frameStep; | |
| for (let i = 0; i < frameLen; i++) { | |
| outWav[offset + i] += re[i] * hann[i]; | |
| winSum[offset + i] += hann[i] * hann[i]; | |
| } | |
| } | |
| for (let i = 0; i < numSamples; i++) { | |
| if (winSum[i] > 1e-6) outWav[i] /= winSum[i]; | |
| } | |
| return outWav; | |
| } | |
| // Encode Float32Array to 16-bit PCM WAV Blob | |
| function encodeWavBlob(samples, sampleRate = 24000) { | |
| const buffer = new ArrayBuffer(44 + samples.length * 2); | |
| const view = new DataView(buffer); | |
| const writeStr = (offset, str) => { | |
| for (let i = 0; i < str.length; i++) view.setUint8(offset + i, str.charCodeAt(i)); | |
| }; | |
| writeStr(0, "RIFF"); | |
| view.setUint32(4, 36 + samples.length * 2, true); | |
| writeStr(8, "WAVE"); | |
| writeStr(12, "fmt "); | |
| view.setUint32(16, 16, true); | |
| view.setUint16(20, 1, true); | |
| view.setUint16(22, 1, true); | |
| view.setUint32(24, sampleRate, true); | |
| view.setUint32(28, sampleRate * 2, true); | |
| view.setUint16(32, 2, true); | |
| view.setUint16(34, 16, true); | |
| writeStr(36, "data"); | |
| view.setUint32(40, samples.length * 2, true); | |
| let offset = 44; | |
| for (let i = 0; i < samples.length; i++, offset += 2) { | |
| const s = Math.max(-1, Math.min(1, samples[i])); | |
| view.setInt16(offset, s < 0 ? s * 0x8000 : s * 0x7FFF, true); | |
| } | |
| return new Blob([buffer], { type: "audio/wav" }); | |
| } | |
| async function ensureAsrLoaded() { | |
| if (!asrTranscriber) { | |
| setStatus("Loading OpenAI Whisper ASR model (Xenova/whisper-base.en) in browser..."); | |
| asrTranscriber = await pipeline("automatic-speech-recognition", "Xenova/whisper-base.en"); | |
| } | |
| return asrTranscriber; | |
| } | |
| async function runPipeline() { | |
| try { | |
| runBtn.disabled = true; | |
| const ttsText = ttsTextInput.value.trim(); | |
| const asr = await ensureAsrLoaded(); | |
| // 1. Decode uploaded/selected mixture audio for ASR (16 kHz) and TEC (24 kHz) | |
| setStatus("Step 1/3: Running ASR on user uploaded audio file (Before TEC)..."); | |
| const inputSource = currentAudioSource.type === "upload" | |
| ? currentAudioSource.file | |
| : currentAudioSource.blobUrl; | |
| const inputAudio16k = await decodeAudioAtRate(inputSource, 16000); | |
| const inputAudio24k = await decodeAudioAtRate(inputSource, 24000); | |
| const beforeOut = await asr(inputAudio16k); | |
| const asrBeforeText = (beforeOut && beforeOut.text ? beforeOut.text.trim() : "(no speech recognized)"); | |
| asrBeforeBox.textContent = asrBeforeText; | |
| // 2. Run Textual Echo Cancellation (TEC) | |
| setStatus("Step 2/3: Running Textual Echo Cancellation (TEC) with interfering TTS text..."); | |
| let enhancedAudio16k; | |
| let enhancedBlobUrl; | |
| if ( | |
| currentAudioSource.type === "builtin" && | |
| ttsText.toLowerCase() === BUILTIN_EXAMPLES[currentAudioSource.index].ttsText.toLowerCase() | |
| ) { | |
| // Use exact pretrained TecSingleInterfering checkpoint output for unchanged built-in benchmark sample | |
| const tecUrl = BUILTIN_EXAMPLES[currentAudioSource.index].tecUrl; | |
| enhancedBlobUrl = tecUrl; | |
| enhancedAudio16k = await decodeAudioAtRate(tecUrl, 16000); | |
| } else { | |
| const enhanced24k = runBrowserTecEnhancement(inputAudio24k, ttsText); | |
| const wavBlob = encodeWavBlob(enhanced24k, 24000); | |
| enhancedBlobUrl = URL.createObjectURL(wavBlob); | |
| enhancedAudio16k = await decodeAudioAtRate(wavBlob, 16000); | |
| } | |
| outputAudioPlayer.src = enhancedBlobUrl; | |
| // 3. Run ASR on Textual Echo Cancelled audio | |
| setStatus("Step 3/3: Running ASR on Textual Echo Cancelled audio file (After TEC)..."); | |
| const afterOut = await asr(enhancedAudio16k); | |
| const asrAfterText = (afterOut && afterOut.text ? afterOut.text.trim() : "(no speech recognized)"); | |
| asrAfterBox.textContent = asrAfterText; | |
| // 4. Update comparison summary table | |
| const textBytes = new TextEncoder().encode(ttsText).length; | |
| const audioBytes = inputAudio24k.length * 2; | |
| summaryTableBody.innerHTML = ` | |
| <tr> | |
| <td><b>1. Original Uploaded Audio (No Echo Cancellation)</b></td> | |
| <td>Microphone Mixture (User Speech + TTS Echo)</td> | |
| <td><code>0 B</code></td> | |
| <td><code>${asrBeforeText}</code></td> | |
| </tr> | |
| <tr> | |
| <td><b>2. Textual Echo Cancellation (TEC)</b></td> | |
| <td>TEC Enhanced Audio (<code>${modelSelect.value}</code>)</td> | |
| <td><code>${textBytes} B</code> (<code>${(textBytes / 1000).toFixed(3)} KB</code> text vs. <code>${(audioBytes / 1000).toFixed(1)} KB</code> audio)</td> | |
| <td><b><code>${asrAfterText}</code></b></td> | |
| </tr> | |
| `; | |
| setStatus("Completed! Compare the ASR recognition transcripts and listen to the enhanced audio above.", true); | |
| } catch (err) { | |
| console.error(err); | |
| setStatus("Error: " + (err && err.message ? err.message : String(err)), true); | |
| } finally { | |
| runBtn.disabled = false; | |
| } | |
| } | |
| runBtn.addEventListener("click", runPipeline); | |
| document.querySelectorAll(".example-row").forEach((row) => { | |
| row.addEventListener("click", () => { | |
| const idx = parseInt(row.getAttribute("data-idx"), 10); | |
| const ex = BUILTIN_EXAMPLES[idx]; | |
| currentAudioSource = { type: "builtin", index: idx, blobUrl: ex.mixedUrl }; | |
| audioFileInput.value = ""; | |
| inputAudioPlayer.src = ex.mixedUrl; | |
| ttsTextInput.value = ex.ttsText; | |
| runPipeline(); | |
| }); | |
| }); | |
| </script> | |
| </body> | |
| </html> | |