// Hearlark: transcript formatting (TXT, SRT, VTT, JSON). Pure functions, no DOM, so they run in Node tests too. // © 2026 CyberMax. /** Seconds → "HH:MM:SS,mmm" (SRT) or "HH:MM:SS.mmm" (VTT). */ export function stamp(seconds, sep = ',') { const ms = Math.max(0, Math.round((Number(seconds) || 0) * 1000)); const h = Math.floor(ms / 3600000); const m = Math.floor((ms % 3600000) / 60000); const s = Math.floor((ms % 60000) / 1000); const r = ms % 1000; const p = (n, w = 2) => String(n).padStart(w, '0'); return `${p(h)}:${p(m)}:${p(s)}${sep}${p(r, 3)}`; } /** Short clock for the on-page list: "m:ss" or "h:mm:ss". */ export function clock(seconds) { const t = Math.max(0, Math.floor(Number(seconds) || 0)); const h = Math.floor(t / 3600); const m = Math.floor((t % 3600) / 60); const s = t % 60; return h ? `${h}:${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}` : `${m}:${String(s).padStart(2, '0')}`; } /** * Normalises the ASR pipeline output ({ text, chunks: [{ timestamp: [start, end], text }] }) into clean segments. * Fixes what Whisper output commonly has: null end times, overlapping or backwards times, empty or * whitespace-only chunks and runs of the same line repeated (a known Whisper hallucination on silence). */ export function toSegments(result, duration = null) { const chunks = Array.isArray(result?.chunks) ? result.chunks : []; const out = []; for (const c of chunks) { const text = String(c?.text ?? '').replace(/\s+/g, ' ').trim(); if (!text) continue; let [start, end] = Array.isArray(c.timestamp) ? c.timestamp : [null, null]; start = Number.isFinite(start) ? start : (out.length ? out[out.length - 1].end : 0); if (out.length && start < out[out.length - 1].end) start = out[out.length - 1].end; if (!Number.isFinite(end) || end <= start) end = Number.isFinite(duration) && duration > start ? Math.min(duration, start + 10) : start + 2; const prev = out[out.length - 1]; if (prev && prev.text === text && start - prev.end < 1) { prev.end = end; continue; } out.push({ start: +start.toFixed(3), end: +end.toFixed(3), text }); } if (!out.length && String(result?.text ?? '').trim()) { out.push({ start: 0, end: Number.isFinite(duration) && duration > 0 ? +duration.toFixed(3) : 0, text: String(result.text).replace(/\s+/g, ' ').trim() }); } return out; } export function toTxt(segments, { timestamps = false } = {}) { if (!timestamps) return segments.map((s) => s.text).join(' ').replace(/\s+([,.!?;:])/g, '$1').trim() + '\n'; return segments.map((s) => `[${clock(s.start)}] ${s.text}`).join('\n') + '\n'; } export function toSrt(segments) { return segments.map((s, i) => `${i + 1}\n${stamp(s.start)} --> ${stamp(s.end)}\n${s.text}\n`).join('\n'); } export function toVtt(segments) { return 'WEBVTT\n\n' + segments.map((s) => `${stamp(s.start, '.')} --> ${stamp(s.end, '.')}\n${s.text}\n`).join('\n'); } export function toJson(segments, meta = {}) { return JSON.stringify({ ...meta, segments, text: toTxt(segments).trim() }, null, 2) + '\n'; } /** Approximate number of 30 s windows the pipeline will decode (20 s step with a 5 s stride each side). */ export function expectedChunks(duration) { const d = Number(duration) || 0; return d <= 30 ? 1 : 1 + Math.ceil((d - 30) / 20); } /** Mixes decoded channels down to one Float32Array. */ export function toMono(channels) { if (!channels.length) return new Float32Array(0); if (channels.length === 1) return channels[0]; const out = new Float32Array(channels[0].length); for (const ch of channels) for (let i = 0; i < out.length; i++) out[i] += ch[i] / channels.length; return out; } /** "My Talk (final).mp3" → "My Talk (final)" for download names. */ export function baseName(name) { const s = String(name || 'transcript').replace(/\.[^.\/\\]+$/, '').replace(/[\\/:*?"<>|]+/g, '-').trim(); return s || 'transcript'; }