hushscribe / format.js
CyberMax-tools's picture
Rename products (unique names)
d0fb3e3 verified
Raw History Blame Contribute Delete
3.93 kB
// Hearlark: transcript formatting (TXT, SRT, VTT, JSON). Pure functions, no DOM, so they run in Node tests too.
// © 2026 CyberMax.
/** Seconds → "HH:MM:SS,mmm" (SRT) or "HH:MM:SS.mmm" (VTT). */
export function stamp(seconds, sep = ',') {
const ms = Math.max(0, Math.round((Number(seconds) || 0) * 1000));
const h = Math.floor(ms / 3600000);
const m = Math.floor((ms % 3600000) / 60000);
const s = Math.floor((ms % 60000) / 1000);
const r = ms % 1000;
const p = (n, w = 2) => String(n).padStart(w, '0');
return `${p(h)}:${p(m)}:${p(s)}${sep}${p(r, 3)}`;
}
/** Short clock for the on-page list: "m:ss" or "h:mm:ss". */
export function clock(seconds) {
const t = Math.max(0, Math.floor(Number(seconds) || 0));
const h = Math.floor(t / 3600);
const m = Math.floor((t % 3600) / 60);
const s = t % 60;
return h ? `${h}:${String(m).padStart(2, '0')}:${String(s).padStart(2, '0')}` : `${m}:${String(s).padStart(2, '0')}`;
}
/**
* Normalises the ASR pipeline output ({ text, chunks: [{ timestamp: [start, end], text }] }) into clean segments.
* Fixes what Whisper output commonly has: null end times, overlapping or backwards times, empty or
* whitespace-only chunks and runs of the same line repeated (a known Whisper hallucination on silence).
*/
export function toSegments(result, duration = null) {
const chunks = Array.isArray(result?.chunks) ? result.chunks : [];
const out = [];
for (const c of chunks) {
const text = String(c?.text ?? '').replace(/\s+/g, ' ').trim();
if (!text) continue;
let [start, end] = Array.isArray(c.timestamp) ? c.timestamp : [null, null];
start = Number.isFinite(start) ? start : (out.length ? out[out.length - 1].end : 0);
if (out.length && start < out[out.length - 1].end) start = out[out.length - 1].end;
if (!Number.isFinite(end) || end <= start) end = Number.isFinite(duration) && duration > start ? Math.min(duration, start + 10) : start + 2;
const prev = out[out.length - 1];
if (prev && prev.text === text && start - prev.end < 1) { prev.end = end; continue; }
out.push({ start: +start.toFixed(3), end: +end.toFixed(3), text });
}
if (!out.length && String(result?.text ?? '').trim()) {
out.push({ start: 0, end: Number.isFinite(duration) && duration > 0 ? +duration.toFixed(3) : 0, text: String(result.text).replace(/\s+/g, ' ').trim() });
}
return out;
}
export function toTxt(segments, { timestamps = false } = {}) {
if (!timestamps) return segments.map((s) => s.text).join(' ').replace(/\s+([,.!?;:])/g, '$1').trim() + '\n';
return segments.map((s) => `[${clock(s.start)}] ${s.text}`).join('\n') + '\n';
}
export function toSrt(segments) {
return segments.map((s, i) => `${i + 1}\n${stamp(s.start)} --> ${stamp(s.end)}\n${s.text}\n`).join('\n');
}
export function toVtt(segments) {
return 'WEBVTT\n\n' + segments.map((s) => `${stamp(s.start, '.')} --> ${stamp(s.end, '.')}\n${s.text}\n`).join('\n');
}
export function toJson(segments, meta = {}) {
return JSON.stringify({ ...meta, segments, text: toTxt(segments).trim() }, null, 2) + '\n';
}
/** Approximate number of 30 s windows the pipeline will decode (20 s step with a 5 s stride each side). */
export function expectedChunks(duration) {
const d = Number(duration) || 0;
return d <= 30 ? 1 : 1 + Math.ceil((d - 30) / 20);
}
/** Mixes decoded channels down to one Float32Array. */
export function toMono(channels) {
if (!channels.length) return new Float32Array(0);
if (channels.length === 1) return channels[0];
const out = new Float32Array(channels[0].length);
for (const ch of channels) for (let i = 0; i < out.length; i++) out[i] += ch[i] / channels.length;
return out;
}
/** "My Talk (final).mp3" → "My Talk (final)" for download names. */
export function baseName(name) {
const s = String(name || 'transcript').replace(/\.[^.\/\\]+$/, '').replace(/[\\/:*?"<>|]+/g, '-').trim();
return s || 'transcript';
}