Download src/video-engine/physics/captionSegmentation.ts from asnannp/DocDoeAI: direct link, hf CLI and curl.
- Browser
- Download file 14.4 kB
-
https://huggingface.co/spaces/asnannp/DocDoeAI/resolve/main/src/video-engine/physics/captionSegmentation.ts
- Command line
-
hf download hf://spaces/asnannp/DocDoeAI/src/video-engine/physics/captionSegmentation.ts
-
curl -L -o captionSegmentation.ts https://huggingface.co/spaces/asnannp/DocDoeAI/resolve/main/src/video-engine/physics/captionSegmentation.ts
14.4 kB
| // Caption segmentation for the Physics production engine. | |
| // | |
| // Goal: turn a narration chunk's full text into short, natural, readable caption | |
| // cues that never break a fixed scientific phrase, never split a formula or a | |
| // number from its unit, never leave a one-word orphan, and never end a cue on a | |
| // dangling conjunction that makes the sentence look incomplete on screen. | |
| // | |
| // This is intentionally deterministic and dependency-free so it can run inside | |
| // the render-package build step and inside caption QA. | |
| export type CaptionCue = { | |
| id: string; | |
| sceneId?: string; | |
| startSecond: number; | |
| endSecond: number; | |
| text: string; | |
| }; | |
| // Multi-word scientific phrases that must stay on one caption line together. | |
| // Longest phrases first so greedy matching prefers the most specific term. | |
| export const PROTECTED_PHRASES: string[] = [ | |
| "induced potential difference", | |
| "electromagnetic induction", | |
| "compression and rarefaction", | |
| "centre of curvature", | |
| "right-hand thumb rule", | |
| "first-order lever", | |
| "second-order lever", | |
| "third-order lever", | |
| "potential difference", | |
| "principal focus", | |
| "principal axis", | |
| "optical centre", | |
| "optic centre", | |
| "focal length", | |
| "longitudinal wave", | |
| "transverse wave", | |
| "electric current", | |
| "electric power", | |
| "electric energy", | |
| "magnetic field", | |
| "induced current", | |
| "mechanical advantage", | |
| "velocity ratio", | |
| "effort arm", | |
| "load arm", | |
| "convex lens", | |
| "concave lens", | |
| "real image", | |
| "virtual image", | |
| "inverted image", | |
| "erect image", | |
| "diminished image", | |
| "magnified image", | |
| "lens formula", | |
| "sign convention", | |
| "power of a lens", | |
| "angle of incidence", | |
| "angle of refraction", | |
| "speed of sound", | |
| "human ear", | |
| ].sort((a, b) => b.split(/\s+/).length - a.split(/\s+/).length); | |
| // Units that must stay attached to the number in front of them. | |
| const UNITS = new Set([ | |
| "cm", "mm", "m", "km", "s", "ms", "hz", "khz", "w", "kw", "v", "kv", | |
| "a", "ma", "ω", "ohm", "ohms", "j", "joule", "joules", "n", "kg", "g", | |
| "dioptre", "dioptres", "d", "m/s", "°", "%", | |
| ]); | |
| // Words a caption cue should not END on: ending here makes the sentence look | |
| // cut off mid-thought on screen. | |
| const WEAK_TRAILING = new Set([ | |
| "and", "or", "but", "the", "a", "an", "to", "of", "for", "with", "by", "in", | |
| "on", "at", "as", "is", "are", "was", "were", "its", "their", "that", "which", | |
| "when", "while", "than", "then", "so", "we", "will", "you", "be", "able", | |
| "this", "these", "those", "into", "from", "if", "it", "our", "your", "his", | |
| "her", "not", "no", "how", "why", "what", "up", "out", "about", "over", | |
| ]); | |
| const MIN_WORDS = 4; | |
| const SOFT_MAX_WORDS = 10; | |
| const HARD_MAX_WORDS = 13; // only reached when a long protected phrase/formula forces it | |
| function hasMathChar(token: string): boolean { | |
| return /[=/×·²³]/.test(token) || token === "−"; | |
| } | |
| function stripPunct(token: string): string { | |
| return token.replace(/[.,;:!?…—–)"']+$/g, "").replace(/^["'(]+/g, "").toLowerCase(); | |
| } | |
| function trailingPunct(token: string): "sentence" | "clause" | null { | |
| if (/[.!?…]$/.test(token)) return "sentence"; | |
| if (/[,;:—–]$/.test(token)) return "clause"; | |
| return null; | |
| } | |
| type Atom = { | |
| text: string; | |
| words: number; // word count for cap accounting | |
| boundary: "sentence" | "clause" | null; | |
| protectedOrFormula: boolean; | |
| weakEnd: boolean; // ends on a weak word (only relevant when boundary === null) | |
| startsConjunction: boolean; // first word is and/or/but | |
| }; | |
| // Break text into atoms: protected phrases, number+unit pairs, and formula runs | |
| // each collapse into a single indivisible atom. | |
| function toAtoms(text: string): Atom[] { | |
| const tokens = text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean); | |
| const atoms: Atom[] = []; | |
| let i = 0; | |
| while (i < tokens.length) { | |
| // 1) Protected phrase match (case-insensitive, punctuation-tolerant). | |
| let matched = false; | |
| for (const phrase of PROTECTED_PHRASES) { | |
| const parts = phrase.split(" "); | |
| if (i + parts.length > tokens.length) continue; | |
| const windowWords = tokens.slice(i, i + parts.length).map(stripPunct); | |
| // Allow a plural on the final word ("concave lens" also matches "concave lenses"). | |
| const isMatch = windowWords.every((w, idx) => | |
| idx < parts.length - 1 | |
| ? w === parts[idx] | |
| : w === parts[idx] || w === `${parts[idx]}s` || w === `${parts[idx]}es`); | |
| if (isMatch) { | |
| const raw = tokens.slice(i, i + parts.length).join(" "); | |
| atoms.push({ | |
| text: raw, | |
| words: parts.length, | |
| boundary: trailingPunct(raw), | |
| protectedOrFormula: true, | |
| weakEnd: false, | |
| startsConjunction: false, | |
| }); | |
| i += parts.length; | |
| matched = true; | |
| break; | |
| } | |
| } | |
| if (matched) continue; | |
| // 2) Formula run: consecutive tokens carrying math characters / operators / | |
| // single-letter variables glued by them. | |
| if (hasMathChar(tokens[i]) || (/^[a-zA-Z]$/.test(tokens[i]) && i + 1 < tokens.length && hasMathChar(tokens[i + 1]))) { | |
| let j = i; | |
| const runTokens: string[] = []; | |
| while (j < tokens.length) { | |
| const t = tokens[j]; | |
| const isOperatorOrVar = hasMathChar(t) || /^[a-zA-Z]$/.test(t) || /^[a-zA-Z]?\d/.test(t) || /\d/.test(t); | |
| if (runTokens.length === 0 || isOperatorOrVar) { | |
| runTokens.push(t); | |
| j += 1; | |
| if (trailingPunct(t)) break; // formula ended a clause/sentence | |
| } else break; | |
| } | |
| const raw = runTokens.join(" "); | |
| atoms.push({ | |
| text: raw, | |
| words: Math.max(2, Math.min(runTokens.length, 3)), | |
| boundary: trailingPunct(raw), | |
| protectedOrFormula: true, | |
| weakEnd: false, | |
| startsConjunction: false, | |
| }); | |
| i = j; | |
| continue; | |
| } | |
| // 3) Number + unit pair (e.g. "20 cm", "50 Hz"). | |
| if (/^\d[\d.,]*$/.test(stripPunct(tokens[i])) && i + 1 < tokens.length && UNITS.has(stripPunct(tokens[i + 1]))) { | |
| const raw = `${tokens[i]} ${tokens[i + 1]}`; | |
| atoms.push({ | |
| text: raw, | |
| words: 2, | |
| boundary: trailingPunct(tokens[i + 1]), | |
| protectedOrFormula: true, | |
| weakEnd: false, | |
| startsConjunction: false, | |
| }); | |
| i += 2; | |
| continue; | |
| } | |
| // 4) Plain word. | |
| const t = tokens[i]; | |
| const bare = stripPunct(t); | |
| atoms.push({ | |
| text: t, | |
| words: 1, | |
| boundary: trailingPunct(t), | |
| protectedOrFormula: false, | |
| weakEnd: WEAK_TRAILING.has(bare), | |
| startsConjunction: bare === "and" || bare === "or" || bare === "but", | |
| }); | |
| i += 1; | |
| } | |
| return atoms; | |
| } | |
| // Split a narration chunk into sentences without breaking decimals ("2.5"), | |
| // formulas ("1/f"), or abbreviations glued to digits. A sentence ends on | |
| // . ! ? … followed by whitespace and a capital letter or digit. | |
| function splitSentences(text: string): string[] { | |
| return text | |
| .replace(/\s+/g, " ") | |
| .trim() | |
| .split(/(?<=[.!?…])\s+(?=[A-Z0-9])/) | |
| .map((s) => s.trim()) | |
| .filter(Boolean); | |
| } | |
| const wordsOf = (group: Atom[]) => group.reduce((sum, a) => sum + a.words, 0); | |
| // Segment a single sentence's atoms into cues. Every cue here stays inside the | |
| // sentence, so no cue ever crosses a full stop. | |
| function segmentSentence(atoms: Atom[]): Atom[][] { | |
| if (!atoms.length) return []; | |
| // Keep a whole sentence on one cue unless it is genuinely long. Completeness | |
| // beats brevity — a student should read a full thought, not a fragment. | |
| if (wordsOf(atoms) <= HARD_MAX_WORDS) return [atoms]; | |
| const cues: Atom[][] = []; | |
| let current: Atom[] = []; | |
| let words = 0; | |
| for (let k = 0; k < atoms.length; k += 1) { | |
| const atom = atoms[k]; | |
| current.push(atom); | |
| words += atom.words; | |
| const next = atoms[k + 1]; | |
| const isLast = !next; | |
| const endsClause = atom.boundary === "clause"; | |
| const nextIsConjunction = next?.startsConjunction ?? false; | |
| const remaining = atoms.slice(k + 1).reduce((sum, a) => sum + a.words, 0); | |
| if (isLast) break; // sentence terminal — always closes below | |
| let close = false; | |
| if (endsClause && words >= MIN_WORDS && remaining >= MIN_WORDS && !nextIsConjunction) { | |
| close = true; | |
| } else if (words >= SOFT_MAX_WORDS && !atom.weakEnd && !nextIsConjunction) { | |
| close = true; | |
| } else if (words >= HARD_MAX_WORDS) { | |
| close = true; // safety valve for a long formula-heavy clause | |
| } | |
| if (close) { | |
| cues.push(current); | |
| current = []; | |
| words = 0; | |
| } | |
| } | |
| if (current.length) cues.push(current); | |
| // Merge a trailing incomplete fragment (< MIN_WORDS) back into its neighbour | |
| // within the same sentence — never leave a one/two-word dangling line. | |
| for (let k = cues.length - 1; k >= 0; k -= 1) { | |
| if (wordsOf(cues[k]) < MIN_WORDS && cues.length > 1) { | |
| if (k > 0) { | |
| cues[k - 1] = cues[k - 1].concat(cues[k]); | |
| cues.splice(k, 1); | |
| } else { | |
| cues[1] = cues[0].concat(cues[1]); | |
| cues.splice(0, 1); | |
| } | |
| } | |
| } | |
| // Pull a dangling weak word forward so no non-final cue ends on "and", "to"… | |
| for (let k = 0; k < cues.length - 1; k += 1) { | |
| let last = cues[k][cues[k].length - 1]; | |
| while (cues[k].length > MIN_WORDS && last && last.weakEnd && last.boundary === null) { | |
| const moved = cues[k].pop() as Atom; | |
| cues[k + 1].unshift(moved); | |
| last = cues[k][cues[k].length - 1]; | |
| } | |
| } | |
| return cues; | |
| } | |
| /** | |
| * Segment a narration chunk's full text into ordered caption cue strings. | |
| * Sentences are segmented independently, so a cue is always either a complete | |
| * short sentence or a clean clause — never a run across a full stop. | |
| */ | |
| export function segmentText(text: string): string[] { | |
| const sentences = splitSentences(text); | |
| const cues: Atom[][] = []; | |
| for (const sentence of sentences) { | |
| cues.push(...segmentSentence(toAtoms(sentence))); | |
| } | |
| // Merge any one-word cue (e.g. a lone "Example." sentence) into a neighbour so | |
| // no single-word line is ever shown — unless it is the only cue in the chunk. | |
| for (let k = cues.length - 1; k >= 0; k -= 1) { | |
| if (wordsOf(cues[k]) <= 1 && cues.length > 1) { | |
| if (k > 0) { | |
| cues[k - 1] = cues[k - 1].concat(cues[k]); | |
| cues.splice(k, 1); | |
| } else { | |
| cues[1] = cues[0].concat(cues[1]); | |
| cues.splice(0, 1); | |
| } | |
| } | |
| } | |
| return cues.map((group) => group.map((a) => a.text).join(" ").replace(/\s+/g, " ").trim()); | |
| } | |
| // Build a per-word timing map from the ORIGINAL cues. Each original cue is | |
| // already aligned to the spoken audio, so we linearly place its words across | |
| // its own [start, end] by character offset. This keeps new cue boundaries | |
| // anchored to real audio timing instead of guessing over the whole chunk. | |
| type WordTime = { word: string; start: number; end: number }; | |
| function wordTimeline(cues: CaptionCue[]): WordTime[] { | |
| const timeline: WordTime[] = []; | |
| for (const cue of cues) { | |
| const words = cue.text.replace(/\s+/g, " ").trim().split(" ").filter(Boolean); | |
| if (!words.length) continue; | |
| const span = Math.max(0.001, cue.endSecond - cue.startSecond); | |
| const totalChars = words.reduce((sum, w) => sum + w.length + 1, 0) || 1; | |
| let offset = 0; | |
| for (const word of words) { | |
| const wStart = cue.startSecond + span * (offset / totalChars); | |
| offset += word.length + 1; | |
| const wEnd = cue.startSecond + span * (offset / totalChars); | |
| timeline.push({ word, start: wStart, end: wEnd }); | |
| } | |
| } | |
| return timeline; | |
| } | |
| /** | |
| * Re-segment a group of caption cues that share one narration chunk into clean | |
| * cues. Timing is taken from the original word-level audio alignment so each new | |
| * cue appears and disappears in step with the spoken words. | |
| */ | |
| export function resegmentGroup(group: CaptionCue[]): CaptionCue[] { | |
| if (!group.length) return []; | |
| const sorted = [...group].sort((a, b) => a.startSecond - b.startSecond); | |
| const start = sorted[0].startSecond; | |
| const end = sorted[sorted.length - 1].endSecond; | |
| const sceneId = sorted[0].sceneId; | |
| const baseId = sorted[0].id.replace(/-\d+$/, ""); | |
| const fullText = sorted.map((c) => c.text.trim()).filter(Boolean).join(" "); | |
| const pieces = segmentText(fullText); | |
| if (!pieces.length) return sorted; | |
| const timeline = wordTimeline(sorted); | |
| const totalWords = timeline.length; | |
| // The re-segmented pieces contain exactly the same word sequence, so we can | |
| // walk the word timeline and read each new cue's real start/end from it. | |
| const pieceWordCounts = pieces.map((p) => p.replace(/\s+/g, " ").trim().split(" ").filter(Boolean).length); | |
| const alignable = pieceWordCounts.reduce((a, b) => a + b, 0) === totalWords && totalWords > 0; | |
| const span = Math.max(0.001, end - start); | |
| const totalChars = pieces.reduce((sum, p) => sum + p.length, 0) || 1; | |
| const out: CaptionCue[] = []; | |
| let wordCursor = 0; | |
| let charCursor = start; | |
| for (let index = 0; index < pieces.length; index += 1) { | |
| let cueStart: number; | |
| let cueEnd: number; | |
| if (alignable) { | |
| const first = timeline[wordCursor]; | |
| const lastIndex = wordCursor + pieceWordCounts[index] - 1; | |
| const last = timeline[Math.min(lastIndex, totalWords - 1)]; | |
| cueStart = index === 0 ? start : first.start; | |
| cueEnd = index === pieces.length - 1 ? end : last.end; | |
| wordCursor += pieceWordCounts[index]; | |
| } else { | |
| // Fallback: proportional by character length (rare — only if word counts | |
| // don't line up, e.g. exotic tokenisation). | |
| cueStart = charCursor; | |
| cueEnd = index === pieces.length - 1 ? end : charCursor + span * (pieces[index].length / totalChars); | |
| charCursor = cueEnd; | |
| } | |
| out.push({ id: `${baseId}-${index + 1}`, sceneId, startSecond: cueStart, endSecond: cueEnd, text: pieces[index] }); | |
| } | |
| return out; | |
| } | |
| /** | |
| * Re-segment a full flat caption array by grouping cues that belong to the same | |
| * narration chunk (id without the trailing "-N"). | |
| */ | |
| export function resegmentCaptions(captions: CaptionCue[]): CaptionCue[] { | |
| const groups = new Map<string, CaptionCue[]>(); | |
| const order: string[] = []; | |
| for (const cue of captions) { | |
| const key = cue.id.replace(/-\d+$/, ""); | |
| if (!groups.has(key)) { | |
| groups.set(key, []); | |
| order.push(key); | |
| } | |
| groups.get(key)!.push(cue); | |
| } | |
| const result: CaptionCue[] = []; | |
| for (const key of order) { | |
| result.push(...resegmentGroup(groups.get(key)!)); | |
| } | |
| return result; | |
| } | |