#!/usr/bin/env python3 """Timed-script annotation engine (v2). WHY. Listening to the SFT+DPO model surfaced two failure modes: vocal bursts that run far too long and sound unnatural, and occasional hallucinated content that was never prompted. Both are BUDGET failures -- nothing in the v1 prompt told the model how long anything was allowed to take. v1 carried at most one optional `[7.3 seconds duration]` prefix for the whole clip and bare `(sigh)` cues with no duration at all. v2 renders the target text as a TIMED SCRIPT built from the word-level timestamps and the DETECTED burst spans, so every second of the clip is accounted for: [0.9 seconds pause] [3.4 seconds duration] Die Handelskammer betreibt diesen Test. [0.6 seconds pause] (contented sigh, 0.4 seconds) [1.2 seconds pause] [4.1 seconds duration] Die entwickeln und fuehren ihn durch, wissen Sie. Rules, each asserted by `verify()` on the EMITTED string: 1. `[D seconds duration]` precedes a speech segment and equals that SEGMENT's last-word offset minus its first-word onset -- never the clip's. 2. Segments break at sentence-final punctuation. A segment still longer than MAX_SEG_S is split again at its largest internal word gap, recursively. A 20 s two-sentence utterance therefore renders as `[12.0 ...] [8.0 ...] `. 3. Every gap >= PAUSE_MIN between consecutive events -- including before the first and after the last -- is emitted as `[G seconds pause]`. 4. Every detected burst is emitted at its position, as `(label, D seconds)`. 5. The emitted numbers sum to the clip duration. FOUR THINGS A NAIVE IMPLEMENTATION GETS WRONG, all measured on this corpus: * The transcript must not be rebuilt by joining word tokens. The forced aligner drops numerals ("154 Euro" aligns as "Euro"), so a join silently loses text on 9.9 % of real-speech rows. Words are ALIGNED to the original string by character offset and the original string is SLICED, so output text is byte-identical apart from inserted tags. * The script is built from DETECTED bursts, not from the parenthetical cues already sitting in `text_with_bursts`. On the voice-profile corpus those parentheticals are SYNTHESIS DIRECTIONS -- 39.4 % were never confirmed by the burst detector. A target text carrying a cue that is not in the audio teaches precisely the behaviour being removed. * 49.0 % of detected bursts OVERLAP an aligned word span. Adding such a burst's full length to the budget double-counts time -- it drove the sum 2.1 s past the clip length on the worst row of the first test. A burst's printed duration is its length MINUS its overlap with speech; when nothing is left it is printed WITHOUT a number, which is the honest statement that it is co-articulated and has no slot of its own. * A gap SHORTER than PAUSE_MIN is not worth a tag, but skipping it breaks the sum and the residual then reads as an error in the duration values. Such a gap is absorbed into the adjacent speech segment (never into a burst -- a burst tag must state the burst's own length). """ import re PAUSE_MIN = 0.20 # gaps below this are absorbed rather than tagged MAX_SEG_S = 12.0 # a speech segment longer than this is split again at its largest gap MIN_SPLIT_S = 2.0 # never create a segment shorter than this by splitting BURST_MIN_S = 0.05 # below this a burst prints without a duration _SENT_END = re.compile(r"[.!?…]+[\"'“”’\)\]]*$") def _fmt(x): return f"{x:.1f}" def align_words(text, words): """-> [(char_start, char_end, t_start, t_end)] for the word tokens locatable in `text`.""" out, pos = [], 0 for w in words or []: tok = (w.get("w") or "").strip() if not tok: continue try: s, e = float(w["s"]), float(w["e"]) except Exception: continue if not (e >= s >= -0.001): continue i = text.find(tok, pos) if i < 0: continue j = i + len(tok) out.append((i, j, s, e)) pos = j return out def _sentence_breaks(text, aw): br = {k for k, (i, j, s, e) in enumerate(aw) if _SENT_END.search(text[i:j])} br.add(len(aw) - 1) return br def _split_long(aw, lo, hi): dur = aw[hi][3] - aw[lo][2] if dur <= MAX_SEG_S or hi - lo < 1: return [(lo, hi)] best, bestscore = None, -1e9 for k in range(lo, hi): if (aw[k][3] - aw[lo][2]) < MIN_SPLIT_S or (aw[hi][3] - aw[k + 1][2]) < MIN_SPLIT_S: continue gap = aw[k + 1][2] - aw[k][3] score = gap - 0.02 * abs((aw[k][3] - aw[lo][2]) - dur / 2.0) if score > bestscore: bestscore, best = score, k if best is None: return [(lo, hi)] return _split_long(aw, lo, best) + _split_long(aw, best + 1, hi) def _free_len(s, e, spans): """length of [s,e) not covered by any interval in `spans` (sorted, disjoint).""" free = e - s for (a, b) in spans: if b <= s: continue if a >= e: break free -= min(e, b) - max(s, a) return max(free, 0.0) def build_events(text, words, bursts, dur_s): aw = align_words(text, words) bs = [] for t in bursts or []: try: s, e, lab = float(t[0]), float(t[1]), str(t[2] or "").strip() except Exception: continue if e <= s: continue bs.append((s, e, lab)) bs.sort() if not aw: ev = [{"kind": "burst", "s": s, "e": e, "label": l, "dur": e - s, "timed": True} for (s, e, l) in bs] _absorb(ev, dur_s) return ev, aw # 1. sentence segmentation br = _sentence_breaks(text, aw) segs, lo = [], 0 for k in range(len(aw)): if k in br: segs.append((lo, k)) lo = k + 1 # 2. every burst splits the segment it lands in, at the WORD GAP nearest its midpoint. # (Requiring the burst to sit entirely inside a gap would fail for the 49 % that overlap # a word, and those would then be emitted out of order.) cuts = set() for (s, e, l) in bs: mid = 0.5 * (s + e) for (lo, hi) in segs: if aw[lo][2] <= mid <= aw[hi][3] and hi > lo: k = min(range(lo, hi), key=lambda k: abs(0.5 * (aw[k][3] + aw[k + 1][2]) - mid)) cuts.add(k) break out, = [[]] for (lo, hi) in segs: a = lo for k in range(lo, hi): if k in cuts: out.append((a, k)) a = k + 1 out.append((a, hi)) segs = [x for x in out if x[0] <= x[1]] # 3. length cap out = [] for (lo, hi) in segs: out.extend(_split_long(aw, lo, hi)) segs = out spans = sorted((aw[lo][2], aw[hi][3]) for (lo, hi) in segs) ev = [{"kind": "speech", "s": aw[lo][2], "e": aw[hi][3], "lo": lo, "hi": hi} for (lo, hi) in segs] for (s, e, l) in bs: f = _free_len(s, e, spans) ev.append({"kind": "burst", "s": s, "e": e, "label": l, "dur": f, "timed": f >= BURST_MIN_S}) ev.sort(key=lambda d: (d["s"], 0 if d["kind"] == "speech" else 1)) _absorb(ev, dur_s) return ev, aw def _absorb(ev, dur_s): prev = 0.0 for k, e in enumerate(ev): if e["kind"] == "burst" and not e.get("timed"): continue # occupies no budget of its own gap = e["s"] - prev if 0 < gap < PAUSE_MIN: if e["kind"] == "speech": e["s"] = prev elif k > 0 and ev[k - 1]["kind"] == "speech": ev[k - 1]["e"] = e["s"] prev = max(prev, e["e"] if e["kind"] == "speech" else e["s"] + e["dur"]) if dur_s and ev: last = None for e in reversed(ev): if e["kind"] == "speech": last = e break if last is not None: tail = float(dur_s) - last["e"] if 0 < tail < PAUSE_MIN and last is ev[-1]: last["e"] = float(dur_s) def render(text, words, bursts, dur_s, timed=True, drop_bursts=False, trailing=True, blank_text=False): """-> (script, meta). `timed=False` keeps burst labels but omits every number. `blank_text=True` emits `...` in place of every SPEECH CHUNK and leaves every tag -- pauses, per-segment durations, burst labels and their lengths -- exactly where it was. It exists for the classifier-free-guidance DPO pairs, where the two clips of a pair say different words and the preference must therefore not be decidable from the words. Nothing else about the rendering changes: the segmentation, the numbers and their sum are computed from the real text and the real word timings, so the emitted budget still describes the real clip. """ text = text or "" ev, aw = build_events(text, words, bursts, dur_s) # `not aw` used to be part of this guard, which made the wordless branch of build_events # (line ~131, "no aligned words -> the events ARE the bursts") unreachable: a clip that is # nothing but a vocal burst rendered as its empty text instead of as `(chuckle, 0.5 seconds)`. # `not ev` is equivalent for every other input -- with no words and no bursts build_events # returns an empty list anyway -- so this only changes rows that are wordless AND carry a # burst: 709 of the 3,144,739 corpus rows, plus the standalone-burst training set. if not ev: head = ("..." if text.strip() else "") if blank_text else text.strip() return head, {"n_seg": 0, "n_burst": 0, "n_pause": 0, "n_burst_untimed": 0, "timed": False, "acc": 0.0} parts, t, cut = [], 0.0, 0 n_seg = n_burst = n_pause = n_untimed = 0 acc = 0.0 for k, e in enumerate(ev): if e["kind"] == "burst" and drop_bursts: continue # its span folds into the surrounding pause if e["kind"] == "burst" and not e["timed"]: parts.append(f"({e['label'].lower() or 'vocal burst'})") n_burst += 1 n_untimed += 1 continue gap = e["s"] - t if timed and gap >= PAUSE_MIN: parts.append(f"[{_fmt(gap)} seconds pause]") acc += round(gap, 1) n_pause += 1 if e["kind"] == "speech": is_last_speech = not any(x["kind"] == "speech" for x in ev[k + 1:]) end_char = len(text) if is_last_speech else aw[e["hi"]][1] chunk = text[cut:end_char].strip() cut = end_char if blank_text: chunk = "..." if chunk else "" d = e["e"] - e["s"] if timed: parts.append(f"[{_fmt(d)} seconds duration]") acc += round(d, 1) if chunk: parts.append(chunk) n_seg += 1 t = e["e"] else: lab = e["label"].lower() or "vocal burst" parts.append(f"({lab}, {_fmt(e['dur'])} seconds)" if timed else f"({lab})") if timed: acc += round(e["dur"], 1) n_burst += 1 t = max(t, e["s"] + e["dur"]) if timed and trailing and dur_s and (float(dur_s) - t) >= PAUSE_MIN: parts.append(f"[{_fmt(float(dur_s) - t)} seconds pause]") acc += round(float(dur_s) - t, 1) n_pause += 1 script = re.sub(r"\s{2,}", " ", " ".join(p for p in parts if p)).strip() return script, {"n_seg": n_seg, "n_burst": n_burst, "n_pause": n_pause, "n_burst_untimed": n_untimed, "timed": bool(timed), "acc": round(acc, 3)} # ---------------------------------------------------------------- independent verification _TAG = re.compile(r"\[(\d+\.\d) seconds (?:pause|duration)\]") _BUR = re.compile(r"\([^,()]*, (\d+\.\d) seconds\)") def verify(script, dur_s, text=None): """Re-parses the EMITTED string. Deliberately does not reuse the renderer's accumulator, so a bookkeeping error in the renderer cannot hide behind its own arithmetic.""" tot = sum(float(m.group(1)) for m in _TAG.finditer(script)) tot += sum(float(m.group(1)) for m in _BUR.finditer(script)) stripped = re.sub(r"\([^)]*\)", " ", re.sub(r"\[[^\]]*\]", " ", script)) stripped = re.sub(r"\s+", " ", stripped).strip() return {"sum": round(tot, 2), "dur": round(float(dur_s or 0), 2), "residual": round(float(dur_s or 0) - tot, 2), "text_out": stripped} # ---------------------------------------------------------------- script -> plan _ITEM = re.compile(r"\[(\d+\.\d) seconds (pause|duration)\]|\(([^()]*?), (\d+\.\d) seconds\)|\(([^()]*?)\)") def parse_script(script): """Read an emitted timed script back into a plan. Used by the GRPO reward: the prompt states WHERE each burst should fall and HOW LONG it and every utterance should be, and the reward has to compare that against what the detector found in the generated audio. Parsing the emitted string rather than keeping a parallel structure means the reward can only ever score what the model was actually shown. -> (items, total_s) where each item is {"kind": "pause"|"speech"|"burst", "t": start_s, "dur": s, "label": str|None} """ items, t = [], 0.0 for m in _ITEM.finditer(script or ""): if m.group(1) is not None: d = float(m.group(1)) items.append({"kind": m.group(2), "t": t, "dur": d, "label": None}) t += d elif m.group(3) is not None: d = float(m.group(4)) items.append({"kind": "burst", "t": t, "dur": d, "label": m.group(3).strip()}) t += d else: items.append({"kind": "burst", "t": t, "dur": None, "label": m.group(5).strip()}) return items, round(t, 2) def prompted_bursts(script): items, _ = parse_script(script) return [(i["label"], i["t"], i["dur"]) for i in items if i["kind"] == "burst" and i["dur"] is not None]