Download code/anno_timed.py from laion/Humaneness-Voice-Small: direct link, hf CLI and curl.
- Browser
- Download file 14.1 kB
-
https://huggingface.co/laion/Humaneness-Voice-Small/resolve/main/code/anno_timed.py
- Command line
-
hf download hf://laion/Humaneness-Voice-Small/code/anno_timed.py
-
curl -L -o anno_timed.py https://huggingface.co/laion/Humaneness-Voice-Small/resolve/main/code/anno_timed.py
14.1 kB
| #!/usr/bin/env python3 | |
| """Timed-script annotation engine (v2). | |
| WHY. Listening to the SFT+DPO model surfaced two failure modes: vocal bursts that run far too | |
| long and sound unnatural, and occasional hallucinated content that was never prompted. Both are | |
| BUDGET failures -- nothing in the v1 prompt told the model how long anything was allowed to take. | |
| v1 carried at most one optional `[7.3 seconds duration]` prefix for the whole clip and bare | |
| `(sigh)` cues with no duration at all. | |
| v2 renders the target text as a TIMED SCRIPT built from the word-level timestamps and the | |
| DETECTED burst spans, so every second of the clip is accounted for: | |
| [0.9 seconds pause] [3.4 seconds duration] Die Handelskammer betreibt diesen Test. | |
| [0.6 seconds pause] (contented sigh, 0.4 seconds) [1.2 seconds pause] | |
| [4.1 seconds duration] Die entwickeln und fuehren ihn durch, wissen Sie. | |
| Rules, each asserted by `verify()` on the EMITTED string: | |
| 1. `[D seconds duration]` precedes a speech segment and equals that SEGMENT's last-word offset | |
| minus its first-word onset -- never the clip's. | |
| 2. Segments break at sentence-final punctuation. A segment still longer than MAX_SEG_S is | |
| split again at its largest internal word gap, recursively. A 20 s two-sentence utterance | |
| therefore renders as `[12.0 ...] <s1> [8.0 ...] <s2>`. | |
| 3. Every gap >= PAUSE_MIN between consecutive events -- including before the first and after | |
| the last -- is emitted as `[G seconds pause]`. | |
| 4. Every detected burst is emitted at its position, as `(label, D seconds)`. | |
| 5. The emitted numbers sum to the clip duration. | |
| FOUR THINGS A NAIVE IMPLEMENTATION GETS WRONG, all measured on this corpus: | |
| * The transcript must not be rebuilt by joining word tokens. The forced aligner drops | |
| numerals ("154 Euro" aligns as "Euro"), so a join silently loses text on 9.9 % of | |
| real-speech rows. Words are ALIGNED to the original string by character offset and the | |
| original string is SLICED, so output text is byte-identical apart from inserted tags. | |
| * The script is built from DETECTED bursts, not from the parenthetical cues already sitting in | |
| `text_with_bursts`. On the voice-profile corpus those parentheticals are SYNTHESIS | |
| DIRECTIONS -- 39.4 % were never confirmed by the burst detector. A target text carrying a | |
| cue that is not in the audio teaches precisely the behaviour being removed. | |
| * 49.0 % of detected bursts OVERLAP an aligned word span. Adding such a burst's full length | |
| to the budget double-counts time -- it drove the sum 2.1 s past the clip length on the worst | |
| row of the first test. A burst's printed duration is its length MINUS its overlap with | |
| speech; when nothing is left it is printed WITHOUT a number, which is the honest statement | |
| that it is co-articulated and has no slot of its own. | |
| * A gap SHORTER than PAUSE_MIN is not worth a tag, but skipping it breaks the sum and the | |
| residual then reads as an error in the duration values. Such a gap is absorbed into the | |
| adjacent speech segment (never into a burst -- a burst tag must state the burst's own length). | |
| """ | |
| import re | |
| PAUSE_MIN = 0.20 # gaps below this are absorbed rather than tagged | |
| MAX_SEG_S = 12.0 # a speech segment longer than this is split again at its largest gap | |
| MIN_SPLIT_S = 2.0 # never create a segment shorter than this by splitting | |
| BURST_MIN_S = 0.05 # below this a burst prints without a duration | |
| _SENT_END = re.compile(r"[.!?…]+[\"'“”’\)\]]*$") | |
| def _fmt(x): | |
| return f"{x:.1f}" | |
| def align_words(text, words): | |
| """-> [(char_start, char_end, t_start, t_end)] for the word tokens locatable in `text`.""" | |
| out, pos = [], 0 | |
| for w in words or []: | |
| tok = (w.get("w") or "").strip() | |
| if not tok: | |
| continue | |
| try: | |
| s, e = float(w["s"]), float(w["e"]) | |
| except Exception: | |
| continue | |
| if not (e >= s >= -0.001): | |
| continue | |
| i = text.find(tok, pos) | |
| if i < 0: | |
| continue | |
| j = i + len(tok) | |
| out.append((i, j, s, e)) | |
| pos = j | |
| return out | |
| def _sentence_breaks(text, aw): | |
| br = {k for k, (i, j, s, e) in enumerate(aw) if _SENT_END.search(text[i:j])} | |
| br.add(len(aw) - 1) | |
| return br | |
| def _split_long(aw, lo, hi): | |
| dur = aw[hi][3] - aw[lo][2] | |
| if dur <= MAX_SEG_S or hi - lo < 1: | |
| return [(lo, hi)] | |
| best, bestscore = None, -1e9 | |
| for k in range(lo, hi): | |
| if (aw[k][3] - aw[lo][2]) < MIN_SPLIT_S or (aw[hi][3] - aw[k + 1][2]) < MIN_SPLIT_S: | |
| continue | |
| gap = aw[k + 1][2] - aw[k][3] | |
| score = gap - 0.02 * abs((aw[k][3] - aw[lo][2]) - dur / 2.0) | |
| if score > bestscore: | |
| bestscore, best = score, k | |
| if best is None: | |
| return [(lo, hi)] | |
| return _split_long(aw, lo, best) + _split_long(aw, best + 1, hi) | |
| def _free_len(s, e, spans): | |
| """length of [s,e) not covered by any interval in `spans` (sorted, disjoint).""" | |
| free = e - s | |
| for (a, b) in spans: | |
| if b <= s: | |
| continue | |
| if a >= e: | |
| break | |
| free -= min(e, b) - max(s, a) | |
| return max(free, 0.0) | |
| def build_events(text, words, bursts, dur_s): | |
| aw = align_words(text, words) | |
| bs = [] | |
| for t in bursts or []: | |
| try: | |
| s, e, lab = float(t[0]), float(t[1]), str(t[2] or "").strip() | |
| except Exception: | |
| continue | |
| if e <= s: | |
| continue | |
| bs.append((s, e, lab)) | |
| bs.sort() | |
| if not aw: | |
| ev = [{"kind": "burst", "s": s, "e": e, "label": l, "dur": e - s, "timed": True} | |
| for (s, e, l) in bs] | |
| _absorb(ev, dur_s) | |
| return ev, aw | |
| # 1. sentence segmentation | |
| br = _sentence_breaks(text, aw) | |
| segs, lo = [], 0 | |
| for k in range(len(aw)): | |
| if k in br: | |
| segs.append((lo, k)) | |
| lo = k + 1 | |
| # 2. every burst splits the segment it lands in, at the WORD GAP nearest its midpoint. | |
| # (Requiring the burst to sit entirely inside a gap would fail for the 49 % that overlap | |
| # a word, and those would then be emitted out of order.) | |
| cuts = set() | |
| for (s, e, l) in bs: | |
| mid = 0.5 * (s + e) | |
| for (lo, hi) in segs: | |
| if aw[lo][2] <= mid <= aw[hi][3] and hi > lo: | |
| k = min(range(lo, hi), key=lambda k: abs(0.5 * (aw[k][3] + aw[k + 1][2]) - mid)) | |
| cuts.add(k) | |
| break | |
| out, = [[]] | |
| for (lo, hi) in segs: | |
| a = lo | |
| for k in range(lo, hi): | |
| if k in cuts: | |
| out.append((a, k)) | |
| a = k + 1 | |
| out.append((a, hi)) | |
| segs = [x for x in out if x[0] <= x[1]] | |
| # 3. length cap | |
| out = [] | |
| for (lo, hi) in segs: | |
| out.extend(_split_long(aw, lo, hi)) | |
| segs = out | |
| spans = sorted((aw[lo][2], aw[hi][3]) for (lo, hi) in segs) | |
| ev = [{"kind": "speech", "s": aw[lo][2], "e": aw[hi][3], "lo": lo, "hi": hi} for (lo, hi) in segs] | |
| for (s, e, l) in bs: | |
| f = _free_len(s, e, spans) | |
| ev.append({"kind": "burst", "s": s, "e": e, "label": l, | |
| "dur": f, "timed": f >= BURST_MIN_S}) | |
| ev.sort(key=lambda d: (d["s"], 0 if d["kind"] == "speech" else 1)) | |
| _absorb(ev, dur_s) | |
| return ev, aw | |
| def _absorb(ev, dur_s): | |
| prev = 0.0 | |
| for k, e in enumerate(ev): | |
| if e["kind"] == "burst" and not e.get("timed"): | |
| continue # occupies no budget of its own | |
| gap = e["s"] - prev | |
| if 0 < gap < PAUSE_MIN: | |
| if e["kind"] == "speech": | |
| e["s"] = prev | |
| elif k > 0 and ev[k - 1]["kind"] == "speech": | |
| ev[k - 1]["e"] = e["s"] | |
| prev = max(prev, e["e"] if e["kind"] == "speech" else e["s"] + e["dur"]) | |
| if dur_s and ev: | |
| last = None | |
| for e in reversed(ev): | |
| if e["kind"] == "speech": | |
| last = e | |
| break | |
| if last is not None: | |
| tail = float(dur_s) - last["e"] | |
| if 0 < tail < PAUSE_MIN and last is ev[-1]: | |
| last["e"] = float(dur_s) | |
| def render(text, words, bursts, dur_s, timed=True, drop_bursts=False, trailing=True, | |
| blank_text=False): | |
| """-> (script, meta). `timed=False` keeps burst labels but omits every number. | |
| `blank_text=True` emits `...` in place of every SPEECH CHUNK and leaves every tag -- pauses, | |
| per-segment durations, burst labels and their lengths -- exactly where it was. It exists for | |
| the classifier-free-guidance DPO pairs, where the two clips of a pair say different words and | |
| the preference must therefore not be decidable from the words. Nothing else about the | |
| rendering changes: the segmentation, the numbers and their sum are computed from the real text | |
| and the real word timings, so the emitted budget still describes the real clip. | |
| """ | |
| text = text or "" | |
| ev, aw = build_events(text, words, bursts, dur_s) | |
| # `not aw` used to be part of this guard, which made the wordless branch of build_events | |
| # (line ~131, "no aligned words -> the events ARE the bursts") unreachable: a clip that is | |
| # nothing but a vocal burst rendered as its empty text instead of as `(chuckle, 0.5 seconds)`. | |
| # `not ev` is equivalent for every other input -- with no words and no bursts build_events | |
| # returns an empty list anyway -- so this only changes rows that are wordless AND carry a | |
| # burst: 709 of the 3,144,739 corpus rows, plus the standalone-burst training set. | |
| if not ev: | |
| head = ("..." if text.strip() else "") if blank_text else text.strip() | |
| return head, {"n_seg": 0, "n_burst": 0, "n_pause": 0, "n_burst_untimed": 0, | |
| "timed": False, "acc": 0.0} | |
| parts, t, cut = [], 0.0, 0 | |
| n_seg = n_burst = n_pause = n_untimed = 0 | |
| acc = 0.0 | |
| for k, e in enumerate(ev): | |
| if e["kind"] == "burst" and drop_bursts: | |
| continue # its span folds into the surrounding pause | |
| if e["kind"] == "burst" and not e["timed"]: | |
| parts.append(f"({e['label'].lower() or 'vocal burst'})") | |
| n_burst += 1 | |
| n_untimed += 1 | |
| continue | |
| gap = e["s"] - t | |
| if timed and gap >= PAUSE_MIN: | |
| parts.append(f"[{_fmt(gap)} seconds pause]") | |
| acc += round(gap, 1) | |
| n_pause += 1 | |
| if e["kind"] == "speech": | |
| is_last_speech = not any(x["kind"] == "speech" for x in ev[k + 1:]) | |
| end_char = len(text) if is_last_speech else aw[e["hi"]][1] | |
| chunk = text[cut:end_char].strip() | |
| cut = end_char | |
| if blank_text: | |
| chunk = "..." if chunk else "" | |
| d = e["e"] - e["s"] | |
| if timed: | |
| parts.append(f"[{_fmt(d)} seconds duration]") | |
| acc += round(d, 1) | |
| if chunk: | |
| parts.append(chunk) | |
| n_seg += 1 | |
| t = e["e"] | |
| else: | |
| lab = e["label"].lower() or "vocal burst" | |
| parts.append(f"({lab}, {_fmt(e['dur'])} seconds)" if timed else f"({lab})") | |
| if timed: | |
| acc += round(e["dur"], 1) | |
| n_burst += 1 | |
| t = max(t, e["s"] + e["dur"]) | |
| if timed and trailing and dur_s and (float(dur_s) - t) >= PAUSE_MIN: | |
| parts.append(f"[{_fmt(float(dur_s) - t)} seconds pause]") | |
| acc += round(float(dur_s) - t, 1) | |
| n_pause += 1 | |
| script = re.sub(r"\s{2,}", " ", " ".join(p for p in parts if p)).strip() | |
| return script, {"n_seg": n_seg, "n_burst": n_burst, "n_pause": n_pause, | |
| "n_burst_untimed": n_untimed, "timed": bool(timed), "acc": round(acc, 3)} | |
| # ---------------------------------------------------------------- independent verification | |
| _TAG = re.compile(r"\[(\d+\.\d) seconds (?:pause|duration)\]") | |
| _BUR = re.compile(r"\([^,()]*, (\d+\.\d) seconds\)") | |
| def verify(script, dur_s, text=None): | |
| """Re-parses the EMITTED string. Deliberately does not reuse the renderer's accumulator, so | |
| a bookkeeping error in the renderer cannot hide behind its own arithmetic.""" | |
| tot = sum(float(m.group(1)) for m in _TAG.finditer(script)) | |
| tot += sum(float(m.group(1)) for m in _BUR.finditer(script)) | |
| stripped = re.sub(r"\([^)]*\)", " ", re.sub(r"\[[^\]]*\]", " ", script)) | |
| stripped = re.sub(r"\s+", " ", stripped).strip() | |
| return {"sum": round(tot, 2), "dur": round(float(dur_s or 0), 2), | |
| "residual": round(float(dur_s or 0) - tot, 2), "text_out": stripped} | |
| # ---------------------------------------------------------------- script -> plan | |
| _ITEM = re.compile(r"\[(\d+\.\d) seconds (pause|duration)\]|\(([^()]*?), (\d+\.\d) seconds\)|\(([^()]*?)\)") | |
| def parse_script(script): | |
| """Read an emitted timed script back into a plan. | |
| Used by the GRPO reward: the prompt states WHERE each burst should fall and HOW LONG it and | |
| every utterance should be, and the reward has to compare that against what the detector found | |
| in the generated audio. Parsing the emitted string rather than keeping a parallel structure | |
| means the reward can only ever score what the model was actually shown. | |
| -> (items, total_s) where each item is | |
| {"kind": "pause"|"speech"|"burst", "t": start_s, "dur": s, "label": str|None} | |
| """ | |
| items, t = [], 0.0 | |
| for m in _ITEM.finditer(script or ""): | |
| if m.group(1) is not None: | |
| d = float(m.group(1)) | |
| items.append({"kind": m.group(2), "t": t, "dur": d, "label": None}) | |
| t += d | |
| elif m.group(3) is not None: | |
| d = float(m.group(4)) | |
| items.append({"kind": "burst", "t": t, "dur": d, "label": m.group(3).strip()}) | |
| t += d | |
| else: | |
| items.append({"kind": "burst", "t": t, "dur": None, "label": m.group(5).strip()}) | |
| return items, round(t, 2) | |
| def prompted_bursts(script): | |
| items, _ = parse_script(script) | |
| return [(i["label"], i["t"], i["dur"]) for i in items | |
| if i["kind"] == "burst" and i["dur"] is not None] | |