Text-to-Speech
English
German
voice-acting
qwen3
moss-audio-tokenizer-v2
audio-generation
Humaneness-Voice-Small / code /timed_prompt.py
ChristophSchuhmann's picture
Document architecture, prompts, code, and full run statistics
d911efa verified
Raw History Blame Contribute Delete
6.46 kB
#!/usr/bin/env python3
"""Deterministic prompt repair for the S6--S10 timing curriculum.
The packed TAR metadata already contains word timestamps for most rows, but the
``prompt`` field was rendered by procedural-voice-captions PR #4 and contains
only legacy ``[pause 0.4s]`` markers. This module replaces only the prompt
surface at read time. Audio, codec targets, manifests, and sample ordering stay
immutable.
The contract follows prompt_schema_2026-08-25.html: 70% of timing-eligible
presentations are timed; each speech segment gets a duration; gaps and detected
bursts are budgeted; the remaining presentations carry no numbers. Rows without
trustworthy alignment stay untimed rather than receiving invented precision.
"""
from __future__ import annotations
import hashlib
import random
import re
import anno_timed as anno
FORMAT_ID = "m2-cascade-timed-repair-v1-20260920"
P_TIMED = 0.70
P_DROP_BURSTS = 0.10
P_DIRECTION = 0.85
P_DROP_DIRECTION_PER_SEGMENT = 0.15
_LEADING_CUE = re.compile(r"^\s*(\([^()\n]+\))\s*")
_DURATION = re.compile(r"\[(\d+(?:\.\d+)?) seconds duration\]")
_ANY_NUMBERED_TAG = re.compile(
r"\[\d+(?:\.\d+)? seconds (?:duration|pause)\]|"
r"\([^()]*, \d+(?:\.\d+)? seconds\)"
)
def _rng(uid: str, presentation: int, phase: str, mode: str, form: str) -> random.Random:
material = f"{FORMAT_ID}|{phase}|{presentation}|{uid}|{mode}|{form}"
digest = hashlib.blake2b(material.encode(), digest_size=16).digest()
return random.Random(int.from_bytes(digest, "big"))
def _words(meta: dict) -> list[dict]:
"""Normalise packed ``start/end`` words to anno.py's ``s/e`` contract."""
result = []
for row in meta.get("words") or []:
word = row.get("w")
start = row.get("s", row.get("start"))
end = row.get("e", row.get("end"))
if not word or start is None or end is None:
continue
result.append({"w": str(word), "s": float(start), "e": float(end)})
return result
def _bursts(meta: dict) -> list[tuple[float, float, str]]:
result = []
for row in meta.get("bursts") or []:
if isinstance(row, dict):
start, end = row.get("start", row.get("s")), row.get("end", row.get("e"))
label = row.get("written") or row.get("label") or "vocal burst"
else:
try:
start, end, label = row[:3]
except Exception:
continue
if start is None or end is None or float(end) <= float(start):
continue
result.append((float(start), float(end), str(label)))
return result
def _legacy_parts(prompt: str) -> tuple[str, list[str]]:
prompt = prompt or ""
if prompt.startswith("GENERAL: ") and "\nSCRIPT:\n" in prompt:
general, script = prompt[len("GENERAL: "):].split("\nSCRIPT:\n", 1)
else:
general, script = "", prompt
cues = []
for line in script.splitlines():
match = _LEADING_CUE.match(line)
if match:
cues.append(match.group(1))
return general.strip(), cues
def _add_directions(script: str, cues: list[str], rng: random.Random, timed: bool) -> tuple[str, int]:
if not cues or rng.random() >= P_DIRECTION:
return script, 0
if not timed:
if rng.random() < P_DROP_DIRECTION_PER_SEGMENT:
return script, 0
return f"{cues[0]} {script}".strip(), 1
visited = 0
emitted = 0
def replace(match: re.Match) -> str:
nonlocal visited, emitted
cue = cues[min(visited, len(cues) - 1)]
visited += 1
if rng.random() < P_DROP_DIRECTION_PER_SEGMENT:
return match.group(0)
emitted += 1
return f"{cue} {match.group(0)}"
return _DURATION.sub(replace, script), emitted
def repair_prompt(meta: dict, *, caption: str | None, form: str, mode: str,
presentation: int, phase: str,
timed_override: bool | None = None) -> tuple[str, dict]:
"""Return the repaired prompt and auditable per-presentation metadata."""
assert form in ("A", "B")
uid = str(meta["uid"])
rng = _rng(uid, int(presentation), str(phase), str(mode), form)
words = _words(meta)
text = str(meta.get("text") or "").strip()
aligned = anno.align_words(text, words)
alignment_ratio = len(aligned) / max(1, len(words))
eligible = bool(text and words and aligned and alignment_ratio >= 0.50)
timed_draw = rng.random() < P_TIMED if timed_override is None else bool(timed_override)
timed_requested = eligible and timed_draw
drop_bursts = bool(words) and rng.random() < P_DROP_BURSTS
script, annotation = anno.render(
text, words, _bursts(meta), float(meta.get("dur_s") or 0.0),
timed=timed_requested, drop_bursts=drop_bursts,
)
timed = bool(timed_requested and annotation["n_seg"] > 0)
general, cues = _legacy_parts(meta.get("prompt") or "")
direction_count = 0
if form == "A":
script, direction_count = _add_directions(script, cues, rng, timed)
prompt = f"GENERAL: {general}\nSCRIPT:\n{script}" if general else f"SCRIPT:\n{script}"
else:
body = (caption or "").strip()
if not body:
raise ValueError(f"Form B without body caption: {uid}")
prompt = f'CAPTION: {body}\nTRANSCRIPT: "{script}"'
duration_tags = len(_DURATION.findall(prompt))
if timed:
assert duration_tags == annotation["n_seg"] > 0, (uid, duration_tags, annotation)
proof = anno.verify(script, float(meta.get("dur_s") or 0.0), text=text)
else:
assert duration_tags == 0 and not _ANY_NUMBERED_TAG.search(script), uid
proof = {"sum": 0.0, "dur": float(meta.get("dur_s") or 0.0),
"residual": None, "text_out": text}
assert "[pause " not in prompt, f"legacy pause survived repair: {uid}"
return prompt, {
"format_id": FORMAT_ID,
"eligible": eligible,
"timed": timed,
"untimed_reason": None if timed else ("draw" if eligible else "no_trustworthy_alignment"),
"alignment_ratio": alignment_ratio,
"duration_tags": duration_tags,
"segments": int(annotation["n_seg"]),
"pauses": int(annotation["n_pause"]),
"bursts": int(annotation["n_burst"]),
"directions": int(direction_count),
"drop_bursts": drop_bursts,
"numeric_sum_seconds": proof["sum"],
"numeric_residual_seconds": proof["residual"],
}