DocDoeAI / scripts /validate_teaching_audio.py
asnannp's picture
Deploy backend cd4237ff: support routes + rate limit + exam_date nullable + upload 413 fix
7c6ffa6
Raw
History Blame Contribute Delete
6.91 kB
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
REQUIRED_TERMS_BY_CHAPTER: dict[str, set[str]] = {
"sound-waves": {"compression", "rarefaction", "longitudinal", "vacuum", "particles", "vibration"},
"lenses": {"refraction", "convex", "concave", "focal", "image", "virtual"},
"world-of-colours-and-vision": {"retina", "dispersion", "spectrum", "cornea", "wavelength"},
"magnetic-effect-electric-current": {"magnetic", "field", "solenoid", "electromagnet", "current"},
"electric-energy": {"power", "energy", "resistance", "voltage", "watt"},
"electromagnetic-induction": {"induction", "flux", "coil", "generator", "emf"},
"mechanical-advantage": {"lever", "fulcrum", "effort", "load", "mechanical"},
"humanism": {
"renaissance", "humanism", "constantinople", "machiavelli", "brunelleschi",
"gutenberg", "copernicus", "galileo", "vesalius", "reformation", "luther",
},
}
def required_terms_for(props_path: Path) -> set[str]:
"""Infer the chapter from the props path so every chapter gets its own
critical-vocabulary check instead of silently reusing Sound Waves' terms."""
for part in props_path.resolve().parts:
if part in REQUIRED_TERMS_BY_CHAPTER:
return REQUIRED_TERMS_BY_CHAPTER[part]
print(
f"WARNING: could not infer chapter slug from {props_path}; "
"no chapter-specific critical terms were checked.",
file=sys.stderr,
)
return set()
def normalise(text: str) -> list[str]:
return re.findall(r"[a-z0-9]+", text.lower())
def edit_distance(left: list[str], right: list[str]) -> int:
previous = list(range(len(right) + 1))
for left_index, left_word in enumerate(left, start=1):
current = [left_index]
for right_index, right_word in enumerate(right, start=1):
current.append(
min(
current[-1] + 1,
previous[right_index] + 1,
previous[right_index - 1] + (left_word != right_word),
)
)
previous = current
return previous[-1]
def resolve_audio_path(project_root: Path, audio_src: str) -> Path:
candidate = Path(audio_src)
if candidate.is_absolute() and candidate.exists():
return candidate
direct = project_root / str(audio_src).lstrip("/")
if direct.exists():
return direct
return project_root / "public" / str(audio_src).lstrip("/")
def main() -> None:
parser = argparse.ArgumentParser(description="Reject unclear teaching narration before rendering.")
parser.add_argument("props", type=Path)
parser.add_argument("--project-root", type=Path, required=True)
parser.add_argument("--model", default="medium")
parser.add_argument("--max-wer", type=float, default=0.10)
parser.add_argument("--reuse-transcript", action="store_true", help="Reuse the prior full-chapter transcript after changing only the expected-text normalization.")
args = parser.parse_args()
import whisper
props = json.loads(args.props.read_text(encoding="utf-8"))
model = whisper.load_model(args.model, device="cuda")
scene_reports = []
all_expected: list[str] = []
all_actual: list[str] = []
if "scenes" in props:
narration_units = [
{
"scene_id": scene["scene_id"],
"audio_src": scene["audio_src"],
"voice_text": scene["voice_text"],
}
for scene in props["scenes"]
]
else:
# Full production chapters use one loudness-normalised narration track
# plus timed captions, rather than one audio file per scene.
voice_text = " ".join(caption["text"] for caption in props["captions"])
narration_manifest_path = args.props.with_name("narration-manifest.json")
if narration_manifest_path.exists():
narration_manifest = json.loads(narration_manifest_path.read_text(encoding="utf-8"))
source_manifest_value = narration_manifest.get("sourceManifest")
if source_manifest_value:
source_manifest = json.loads((args.project_root / source_manifest_value).read_text(encoding="utf-8"))
selected_ids = {chunk["id"] for chunk in props["audioTimeline"]["chunks"]}
voice_text = " ".join(
chunk.get("spokenText", chunk["text"])
for chunk in source_manifest["chunks"]
if chunk["id"] in selected_ids
)
narration_units = [
{
"scene_id": "full-chapter",
"audio_src": props["narrationAudioSrc"],
"voice_text": voice_text,
}
]
prior_report_path = args.props.with_name("voice-qa-report.json")
prior_report = json.loads(prior_report_path.read_text(encoding="utf-8")) if args.reuse_transcript and prior_report_path.exists() else None
for scene in narration_units:
audio_src = str(scene["audio_src"])
audio_path = resolve_audio_path(args.project_root, audio_src)
expected = normalise(scene["voice_text"])
if prior_report and len(narration_units) == 1 and prior_report.get("scenes"):
transcript = str(prior_report["scenes"][0]["transcript"]).strip()
else:
result = model.transcribe(str(audio_path), language="en", fp16=True, temperature=0)
transcript = str(result["text"]).strip()
actual = normalise(transcript)
wer = edit_distance(expected, actual) / max(1, len(expected))
all_expected.extend(expected)
all_actual.extend(actual)
scene_reports.append(
{
"scene_id": scene["scene_id"],
"wer": round(wer, 4),
"expected": scene["voice_text"],
"transcript": transcript,
}
)
overall_wer = edit_distance(all_expected, all_actual) / max(1, len(all_expected))
expected_terms = required_terms_for(args.props).intersection(all_expected)
missing_terms = sorted(expected_terms.difference(all_actual))
passed = overall_wer <= args.max_wer and not missing_terms
report = {
"passed": passed,
"overall_wer": round(overall_wer, 4),
"max_wer": args.max_wer,
"required_terms_checked": sorted(expected_terms),
"missing_terms": missing_terms,
"scenes": scene_reports,
}
report_path = args.props.with_name("voice-qa-report.json")
report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False), encoding="utf-8")
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8")
print(json.dumps(report, indent=2, ensure_ascii=False))
raise SystemExit(0 if passed else 1)
if __name__ == "__main__":
main()