"""Interruptible voice orchestration, with optional local ASR/TTS adapters. Microphone VAD/streaming ASR and production full-duplex echo cancellation are planned. """ import asyncio import time from .evaluation import word_error_rate class VoiceSession: def __init__(self, reply, speak, stop_audio): self.reply, self.speak, self.stop_audio = reply, speak, stop_audio self.generation = 0 self.task = None self.metrics = [] async def interrupt(self): self.generation += 1 await self.stop_audio() if self.task and not self.task.done(): self.task.cancel() try: await self.task except asyncio.CancelledError: pass async def transcript(self, text, *, speech_end=None): await self.interrupt() generation = self.generation received = time.perf_counter() async def turn(): first = None async for chunk in self.reply(text): if generation != self.generation: return if first is None: first = time.perf_counter() audio_start = await self.speak(chunk) self.metrics.append({"transcript_ms": None if speech_end is None else (received-speech_end)*1000, "first_token_ms": (first-received)*1000, "first_audio_ms": None if audio_start is None else (audio_start-received)*1000}) self.task = asyncio.create_task(turn()) return self.task def transcribe_file(audio_path, model="small", language=None): from faster_whisper import WhisperModel asr = WhisperModel(model, device="cpu", compute_type="int8") start = time.perf_counter() segments, info = asr.transcribe(str(audio_path), language=language, vad_filter=True) text = " ".join(s.text.strip() for s in segments) return {"text": text, "language": info.language, "seconds": time.perf_counter()-start} def speak_local(text): import pyttsx3 engine = pyttsx3.init() engine.say(text) engine.runAndWait()