NEXORA / nexora /voice.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
2.16 kB
"""Interruptible voice orchestration, with optional local ASR/TTS adapters.
Microphone VAD/streaming ASR and production full-duplex echo cancellation are planned.
"""
import asyncio
import time
from .evaluation import word_error_rate
class VoiceSession:
def __init__(self, reply, speak, stop_audio):
self.reply, self.speak, self.stop_audio = reply, speak, stop_audio
self.generation = 0
self.task = None
self.metrics = []
async def interrupt(self):
self.generation += 1
await self.stop_audio()
if self.task and not self.task.done():
self.task.cancel()
try:
await self.task
except asyncio.CancelledError:
pass
async def transcript(self, text, *, speech_end=None):
await self.interrupt()
generation = self.generation
received = time.perf_counter()
async def turn():
first = None
async for chunk in self.reply(text):
if generation != self.generation:
return
if first is None:
first = time.perf_counter()
audio_start = await self.speak(chunk)
self.metrics.append({"transcript_ms": None if speech_end is None else (received-speech_end)*1000,
"first_token_ms": (first-received)*1000,
"first_audio_ms": None if audio_start is None else (audio_start-received)*1000})
self.task = asyncio.create_task(turn())
return self.task
def transcribe_file(audio_path, model="small", language=None):
from faster_whisper import WhisperModel
asr = WhisperModel(model, device="cpu", compute_type="int8")
start = time.perf_counter()
segments, info = asr.transcribe(str(audio_path), language=language, vad_filter=True)
text = " ".join(s.text.strip() for s in segments)
return {"text": text, "language": info.language, "seconds": time.perf_counter()-start}
def speak_local(text):
import pyttsx3
engine = pyttsx3.init()
engine.say(text)
engine.runAndWait()