Download nexora/voice.py from devildasdf/NEXORA: direct link, hf CLI and curl.
- Browser
- Download file 2.16 kB
-
https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/voice.py
- Command line
-
hf download hf://devildasdf/NEXORA/nexora/voice.py
-
curl -L -o voice.py https://huggingface.co/devildasdf/NEXORA/resolve/main/nexora/voice.py
2.16 kB
| """Interruptible voice orchestration, with optional local ASR/TTS adapters. | |
| Microphone VAD/streaming ASR and production full-duplex echo cancellation are planned. | |
| """ | |
| import asyncio | |
| import time | |
| from .evaluation import word_error_rate | |
| class VoiceSession: | |
| def __init__(self, reply, speak, stop_audio): | |
| self.reply, self.speak, self.stop_audio = reply, speak, stop_audio | |
| self.generation = 0 | |
| self.task = None | |
| self.metrics = [] | |
| async def interrupt(self): | |
| self.generation += 1 | |
| await self.stop_audio() | |
| if self.task and not self.task.done(): | |
| self.task.cancel() | |
| try: | |
| await self.task | |
| except asyncio.CancelledError: | |
| pass | |
| async def transcript(self, text, *, speech_end=None): | |
| await self.interrupt() | |
| generation = self.generation | |
| received = time.perf_counter() | |
| async def turn(): | |
| first = None | |
| async for chunk in self.reply(text): | |
| if generation != self.generation: | |
| return | |
| if first is None: | |
| first = time.perf_counter() | |
| audio_start = await self.speak(chunk) | |
| self.metrics.append({"transcript_ms": None if speech_end is None else (received-speech_end)*1000, | |
| "first_token_ms": (first-received)*1000, | |
| "first_audio_ms": None if audio_start is None else (audio_start-received)*1000}) | |
| self.task = asyncio.create_task(turn()) | |
| return self.task | |
| def transcribe_file(audio_path, model="small", language=None): | |
| from faster_whisper import WhisperModel | |
| asr = WhisperModel(model, device="cpu", compute_type="int8") | |
| start = time.perf_counter() | |
| segments, info = asr.transcribe(str(audio_path), language=language, vad_filter=True) | |
| text = " ".join(s.text.strip() for s in segments) | |
| return {"text": text, "language": info.language, "seconds": time.perf_counter()-start} | |
| def speak_local(text): | |
| import pyttsx3 | |
| engine = pyttsx3.init() | |
| engine.say(text) | |
| engine.runAndWait() | |