from typing import Any, Literal from pydantic import BaseModel, Field ACTS = ["backchannel", "confirm", "request", "correct", "cancel", "other"] INTENTS = ["none", "reservation", "order", "change", "cancel"] MAX_AUDIO_SECONDS = 30 SAMPLE_RATE = 16000 MAX_AUDIO_SAMPLES = MAX_AUDIO_SECONDS * SAMPLE_RATE SOUND_SAMPLE_RATE = 32000 SOUND_LABELS = ("animal", "nature", "music", "environment") # These are the only labels for which the prototype has a trained head. The # OpenJev-shaped endpoint deliberately uses this allowlist instead of treating # arbitrary natural-language questions as if they had been trained. SUPPORTED_SYSTEMONE_FIELDS = ("speech_act", "intent", "turn_complete") class Interaction(BaseModel): turn_complete_probability: float = Field(ge=0, le=1) speech_act: dict[str, float] intent: dict[str, float] calibration: Literal["uncalibrated_synthetic_prototype"] = "uncalibrated_synthetic_prototype" class SoundEvent(BaseModel): """One observed sound-model window. ``scores`` are the model's raw multilabel scores. They deliberately have no probability bounds or calibration claim. ``labels`` contains the coarse labels represented by the event; when a trained head supplies thresholds, these are its predicted labels. ``original_labels`` keeps source-model diagnostics when supplied. """ start_ms: int = Field(ge=0) end_ms: int = Field(ge=0) labels: list[str] scores: dict[str, float] original_labels: dict[str, float] = Field(default_factory=dict) thresholds: dict[str, float] | None = None class VADSegment(BaseModel): """Observed voice-activity interval from the optional VAD companion.""" start_ms: int = Field(ge=0) end_ms: int = Field(ge=0) speech_probability: float = Field(ge=0, le=1) class SpeakerSegment(BaseModel): """Anonymous speaker timeline interval; it is not a speech attribution.""" start_ms: int = Field(ge=0) end_ms: int = Field(ge=0) speaker_id: str | None = None class SlotValue(BaseModel): value: str | int evidence: str source: Literal["transcript_rule"] = "transcript_rule" class Event(BaseModel): utterance_id: str revision: int = Field(ge=1) audio_until_ms: int status: Literal["provisional", "final"] transcript: str transcript_source: Literal["whisper", "provided"] interaction: Interaction slots: dict[str, SlotValue] missing_fields: list[str] next_step: Literal["listen", "ask_clarification", "review", "acknowledge"] action_executable: Literal[False] = False latency_ms: dict[str, float] sound_events: list[SoundEvent] | None = Field( default=None, exclude_if=lambda value: value is None ) vad_segments: list[VADSegment] | None = Field( default=None, exclude_if=lambda value: value is None ) speaker_segments: list[SpeakerSegment] | None = Field( default=None, exclude_if=lambda value: value is None ) speaker_count: int | None = Field(default=None, ge=0) segmentation: dict[str, Any] | None = Field( default=None, exclude_if=lambda value: value is None ) model: str = "whisper-jev-v0.1" class SystemOneQuestion(BaseModel): """Small, local subset of OpenJev's typed question shape. The HTTP endpoint performs the field/criteria checks as well because the accepted field determines which learned head can answer a question. """ type: Literal["choice", "noul", "score"] instructions: str | dict | list | None = None criteria: dict | list | None = None class SystemOneRequest(BaseModel): state: str | dict | list model: str = "whisper-jev-v0.1" questions: dict[str, SystemOneQuestion]