gojiteji's picture
Audio Decision Model demo
148af80
Raw History Blame Contribute Delete
3.73 kB
from typing import Any, Literal
from pydantic import BaseModel, Field
ACTS = ["backchannel", "confirm", "request", "correct", "cancel", "other"]
INTENTS = ["none", "reservation", "order", "change", "cancel"]
MAX_AUDIO_SECONDS = 30
SAMPLE_RATE = 16000
MAX_AUDIO_SAMPLES = MAX_AUDIO_SECONDS * SAMPLE_RATE
SOUND_SAMPLE_RATE = 32000
SOUND_LABELS = ("animal", "nature", "music", "environment")
# These are the only labels for which the prototype has a trained head. The
# OpenJev-shaped endpoint deliberately uses this allowlist instead of treating
# arbitrary natural-language questions as if they had been trained.
SUPPORTED_SYSTEMONE_FIELDS = ("speech_act", "intent", "turn_complete")
class Interaction(BaseModel):
turn_complete_probability: float = Field(ge=0, le=1)
speech_act: dict[str, float]
intent: dict[str, float]
calibration: Literal["uncalibrated_synthetic_prototype"] = "uncalibrated_synthetic_prototype"
class SoundEvent(BaseModel):
"""One observed sound-model window.
``scores`` are the model's raw multilabel scores. They deliberately have
no probability bounds or calibration claim. ``labels`` contains the
coarse labels represented by the event; when a trained head supplies
thresholds, these are its predicted labels. ``original_labels`` keeps
source-model diagnostics when supplied.
"""
start_ms: int = Field(ge=0)
end_ms: int = Field(ge=0)
labels: list[str]
scores: dict[str, float]
original_labels: dict[str, float] = Field(default_factory=dict)
thresholds: dict[str, float] | None = None
class VADSegment(BaseModel):
"""Observed voice-activity interval from the optional VAD companion."""
start_ms: int = Field(ge=0)
end_ms: int = Field(ge=0)
speech_probability: float = Field(ge=0, le=1)
class SpeakerSegment(BaseModel):
"""Anonymous speaker timeline interval; it is not a speech attribution."""
start_ms: int = Field(ge=0)
end_ms: int = Field(ge=0)
speaker_id: str | None = None
class SlotValue(BaseModel):
value: str | int
evidence: str
source: Literal["transcript_rule"] = "transcript_rule"
class Event(BaseModel):
utterance_id: str
revision: int = Field(ge=1)
audio_until_ms: int
status: Literal["provisional", "final"]
transcript: str
transcript_source: Literal["whisper", "provided"]
interaction: Interaction
slots: dict[str, SlotValue]
missing_fields: list[str]
next_step: Literal["listen", "ask_clarification", "review", "acknowledge"]
action_executable: Literal[False] = False
latency_ms: dict[str, float]
sound_events: list[SoundEvent] | None = Field(
default=None, exclude_if=lambda value: value is None
)
vad_segments: list[VADSegment] | None = Field(
default=None, exclude_if=lambda value: value is None
)
speaker_segments: list[SpeakerSegment] | None = Field(
default=None, exclude_if=lambda value: value is None
)
speaker_count: int | None = Field(default=None, ge=0)
segmentation: dict[str, Any] | None = Field(
default=None, exclude_if=lambda value: value is None
)
model: str = "whisper-jev-v0.1"
class SystemOneQuestion(BaseModel):
"""Small, local subset of OpenJev's typed question shape.
The HTTP endpoint performs the field/criteria checks as well because the
accepted field determines which learned head can answer a question.
"""
type: Literal["choice", "noul", "score"]
instructions: str | dict | list | None = None
criteria: dict | list | None = None
class SystemOneRequest(BaseModel):
state: str | dict | list
model: str = "whisper-jev-v0.1"
questions: dict[str, SystemOneQuestion]