Spaces:
Running
Running
Download server_runtime/schema.py from mocomoco-inc/AudioDecisionModel: direct link, hf CLI and curl.
- Browser
- Download file 3.73 kB
-
https://huggingface.co/spaces/mocomoco-inc/AudioDecisionModel/resolve/main/server_runtime/schema.py
- Command line
-
hf download hf://spaces/mocomoco-inc/AudioDecisionModel/server_runtime/schema.py
-
curl -L -o schema.py https://huggingface.co/spaces/mocomoco-inc/AudioDecisionModel/resolve/main/server_runtime/schema.py
3.73 kB
| from typing import Any, Literal | |
| from pydantic import BaseModel, Field | |
| ACTS = ["backchannel", "confirm", "request", "correct", "cancel", "other"] | |
| INTENTS = ["none", "reservation", "order", "change", "cancel"] | |
| MAX_AUDIO_SECONDS = 30 | |
| SAMPLE_RATE = 16000 | |
| MAX_AUDIO_SAMPLES = MAX_AUDIO_SECONDS * SAMPLE_RATE | |
| SOUND_SAMPLE_RATE = 32000 | |
| SOUND_LABELS = ("animal", "nature", "music", "environment") | |
| # These are the only labels for which the prototype has a trained head. The | |
| # OpenJev-shaped endpoint deliberately uses this allowlist instead of treating | |
| # arbitrary natural-language questions as if they had been trained. | |
| SUPPORTED_SYSTEMONE_FIELDS = ("speech_act", "intent", "turn_complete") | |
| class Interaction(BaseModel): | |
| turn_complete_probability: float = Field(ge=0, le=1) | |
| speech_act: dict[str, float] | |
| intent: dict[str, float] | |
| calibration: Literal["uncalibrated_synthetic_prototype"] = "uncalibrated_synthetic_prototype" | |
| class SoundEvent(BaseModel): | |
| """One observed sound-model window. | |
| ``scores`` are the model's raw multilabel scores. They deliberately have | |
| no probability bounds or calibration claim. ``labels`` contains the | |
| coarse labels represented by the event; when a trained head supplies | |
| thresholds, these are its predicted labels. ``original_labels`` keeps | |
| source-model diagnostics when supplied. | |
| """ | |
| start_ms: int = Field(ge=0) | |
| end_ms: int = Field(ge=0) | |
| labels: list[str] | |
| scores: dict[str, float] | |
| original_labels: dict[str, float] = Field(default_factory=dict) | |
| thresholds: dict[str, float] | None = None | |
| class VADSegment(BaseModel): | |
| """Observed voice-activity interval from the optional VAD companion.""" | |
| start_ms: int = Field(ge=0) | |
| end_ms: int = Field(ge=0) | |
| speech_probability: float = Field(ge=0, le=1) | |
| class SpeakerSegment(BaseModel): | |
| """Anonymous speaker timeline interval; it is not a speech attribution.""" | |
| start_ms: int = Field(ge=0) | |
| end_ms: int = Field(ge=0) | |
| speaker_id: str | None = None | |
| class SlotValue(BaseModel): | |
| value: str | int | |
| evidence: str | |
| source: Literal["transcript_rule"] = "transcript_rule" | |
| class Event(BaseModel): | |
| utterance_id: str | |
| revision: int = Field(ge=1) | |
| audio_until_ms: int | |
| status: Literal["provisional", "final"] | |
| transcript: str | |
| transcript_source: Literal["whisper", "provided"] | |
| interaction: Interaction | |
| slots: dict[str, SlotValue] | |
| missing_fields: list[str] | |
| next_step: Literal["listen", "ask_clarification", "review", "acknowledge"] | |
| action_executable: Literal[False] = False | |
| latency_ms: dict[str, float] | |
| sound_events: list[SoundEvent] | None = Field( | |
| default=None, exclude_if=lambda value: value is None | |
| ) | |
| vad_segments: list[VADSegment] | None = Field( | |
| default=None, exclude_if=lambda value: value is None | |
| ) | |
| speaker_segments: list[SpeakerSegment] | None = Field( | |
| default=None, exclude_if=lambda value: value is None | |
| ) | |
| speaker_count: int | None = Field(default=None, ge=0) | |
| segmentation: dict[str, Any] | None = Field( | |
| default=None, exclude_if=lambda value: value is None | |
| ) | |
| model: str = "whisper-jev-v0.1" | |
| class SystemOneQuestion(BaseModel): | |
| """Small, local subset of OpenJev's typed question shape. | |
| The HTTP endpoint performs the field/criteria checks as well because the | |
| accepted field determines which learned head can answer a question. | |
| """ | |
| type: Literal["choice", "noul", "score"] | |
| instructions: str | dict | list | None = None | |
| criteria: dict | list | None = None | |
| class SystemOneRequest(BaseModel): | |
| state: str | dict | list | |
| model: str = "whisper-jev-v0.1" | |
| questions: dict[str, SystemOneQuestion] | |