File size: 3,727 Bytes
148af80
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
from typing import Any, Literal
from pydantic import BaseModel, Field

ACTS = ["backchannel", "confirm", "request", "correct", "cancel", "other"]
INTENTS = ["none", "reservation", "order", "change", "cancel"]
MAX_AUDIO_SECONDS = 30
SAMPLE_RATE = 16000
MAX_AUDIO_SAMPLES = MAX_AUDIO_SECONDS * SAMPLE_RATE
SOUND_SAMPLE_RATE = 32000
SOUND_LABELS = ("animal", "nature", "music", "environment")

# These are the only labels for which the prototype has a trained head.  The
# OpenJev-shaped endpoint deliberately uses this allowlist instead of treating
# arbitrary natural-language questions as if they had been trained.
SUPPORTED_SYSTEMONE_FIELDS = ("speech_act", "intent", "turn_complete")


class Interaction(BaseModel):
    turn_complete_probability: float = Field(ge=0, le=1)
    speech_act: dict[str, float]
    intent: dict[str, float]
    calibration: Literal["uncalibrated_synthetic_prototype"] = "uncalibrated_synthetic_prototype"


class SoundEvent(BaseModel):
    """One observed sound-model window.

    ``scores`` are the model's raw multilabel scores.  They deliberately have
    no probability bounds or calibration claim. ``labels`` contains the
    coarse labels represented by the event; when a trained head supplies
    thresholds, these are its predicted labels. ``original_labels`` keeps
    source-model diagnostics when supplied.
    """

    start_ms: int = Field(ge=0)
    end_ms: int = Field(ge=0)
    labels: list[str]
    scores: dict[str, float]
    original_labels: dict[str, float] = Field(default_factory=dict)
    thresholds: dict[str, float] | None = None


class VADSegment(BaseModel):
    """Observed voice-activity interval from the optional VAD companion."""

    start_ms: int = Field(ge=0)
    end_ms: int = Field(ge=0)
    speech_probability: float = Field(ge=0, le=1)


class SpeakerSegment(BaseModel):
    """Anonymous speaker timeline interval; it is not a speech attribution."""

    start_ms: int = Field(ge=0)
    end_ms: int = Field(ge=0)
    speaker_id: str | None = None


class SlotValue(BaseModel):
    value: str | int
    evidence: str
    source: Literal["transcript_rule"] = "transcript_rule"


class Event(BaseModel):
    utterance_id: str
    revision: int = Field(ge=1)
    audio_until_ms: int
    status: Literal["provisional", "final"]
    transcript: str
    transcript_source: Literal["whisper", "provided"]
    interaction: Interaction
    slots: dict[str, SlotValue]
    missing_fields: list[str]
    next_step: Literal["listen", "ask_clarification", "review", "acknowledge"]
    action_executable: Literal[False] = False
    latency_ms: dict[str, float]
    sound_events: list[SoundEvent] | None = Field(
        default=None, exclude_if=lambda value: value is None
    )
    vad_segments: list[VADSegment] | None = Field(
        default=None, exclude_if=lambda value: value is None
    )
    speaker_segments: list[SpeakerSegment] | None = Field(
        default=None, exclude_if=lambda value: value is None
    )
    speaker_count: int | None = Field(default=None, ge=0)
    segmentation: dict[str, Any] | None = Field(
        default=None, exclude_if=lambda value: value is None
    )
    model: str = "whisper-jev-v0.1"


class SystemOneQuestion(BaseModel):
    """Small, local subset of OpenJev's typed question shape.

    The HTTP endpoint performs the field/criteria checks as well because the
    accepted field determines which learned head can answer a question.
    """

    type: Literal["choice", "noul", "score"]
    instructions: str | dict | list | None = None
    criteria: dict | list | None = None


class SystemOneRequest(BaseModel):
    state: str | dict | list
    model: str = "whisper-jev-v0.1"
    questions: dict[str, SystemOneQuestion]