File size: 4,134 Bytes
f6588d9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
"""Build the one frozen answer-free realtime inference-row surface."""

from __future__ import annotations

from collections.abc import Sequence
from decimal import Decimal
from itertools import pairwise
from pathlib import Path
from typing import Any

from .extractor import SELECTED_FRAME_COUNT, endpoint_inclusive_indices
from .prompt import build_absolute_timeline_prefix
from .realtime import TimedVideo, classify_flat_blank

MIN_RETAINED_FRAME_COUNT = 74
MAX_RETAINED_FRAME_COUNT = SELECTED_FRAME_COUNT
MAX_PIXELS = 147456


def _snapshot_sequence(value: Sequence[Any], name: str) -> tuple[Any, ...]:
    if not isinstance(value, Sequence) or isinstance(value, (str, bytes, bytearray)):
        raise TypeError(f"{name} must be a non-string sequence")
    return tuple(value)


def build_inference_row(
    frame_paths: Sequence[str | Path],
    source_pts_ms: Sequence[int],
    clip_end_ms: int,
    question: str,
) -> dict[str, Any]:
    """Select, blank-filter, and package one exact frozen inference row.

    ``frame_paths`` and ``source_pts_ms`` are the complete endpoint-inclusive
    clip preselection. Exactly 256 positions are selected first with
    ``(i * (M - 1) + 127) // 255``. High-confidence flat black/blue frames are
    then dropped without refill, preserving absolute PTS gaps.
    """

    paths = _snapshot_sequence(frame_paths, "frame_paths")
    pts = _snapshot_sequence(source_pts_ms, "source_pts_ms")
    if len(paths) != len(pts):
        raise ValueError("frame_paths and source_pts_ms must have equal length")
    if len(paths) < SELECTED_FRAME_COUNT:
        raise ValueError(
            f"preselection has {len(paths)} frames; need at least "
            f"{SELECTED_FRAME_COUNT}"
        )
    if any(not isinstance(value, (str, Path)) for value in paths):
        raise TypeError("frame_paths must contain only str or Path values")
    if any(isinstance(value, bool) or not isinstance(value, int) for value in pts):
        raise TypeError("source_pts_ms must contain integers")
    if any(value < 0 for value in pts):
        raise ValueError("source_pts_ms must be non-negative")
    if any(left >= right for left, right in pairwise(pts)):
        raise ValueError("preselection source_pts_ms must be strictly increasing")
    if isinstance(clip_end_ms, bool) or not isinstance(clip_end_ms, int):
        raise TypeError("clip_end_ms must be an integer")
    if clip_end_ms < 0:
        raise ValueError("clip_end_ms must be non-negative")
    if not isinstance(question, str):
        raise TypeError("question must be a string")
    if not question:
        raise ValueError("question must be non-empty")

    retained_paths: list[str] = []
    retained_pts: list[int] = []
    for index in endpoint_inclusive_indices(len(paths)):
        resolved = str(Path(paths[index]).resolve())
        if classify_flat_blank(resolved) is None:
            retained_paths.append(resolved)
            retained_pts.append(pts[index])

    retained_count = len(retained_paths)
    if retained_count == 0:
        raise ValueError("all 256 selected frames were high-confidence blanks")
    if retained_count < MIN_RETAINED_FRAME_COUNT:
        raise ValueError(
            f"only {retained_count} selected frames remain after blank removal; "
            f"need at least {MIN_RETAINED_FRAME_COUNT}"
        )
    if retained_count > MAX_RETAINED_FRAME_COUNT:
        raise AssertionError("retained frame count exceeds frozen selection")

    source_duration_ms = max(clip_end_ms, retained_pts[-1])
    timed_video = TimedVideo.build(
        retained_paths,
        retained_pts,
        source_duration_ms,
    )
    prefix = build_absolute_timeline_prefix(
        Decimal(retained_pts[0]) / Decimal(1000),
        Decimal(retained_pts[-1]) / Decimal(1000),
    )
    return {
        "messages": [
            {
                "role": "user",
                "content": "<video>" + prefix + question,
            }
        ],
        "videos": [timed_video.as_dict()],
        "chat_template_kwargs": {
            "enable_thinking": False,
            "max_pixels": MAX_PIXELS,
        },
    }