gwd200's picture
Publish PROCEDURE ep15 merged3225 self-contained runtime v1
f6588d9 verified
Raw History Blame Contribute Delete
4.13 kB
"""Build the one frozen answer-free realtime inference-row surface."""
from __future__ import annotations
from collections.abc import Sequence
from decimal import Decimal
from itertools import pairwise
from pathlib import Path
from typing import Any
from .extractor import SELECTED_FRAME_COUNT, endpoint_inclusive_indices
from .prompt import build_absolute_timeline_prefix
from .realtime import TimedVideo, classify_flat_blank
MIN_RETAINED_FRAME_COUNT = 74
MAX_RETAINED_FRAME_COUNT = SELECTED_FRAME_COUNT
MAX_PIXELS = 147456
def _snapshot_sequence(value: Sequence[Any], name: str) -> tuple[Any, ...]:
if not isinstance(value, Sequence) or isinstance(value, (str, bytes, bytearray)):
raise TypeError(f"{name} must be a non-string sequence")
return tuple(value)
def build_inference_row(
frame_paths: Sequence[str | Path],
source_pts_ms: Sequence[int],
clip_end_ms: int,
question: str,
) -> dict[str, Any]:
"""Select, blank-filter, and package one exact frozen inference row.
``frame_paths`` and ``source_pts_ms`` are the complete endpoint-inclusive
clip preselection. Exactly 256 positions are selected first with
``(i * (M - 1) + 127) // 255``. High-confidence flat black/blue frames are
then dropped without refill, preserving absolute PTS gaps.
"""
paths = _snapshot_sequence(frame_paths, "frame_paths")
pts = _snapshot_sequence(source_pts_ms, "source_pts_ms")
if len(paths) != len(pts):
raise ValueError("frame_paths and source_pts_ms must have equal length")
if len(paths) < SELECTED_FRAME_COUNT:
raise ValueError(
f"preselection has {len(paths)} frames; need at least "
f"{SELECTED_FRAME_COUNT}"
)
if any(not isinstance(value, (str, Path)) for value in paths):
raise TypeError("frame_paths must contain only str or Path values")
if any(isinstance(value, bool) or not isinstance(value, int) for value in pts):
raise TypeError("source_pts_ms must contain integers")
if any(value < 0 for value in pts):
raise ValueError("source_pts_ms must be non-negative")
if any(left >= right for left, right in pairwise(pts)):
raise ValueError("preselection source_pts_ms must be strictly increasing")
if isinstance(clip_end_ms, bool) or not isinstance(clip_end_ms, int):
raise TypeError("clip_end_ms must be an integer")
if clip_end_ms < 0:
raise ValueError("clip_end_ms must be non-negative")
if not isinstance(question, str):
raise TypeError("question must be a string")
if not question:
raise ValueError("question must be non-empty")
retained_paths: list[str] = []
retained_pts: list[int] = []
for index in endpoint_inclusive_indices(len(paths)):
resolved = str(Path(paths[index]).resolve())
if classify_flat_blank(resolved) is None:
retained_paths.append(resolved)
retained_pts.append(pts[index])
retained_count = len(retained_paths)
if retained_count == 0:
raise ValueError("all 256 selected frames were high-confidence blanks")
if retained_count < MIN_RETAINED_FRAME_COUNT:
raise ValueError(
f"only {retained_count} selected frames remain after blank removal; "
f"need at least {MIN_RETAINED_FRAME_COUNT}"
)
if retained_count > MAX_RETAINED_FRAME_COUNT:
raise AssertionError("retained frame count exceeds frozen selection")
source_duration_ms = max(clip_end_ms, retained_pts[-1])
timed_video = TimedVideo.build(
retained_paths,
retained_pts,
source_duration_ms,
)
prefix = build_absolute_timeline_prefix(
Decimal(retained_pts[0]) / Decimal(1000),
Decimal(retained_pts[-1]) / Decimal(1000),
)
return {
"messages": [
{
"role": "user",
"content": "<video>" + prefix + question,
}
],
"videos": [timed_video.as_dict()],
"chat_template_kwargs": {
"enable_thinking": False,
"max_pixels": MAX_PIXELS,
},
}