File size: 3,705 Bytes
4dd2463
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
489ae67
 
9c25ca5
 
 
 
4dd2463
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
9c25ca5
4dd2463
 
 
 
 
 
 
 
cfc3b0a
 
4dd2463
 
 
 
9c25ca5
4dd2463
cfc3b0a
4dd2463
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
"""
TurboVision open-source miner for element `manak0/Detect-Person`.

Contract (enforced by the chute template):
- file must be `miner.py` at the root of the Hugging Face repo
- class must be named `Miner`
- `predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]`

Scoring notes for this element:
- objects = ["person"], so every box must use cls_id=0; other ids are dropped
  by the validator's parser.
- Pillars: map50 (weight 0.6) and false_positive (weight 0.4), where
  false_positive = 1 - FFPI/10. Each unmatched prediction per image costs 0.1
  on that pillar, so the confidence threshold below trades recall against FPs.
"""

import os
from pathlib import Path

from numpy import ndarray
from pydantic import BaseModel
from ultralytics import YOLO

PERSON_CLS_ID = 0  # index into the element's objects list AND COCO's person id
CONF_THRESHOLD = float(os.getenv("SV_CONF_THRESHOLD", "0.20"))
IMG_SIZE = int(os.getenv("SV_IMG_SIZE", "416"))
# Fine-tuned weights produce more duplicate boxes on the same person at the
# ultralytics default (0.7); tightening NMS collapses most of them without
# losing genuinely separate closely-spaced people.
NMS_IOU = float(os.getenv("SV_NMS_IOU", "0.3"))


class BoundingBox(BaseModel):
    x1: int
    y1: int
    x2: int
    y2: int
    cls_id: int
    conf: float


class Polygon(BaseModel):
    cls_id: int
    conf: float
    points: list[tuple[int, int]]


class TVFrameResult(BaseModel):
    frame_id: int
    boxes: list[BoundingBox] | None = None
    polygons: list[Polygon] | None = None
    keypoints: list[tuple[int, int]] | None = None


class Miner:
    def __init__(self, path_hf_repo: Path) -> None:
        """Load the person detector, preferring an ONNX export (faster on CPU)."""
        onnx_path = path_hf_repo / "person.onnx"
        pt_path = path_hf_repo / "yolo11n.pt"
        if onnx_path.exists():
            self.model = YOLO(str(onnx_path), task="detect")
            self.model_name = onnx_path.name
        else:
            self.model = YOLO(str(pt_path))
            self.model_name = pt_path.name
        print(f"✅ Person detector loaded: {self.model_name}")

    def __repr__(self) -> str:
        return (
            f"Person detector: {self.model_name} "
            f"(conf={CONF_THRESHOLD}, imgsz={IMG_SIZE}, iou={NMS_IOU})"
        )

    def predict_batch(
        self,
        batch_images: list[ndarray],
        offset: int,
        n_keypoints: int,
    ) -> list[TVFrameResult]:
        # device="cpu": chute GPUs (Blackwell) lack kernels for the image's
        # torch build, and the ONNX model targets CPU for the latency gate.
        detections = self.model.predict(
            batch_images,
            conf=CONF_THRESHOLD,
            imgsz=IMG_SIZE,
            iou=NMS_IOU,
            classes=[PERSON_CLS_ID],
            device=os.getenv("SV_DEVICE", "cpu"),
            verbose=False,
        )

        results: list[TVFrameResult] = []
        for i, detection in enumerate(detections):
            boxes: list[BoundingBox] = []
            if getattr(detection, "boxes", None) is not None:
                for box in detection.boxes.data:
                    x1, y1, x2, y2, conf, _cls = box.tolist()
                    boxes.append(
                        BoundingBox(
                            x1=int(x1),
                            y1=int(y1),
                            x2=int(x2),
                            y2=int(y2),
                            cls_id=PERSON_CLS_ID,
                            conf=float(conf),
                        )
                    )
            results.append(TVFrameResult(frame_id=offset + i, boxes=boxes))
        return results