dev0524 commited on
Commit
4dd2463
·
verified ·
1 Parent(s): f839f40

scorevision: push artifact

Browse files
Files changed (1) hide show
  1. miner.py +101 -0
miner.py ADDED
@@ -0,0 +1,101 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ TurboVision open-source miner for element `manak0/Detect-Person`.
3
+
4
+ Contract (enforced by the chute template):
5
+ - file must be `miner.py` at the root of the Hugging Face repo
6
+ - class must be named `Miner`
7
+ - `predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]`
8
+
9
+ Scoring notes for this element:
10
+ - objects = ["person"], so every box must use cls_id=0; other ids are dropped
11
+ by the validator's parser.
12
+ - Pillars: map50 (weight 0.6) and false_positive (weight 0.4), where
13
+ false_positive = 1 - FFPI/10. Each unmatched prediction per image costs 0.1
14
+ on that pillar, so the confidence threshold below trades recall against FPs.
15
+ """
16
+
17
+ import os
18
+ from pathlib import Path
19
+
20
+ from numpy import ndarray
21
+ from pydantic import BaseModel
22
+ from ultralytics import YOLO
23
+
24
+ PERSON_CLS_ID = 0 # index into the element's objects list AND COCO's person id
25
+ CONF_THRESHOLD = float(os.getenv("SV_CONF_THRESHOLD", "0.35"))
26
+ IMG_SIZE = int(os.getenv("SV_IMG_SIZE", "480"))
27
+
28
+
29
+ class BoundingBox(BaseModel):
30
+ x1: int
31
+ y1: int
32
+ x2: int
33
+ y2: int
34
+ cls_id: int
35
+ conf: float
36
+
37
+
38
+ class Polygon(BaseModel):
39
+ cls_id: int
40
+ conf: float
41
+ points: list[tuple[int, int]]
42
+
43
+
44
+ class TVFrameResult(BaseModel):
45
+ frame_id: int
46
+ boxes: list[BoundingBox] | None = None
47
+ polygons: list[Polygon] | None = None
48
+ keypoints: list[tuple[int, int]] | None = None
49
+
50
+
51
+ class Miner:
52
+ def __init__(self, path_hf_repo: Path) -> None:
53
+ """Load the person detector, preferring an ONNX export (faster on CPU)."""
54
+ onnx_path = path_hf_repo / "person.onnx"
55
+ pt_path = path_hf_repo / "yolo11n.pt"
56
+ if onnx_path.exists():
57
+ self.model = YOLO(str(onnx_path), task="detect")
58
+ self.model_name = onnx_path.name
59
+ else:
60
+ self.model = YOLO(str(pt_path))
61
+ self.model_name = pt_path.name
62
+ print(f"✅ Person detector loaded: {self.model_name}")
63
+
64
+ def __repr__(self) -> str:
65
+ return (
66
+ f"Person detector: {self.model_name} "
67
+ f"(conf={CONF_THRESHOLD}, imgsz={IMG_SIZE})"
68
+ )
69
+
70
+ def predict_batch(
71
+ self,
72
+ batch_images: list[ndarray],
73
+ offset: int,
74
+ n_keypoints: int,
75
+ ) -> list[TVFrameResult]:
76
+ detections = self.model.predict(
77
+ batch_images,
78
+ conf=CONF_THRESHOLD,
79
+ imgsz=IMG_SIZE,
80
+ classes=[PERSON_CLS_ID],
81
+ verbose=False,
82
+ )
83
+
84
+ results: list[TVFrameResult] = []
85
+ for i, detection in enumerate(detections):
86
+ boxes: list[BoundingBox] = []
87
+ if getattr(detection, "boxes", None) is not None:
88
+ for box in detection.boxes.data:
89
+ x1, y1, x2, y2, conf, _cls = box.tolist()
90
+ boxes.append(
91
+ BoundingBox(
92
+ x1=int(x1),
93
+ y1=int(y1),
94
+ x2=int(x2),
95
+ y2=int(y2),
96
+ cls_id=PERSON_CLS_ID,
97
+ conf=float(conf),
98
+ )
99
+ )
100
+ results.append(TVFrameResult(frame_id=offset + i, boxes=boxes))
101
+ return results