File size: 3,705 Bytes
4dd2463 489ae67 9c25ca5 4dd2463 9c25ca5 4dd2463 cfc3b0a 4dd2463 9c25ca5 4dd2463 cfc3b0a 4dd2463 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 | """
TurboVision open-source miner for element `manak0/Detect-Person`.
Contract (enforced by the chute template):
- file must be `miner.py` at the root of the Hugging Face repo
- class must be named `Miner`
- `predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]`
Scoring notes for this element:
- objects = ["person"], so every box must use cls_id=0; other ids are dropped
by the validator's parser.
- Pillars: map50 (weight 0.6) and false_positive (weight 0.4), where
false_positive = 1 - FFPI/10. Each unmatched prediction per image costs 0.1
on that pillar, so the confidence threshold below trades recall against FPs.
"""
import os
from pathlib import Path
from numpy import ndarray
from pydantic import BaseModel
from ultralytics import YOLO
PERSON_CLS_ID = 0 # index into the element's objects list AND COCO's person id
CONF_THRESHOLD = float(os.getenv("SV_CONF_THRESHOLD", "0.20"))
IMG_SIZE = int(os.getenv("SV_IMG_SIZE", "416"))
# Fine-tuned weights produce more duplicate boxes on the same person at the
# ultralytics default (0.7); tightening NMS collapses most of them without
# losing genuinely separate closely-spaced people.
NMS_IOU = float(os.getenv("SV_NMS_IOU", "0.3"))
class BoundingBox(BaseModel):
x1: int
y1: int
x2: int
y2: int
cls_id: int
conf: float
class Polygon(BaseModel):
cls_id: int
conf: float
points: list[tuple[int, int]]
class TVFrameResult(BaseModel):
frame_id: int
boxes: list[BoundingBox] | None = None
polygons: list[Polygon] | None = None
keypoints: list[tuple[int, int]] | None = None
class Miner:
def __init__(self, path_hf_repo: Path) -> None:
"""Load the person detector, preferring an ONNX export (faster on CPU)."""
onnx_path = path_hf_repo / "person.onnx"
pt_path = path_hf_repo / "yolo11n.pt"
if onnx_path.exists():
self.model = YOLO(str(onnx_path), task="detect")
self.model_name = onnx_path.name
else:
self.model = YOLO(str(pt_path))
self.model_name = pt_path.name
print(f"✅ Person detector loaded: {self.model_name}")
def __repr__(self) -> str:
return (
f"Person detector: {self.model_name} "
f"(conf={CONF_THRESHOLD}, imgsz={IMG_SIZE}, iou={NMS_IOU})"
)
def predict_batch(
self,
batch_images: list[ndarray],
offset: int,
n_keypoints: int,
) -> list[TVFrameResult]:
# device="cpu": chute GPUs (Blackwell) lack kernels for the image's
# torch build, and the ONNX model targets CPU for the latency gate.
detections = self.model.predict(
batch_images,
conf=CONF_THRESHOLD,
imgsz=IMG_SIZE,
iou=NMS_IOU,
classes=[PERSON_CLS_ID],
device=os.getenv("SV_DEVICE", "cpu"),
verbose=False,
)
results: list[TVFrameResult] = []
for i, detection in enumerate(detections):
boxes: list[BoundingBox] = []
if getattr(detection, "boxes", None) is not None:
for box in detection.boxes.data:
x1, y1, x2, y2, conf, _cls = box.tolist()
boxes.append(
BoundingBox(
x1=int(x1),
y1=int(y1),
x2=int(x2),
y2=int(y2),
cls_id=PERSON_CLS_ID,
conf=float(conf),
)
)
results.append(TVFrameResult(frame_id=offset + i, boxes=boxes))
return results
|