ScoreVision / miner.py
dev0524's picture
scorevision: push artifact
9c25ca5 verified
Raw History Blame Contribute Delete
3.71 kB
"""
TurboVision open-source miner for element `manak0/Detect-Person`.
Contract (enforced by the chute template):
- file must be `miner.py` at the root of the Hugging Face repo
- class must be named `Miner`
- `predict_batch(batch_images, offset, n_keypoints) -> list[TVFrameResult]`
Scoring notes for this element:
- objects = ["person"], so every box must use cls_id=0; other ids are dropped
by the validator's parser.
- Pillars: map50 (weight 0.6) and false_positive (weight 0.4), where
false_positive = 1 - FFPI/10. Each unmatched prediction per image costs 0.1
on that pillar, so the confidence threshold below trades recall against FPs.
"""
import os
from pathlib import Path
from numpy import ndarray
from pydantic import BaseModel
from ultralytics import YOLO
PERSON_CLS_ID = 0 # index into the element's objects list AND COCO's person id
CONF_THRESHOLD = float(os.getenv("SV_CONF_THRESHOLD", "0.20"))
IMG_SIZE = int(os.getenv("SV_IMG_SIZE", "416"))
# Fine-tuned weights produce more duplicate boxes on the same person at the
# ultralytics default (0.7); tightening NMS collapses most of them without
# losing genuinely separate closely-spaced people.
NMS_IOU = float(os.getenv("SV_NMS_IOU", "0.3"))
class BoundingBox(BaseModel):
x1: int
y1: int
x2: int
y2: int
cls_id: int
conf: float
class Polygon(BaseModel):
cls_id: int
conf: float
points: list[tuple[int, int]]
class TVFrameResult(BaseModel):
frame_id: int
boxes: list[BoundingBox] | None = None
polygons: list[Polygon] | None = None
keypoints: list[tuple[int, int]] | None = None
class Miner:
def __init__(self, path_hf_repo: Path) -> None:
"""Load the person detector, preferring an ONNX export (faster on CPU)."""
onnx_path = path_hf_repo / "person.onnx"
pt_path = path_hf_repo / "yolo11n.pt"
if onnx_path.exists():
self.model = YOLO(str(onnx_path), task="detect")
self.model_name = onnx_path.name
else:
self.model = YOLO(str(pt_path))
self.model_name = pt_path.name
print(f"✅ Person detector loaded: {self.model_name}")
def __repr__(self) -> str:
return (
f"Person detector: {self.model_name} "
f"(conf={CONF_THRESHOLD}, imgsz={IMG_SIZE}, iou={NMS_IOU})"
)
def predict_batch(
self,
batch_images: list[ndarray],
offset: int,
n_keypoints: int,
) -> list[TVFrameResult]:
# device="cpu": chute GPUs (Blackwell) lack kernels for the image's
# torch build, and the ONNX model targets CPU for the latency gate.
detections = self.model.predict(
batch_images,
conf=CONF_THRESHOLD,
imgsz=IMG_SIZE,
iou=NMS_IOU,
classes=[PERSON_CLS_ID],
device=os.getenv("SV_DEVICE", "cpu"),
verbose=False,
)
results: list[TVFrameResult] = []
for i, detection in enumerate(detections):
boxes: list[BoundingBox] = []
if getattr(detection, "boxes", None) is not None:
for box in detection.boxes.data:
x1, y1, x2, y2, conf, _cls = box.tolist()
boxes.append(
BoundingBox(
x1=int(x1),
y1=int(y1),
x2=int(x2),
y2=int(y2),
cls_id=PERSON_CLS_ID,
conf=float(conf),
)
)
results.append(TVFrameResult(frame_id=offset + i, boxes=boxes))
return results