ScoreVision / miner.py
Chris Leo
scorevision: push artifact
eee67f0 verified
Raw
History Blame Contribute Delete
21.2 kB
from pathlib import Path
import math
import cv2
import numpy as np
import onnxruntime as ort
from numpy import ndarray
from pydantic import BaseModel
class BoundingBox(BaseModel):
x1: int
y1: int
x2: int
y2: int
cls_id: int
conf: float
class TVFrameResult(BaseModel):
frame_id: int
boxes: list[BoundingBox]
keypoints: list[tuple[int, int]]
class Miner:
"""
ONNX Runtime miner for Detect-crime.
Recipe (see recipe.md for derivation):
* per-class confidence thresholds (king's lever)
* per-class IoU thresholds (closer pairs allowed for graffiti / spray paint)
* per-class soft-NMS with sigma=0.5
* horizontal-flip TTA merged with **weighted box fusion** (coord-averaged)
* per-class cluster confidence boost (fixes king's cross-class leak)
* no min-side / min-area filter (small TPs survive)
* no h/w aspect gate beyond a sane outlier cap (8x)
"""
class_names = ["balaclava", "hoodie", "glove", "bat", "spray paint", "graffiti"]
input_size = 1280
max_det = 300
max_aspect_ratio = 8.0
soft_sigma = 0.5
_conf_thres_array = np.array(
# balaclava, hoodie, glove, bat, spray paint, graffiti
[0.50, 0.60, 0.30, 0.20, 0.45, 0.30],
dtype=np.float32,
)
_iou_thres_array = np.array(
# hoodie/balaclava: clean people, tight NMS
# glove/bat: small rare objects, mid
# spray/graffiti: legitimately overlapping marks, loose
[0.60, 0.60, 0.55, 0.55, 0.45, 0.45],
dtype=np.float32,
)
def __init__(self, path_hf_repo: Path) -> None:
model_path = path_hf_repo / "weights.onnx"
print("ORT version:", ort.__version__)
try:
ort.preload_dlls()
print("preload_dlls success")
except Exception as e:
print(f"preload_dlls failed: {e}")
print("ORT providers available:", ort.get_available_providers())
sess_options = ort.SessionOptions()
sess_options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_ENABLE_ALL
try:
self.session = ort.InferenceSession(
str(model_path),
sess_options=sess_options,
providers=["CUDAExecutionProvider", "CPUExecutionProvider"],
)
print("ORT session: CUDA provider")
except Exception as e:
print(f"CUDA session creation failed, falling back to CPU: {e}")
self.session = ort.InferenceSession(
str(model_path),
sess_options=sess_options,
providers=["CPUExecutionProvider"],
)
print("ORT session providers:", self.session.get_providers())
for inp in self.session.get_inputs():
print("INPUT:", inp.name, inp.shape, inp.type)
for out in self.session.get_outputs():
print("OUTPUT:", out.name, out.shape, out.type)
self.input_name = self.session.get_inputs()[0].name
self.output_names = [o.name for o in self.session.get_outputs()]
self.input_shape = self.session.get_inputs()[0].shape
# Detect FP16 vs FP32 input from the session metadata.
input_type = self.session.get_inputs()[0].type
self.input_dtype = np.float16 if "float16" in input_type else np.float32
self.input_height = self._safe_dim(self.input_shape[2], default=self.input_size)
self.input_width = self._safe_dim(self.input_shape[3], default=self.input_size)
print(f"ONNX model loaded: {model_path}")
print("per-class conf: " + ", ".join(
f"{n}={t:.2f}" for n, t in zip(self.class_names,
self._conf_thres_array.tolist())))
print("per-class iou : " + ", ".join(
f"{n}={t:.2f}" for n, t in zip(self.class_names,
self._iou_thres_array.tolist())))
def __repr__(self) -> str:
return (
f"ONNXRuntime(session={type(self.session).__name__}, "
f"providers={self.session.get_providers()})"
)
@staticmethod
def _safe_dim(value, default: int) -> int:
return value if isinstance(value, int) and value > 0 else default
# ---------------------------------------------------------------- preproc
def _letterbox(
self,
image: ndarray,
new_shape: tuple[int, int],
color=(114, 114, 114),
) -> tuple[ndarray, float, tuple[float, float]]:
h, w = image.shape[:2]
new_w, new_h = new_shape
ratio = min(new_w / w, new_h / h)
resized_w = int(round(w * ratio))
resized_h = int(round(h * ratio))
if (resized_w, resized_h) != (w, h):
interp = cv2.INTER_CUBIC if ratio > 1.0 else cv2.INTER_LINEAR
image = cv2.resize(image, (resized_w, resized_h), interpolation=interp)
dw = (new_w - resized_w) / 2.0
dh = (new_h - resized_h) / 2.0
left = int(round(dw - 0.1))
right = int(round(dw + 0.1))
top = int(round(dh - 0.1))
bottom = int(round(dh + 0.1))
padded = cv2.copyMakeBorder(
image, top, bottom, left, right,
borderType=cv2.BORDER_CONSTANT, value=color,
)
return padded, ratio, (dw, dh)
def _preprocess(
self, image: ndarray
) -> tuple[np.ndarray, float, tuple[float, float], tuple[int, int]]:
orig_h, orig_w = image.shape[:2]
img, ratio, pad = self._letterbox(image, (self.input_width, self.input_height))
img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)
img = img.astype(np.float32) / 255.0
img = np.transpose(img, (2, 0, 1))[None, ...]
img = np.ascontiguousarray(img, dtype=self.input_dtype)
return img, ratio, pad, (orig_w, orig_h)
# ---------------------------------------------------------------- helpers
@staticmethod
def _clip_boxes(boxes: np.ndarray, image_size: tuple[int, int]) -> np.ndarray:
w, h = image_size
boxes[:, 0] = np.clip(boxes[:, 0], 0, w - 1)
boxes[:, 1] = np.clip(boxes[:, 1], 0, h - 1)
boxes[:, 2] = np.clip(boxes[:, 2], 0, w - 1)
boxes[:, 3] = np.clip(boxes[:, 3], 0, h - 1)
return boxes
@staticmethod
def _xywh_to_xyxy(boxes: np.ndarray) -> np.ndarray:
out = np.empty_like(boxes)
out[:, 0] = boxes[:, 0] - boxes[:, 2] / 2.0
out[:, 1] = boxes[:, 1] - boxes[:, 3] / 2.0
out[:, 2] = boxes[:, 0] + boxes[:, 2] / 2.0
out[:, 3] = boxes[:, 1] + boxes[:, 3] / 2.0
return out
@staticmethod
def _iou_matrix(a: np.ndarray, b: np.ndarray) -> np.ndarray:
if len(a) == 0 or len(b) == 0:
return np.zeros((len(a), len(b)), dtype=np.float32)
ax1, ay1, ax2, ay2 = a[:, 0:1], a[:, 1:2], a[:, 2:3], a[:, 3:4]
bx1, by1, bx2, by2 = b[:, 0], b[:, 1], b[:, 2], b[:, 3]
ix1 = np.maximum(ax1, bx1)
iy1 = np.maximum(ay1, by1)
ix2 = np.minimum(ax2, bx2)
iy2 = np.minimum(ay2, by2)
inter = np.maximum(0.0, ix2 - ix1) * np.maximum(0.0, iy2 - iy1)
area_a = np.maximum(0.0, ax2 - ax1) * np.maximum(0.0, ay2 - ay1)
area_b = np.maximum(0.0, bx2 - bx1) * np.maximum(0.0, by2 - by1)
union = area_a + area_b - inter + 1e-7
return (inter / union).astype(np.float32)
# ---------------------------------------------------------------- NMS
def _soft_nms(
self, boxes: np.ndarray, scores: np.ndarray, sigma: float,
score_thresh: float = 0.001,
) -> tuple[np.ndarray, np.ndarray]:
n = len(boxes)
if n == 0:
return np.array([], dtype=np.intp), np.array([], dtype=np.float32)
boxes = boxes.astype(np.float32, copy=True)
scores = scores.astype(np.float32, copy=True)
order = np.arange(n)
for i in range(n):
max_pos = i + int(np.argmax(scores[i:]))
boxes[[i, max_pos]] = boxes[[max_pos, i]]
scores[[i, max_pos]] = scores[[max_pos, i]]
order[[i, max_pos]] = order[[max_pos, i]]
if i + 1 >= n:
break
xx1 = np.maximum(boxes[i, 0], boxes[i + 1:, 0])
yy1 = np.maximum(boxes[i, 1], boxes[i + 1:, 1])
xx2 = np.minimum(boxes[i, 2], boxes[i + 1:, 2])
yy2 = np.minimum(boxes[i, 3], boxes[i + 1:, 3])
inter = np.maximum(0.0, xx2 - xx1) * np.maximum(0.0, yy2 - yy1)
a_i = max(0.0, float(
(boxes[i, 2] - boxes[i, 0]) * (boxes[i, 3] - boxes[i, 1])
))
a_j = (
np.maximum(0.0, boxes[i + 1:, 2] - boxes[i + 1:, 0]) *
np.maximum(0.0, boxes[i + 1:, 3] - boxes[i + 1:, 1])
)
iou = inter / (a_i + a_j - inter + 1e-7)
scores[i + 1:] *= np.exp(-(iou ** 2) / sigma)
mask = scores > score_thresh
return order[mask], scores[mask]
def _per_class_soft_nms(
self, boxes: np.ndarray, scores: np.ndarray, cls_ids: np.ndarray,
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
if len(boxes) == 0:
return boxes, scores, cls_ids
out_b: list = []
out_s: list = []
out_c: list = []
for c in np.unique(cls_ids):
mask = cls_ids == c
sub_b = boxes[mask]
sub_s = scores[mask]
idx, decayed = self._soft_nms(sub_b, sub_s, self.soft_sigma)
if len(idx) == 0:
continue
out_b.append(sub_b[idx])
out_s.append(decayed)
out_c.append(np.full(len(idx), c, dtype=cls_ids.dtype))
if not out_b:
return (np.empty((0, 4), dtype=np.float32),
np.empty((0,), dtype=np.float32),
np.empty((0,), dtype=cls_ids.dtype))
return (np.concatenate(out_b, axis=0),
np.concatenate(out_s, axis=0),
np.concatenate(out_c, axis=0))
# ---------------------------------------------------------------- WBF
def _weighted_box_fusion(
self,
boxes: np.ndarray,
scores: np.ndarray,
cls_ids: np.ndarray,
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
"""Per-class confidence-weighted box fusion across orig+flip detections."""
if len(boxes) == 0:
return boxes, scores, cls_ids
fused_b: list = []
fused_s: list = []
fused_c: list = []
for c in np.unique(cls_ids):
mask = cls_ids == c
sub_b = boxes[mask].astype(np.float32)
sub_s = scores[mask].astype(np.float32)
iou_thr = float(self._iou_thres_array[int(c)])
order = np.argsort(-sub_s)
sub_b = sub_b[order]
sub_s = sub_s[order]
used = np.zeros(len(sub_b), dtype=bool)
for i in range(len(sub_b)):
if used[i]:
continue
used[i] = True
cluster_b = [sub_b[i]]
cluster_s = [sub_s[i]]
if i + 1 < len(sub_b):
rest = sub_b[i + 1:]
ious = self._iou_matrix(sub_b[i:i + 1], rest)[0]
for j_offset, iou in enumerate(ious):
j = i + 1 + j_offset
if used[j]:
continue
if iou >= iou_thr:
used[j] = True
cluster_b.append(sub_b[j])
cluster_s.append(sub_s[j])
ws = np.asarray(cluster_s, dtype=np.float32)
bs = np.asarray(cluster_b, dtype=np.float32)
w_sum = float(ws.sum())
if w_sum <= 0:
continue
fused_box = (bs * ws[:, None]).sum(axis=0) / w_sum
# cluster confidence: max member, slight boost when >1 supporter
support = len(cluster_s)
fused_conf = float(ws.max())
if support > 1:
fused_conf = min(1.0, fused_conf * (1.0 + 0.10 * (support - 1)))
fused_b.append(fused_box)
fused_s.append(fused_conf)
fused_c.append(int(c))
if not fused_b:
return (np.empty((0, 4), dtype=np.float32),
np.empty((0,), dtype=np.float32),
np.empty((0,), dtype=cls_ids.dtype))
return (
np.asarray(fused_b, dtype=np.float32),
np.asarray(fused_s, dtype=np.float32),
np.asarray(fused_c, dtype=cls_ids.dtype),
)
# ---------------------------------------------------------------- sanity
def _filter_sane_boxes(
self, boxes: np.ndarray, scores: np.ndarray, cls_ids: np.ndarray,
orig_size: tuple[int, int],
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
if len(boxes) == 0:
return boxes, scores, cls_ids
orig_w, orig_h = orig_size
image_area = float(orig_w * orig_h)
bw = np.maximum(0.0, boxes[:, 2] - boxes[:, 0])
bh = np.maximum(0.0, boxes[:, 3] - boxes[:, 1])
area = bw * bh
ar = np.where(
(bw > 0) & (bh > 0),
np.maximum(bw / np.maximum(bh, 1e-6), bh / np.maximum(bw, 1e-6)),
np.inf,
)
keep = (area > 0) & (area <= 0.95 * image_area) & (ar <= self.max_aspect_ratio)
return boxes[keep], scores[keep], cls_ids[keep]
# ---------------------------------------------------------------- decode
def _per_view_pipeline(
self, boxes: np.ndarray, scores: np.ndarray, cls_ids: np.ndarray,
orig_size: tuple[int, int],
) -> tuple[np.ndarray, np.ndarray, np.ndarray]:
boxes, scores, cls_ids = self._filter_sane_boxes(boxes, scores, cls_ids, orig_size)
if len(boxes) == 0:
return boxes, scores, cls_ids
if len(boxes) > 1:
boxes, scores, cls_ids = self._per_class_soft_nms(boxes, scores, cls_ids)
if len(scores) > self.max_det:
top = np.argsort(-scores)[: self.max_det]
boxes, scores, cls_ids = boxes[top], scores[top], cls_ids[top]
return boxes, scores, cls_ids
def _decode_final_dets(
self, preds: np.ndarray, ratio: float, pad: tuple[float, float],
orig_size: tuple[int, int],
) -> list[BoundingBox]:
if preds.ndim == 3 and preds.shape[0] == 1:
preds = preds[0]
if preds.ndim != 2 or preds.shape[1] < 6:
raise ValueError(f"Unexpected final-det output shape: {preds.shape}")
boxes = preds[:, :4].astype(np.float32)
scores = preds[:, 4].astype(np.float32)
cls_ids = preds[:, 5].astype(np.int32)
keep = scores >= self._conf_thres_array[cls_ids]
boxes, scores, cls_ids = boxes[keep], scores[keep], cls_ids[keep]
if len(boxes) == 0:
return []
pad_w, pad_h = pad
boxes[:, [0, 2]] -= pad_w
boxes[:, [1, 3]] -= pad_h
boxes /= ratio
boxes = self._clip_boxes(boxes, orig_size)
boxes, scores, cls_ids = self._per_view_pipeline(boxes, scores, cls_ids, orig_size)
return self._build_results(boxes, scores, cls_ids)
def _decode_raw_yolo(
self, preds: np.ndarray, ratio: float, pad: tuple[float, float],
orig_size: tuple[int, int],
) -> list[BoundingBox]:
if preds.ndim != 3 or preds.shape[0] != 1:
raise ValueError(f"Unexpected raw output shape: {preds.shape}")
preds = preds[0]
if preds.shape[0] <= 16 and preds.shape[1] > preds.shape[0]:
preds = preds.T
if preds.ndim != 2 or preds.shape[1] < 5:
raise ValueError(f"Unexpected raw output shape: {preds.shape}")
boxes_xywh = preds[:, :4].astype(np.float32)
cls_part = preds[:, 4:].astype(np.float32)
if cls_part.shape[1] == 1:
scores = cls_part[:, 0]
cls_ids = np.zeros(len(scores), dtype=np.int32)
else:
cls_ids = np.argmax(cls_part, axis=1).astype(np.int32)
scores = cls_part[np.arange(len(cls_part)), cls_ids]
keep = scores >= self._conf_thres_array[cls_ids]
boxes_xywh, scores, cls_ids = boxes_xywh[keep], scores[keep], cls_ids[keep]
if len(boxes_xywh) == 0:
return []
boxes = self._xywh_to_xyxy(boxes_xywh)
pad_w, pad_h = pad
boxes[:, [0, 2]] -= pad_w
boxes[:, [1, 3]] -= pad_h
boxes /= ratio
boxes = self._clip_boxes(boxes, orig_size)
boxes, scores, cls_ids = self._per_view_pipeline(boxes, scores, cls_ids, orig_size)
return self._build_results(boxes, scores, cls_ids)
@staticmethod
def _build_results(
boxes: np.ndarray, scores: np.ndarray, cls_ids: np.ndarray,
) -> list[BoundingBox]:
results: list[BoundingBox] = []
for box, conf, cls_id in zip(boxes, scores, cls_ids):
x1, y1, x2, y2 = box.tolist()
if x2 <= x1 or y2 <= y1:
continue
results.append(
BoundingBox(
x1=int(math.floor(x1)),
y1=int(math.floor(y1)),
x2=int(math.ceil(x2)),
y2=int(math.ceil(y2)),
cls_id=int(cls_id),
conf=float(conf),
)
)
return results
def _postprocess(
self, output: np.ndarray, ratio: float, pad: tuple[float, float],
orig_size: tuple[int, int],
) -> list[BoundingBox]:
if output.ndim == 2 and output.shape[1] >= 6:
return self._decode_final_dets(output, ratio, pad, orig_size)
if output.ndim == 3 and output.shape[0] == 1 and output.shape[2] == 6:
return self._decode_final_dets(output, ratio, pad, orig_size)
return self._decode_raw_yolo(output, ratio, pad, orig_size)
# ---------------------------------------------------------------- inference
def _predict_single(self, image: np.ndarray) -> list[BoundingBox]:
if image is None:
raise ValueError("Input image is None")
if not isinstance(image, np.ndarray):
raise TypeError(f"Input is not numpy array: {type(image)}")
if image.ndim != 3:
raise ValueError(f"Expected HWC image, got shape={image.shape}")
if image.shape[2] != 3:
raise ValueError(f"Expected 3 channels, got shape={image.shape}")
if image.dtype != np.uint8:
image = image.astype(np.uint8)
input_tensor, ratio, pad, orig_size = self._preprocess(image)
expected = (1, 3, self.input_height, self.input_width)
if input_tensor.shape != expected:
raise ValueError(
f"Bad input tensor shape={input_tensor.shape}, expected={expected}"
)
outputs = self.session.run(self.output_names, {self.input_name: input_tensor})
return self._postprocess(outputs[0], ratio, pad, orig_size)
def _predict_tta(self, image: np.ndarray) -> list[BoundingBox]:
boxes_orig = self._predict_single(image)
flipped = cv2.flip(image, 1)
boxes_flip = self._predict_single(flipped)
w = image.shape[1]
boxes_flip = [
BoundingBox(
x1=w - b.x2, y1=b.y1, x2=w - b.x1, y2=b.y2,
cls_id=b.cls_id, conf=b.conf,
)
for b in boxes_flip
]
all_boxes = boxes_orig + boxes_flip
if not all_boxes:
return []
coords = np.array(
[[b.x1, b.y1, b.x2, b.y2] for b in all_boxes], dtype=np.float32
)
scores = np.array([b.conf for b in all_boxes], dtype=np.float32)
cls_ids = np.array([b.cls_id for b in all_boxes], dtype=np.int32)
# Per-class weighted box fusion (replaces hard-NMS + cluster-boost)
boxes_f, scores_f, cls_f = self._weighted_box_fusion(coords, scores, cls_ids)
if len(boxes_f) == 0:
return []
if len(scores_f) > self.max_det:
top = np.argsort(-scores_f)[: self.max_det]
boxes_f, scores_f, cls_f = boxes_f[top], scores_f[top], cls_f[top]
return self._build_results(boxes_f, scores_f, cls_f)
def predict_batch(
self, batch_images: list[ndarray], offset: int, n_keypoints: int,
) -> list[TVFrameResult]:
results: list[TVFrameResult] = []
for frame_number_in_batch, image in enumerate(batch_images):
try:
boxes = self._predict_tta(image)
except Exception as e:
print(
f"Inference failed for frame "
f"{offset + frame_number_in_batch}: {e}"
)
boxes = []
results.append(
TVFrameResult(
frame_id=offset + frame_number_in_batch,
boxes=boxes,
keypoints=[(0, 0) for _ in range(max(0, int(n_keypoints)))],
)
)
return results