File size: 4,353 Bytes
9c98083 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 | """Subject bbox detection for smart cropping.
Priority: mediapipe face detection -> OpenCV Haar cascade face detection
(no model download required, always available with opencv-python) ->
ultralytics YOLOv8n person detection -> full frame (no crop).
All bboxes are returned as (x1, y1, x2, y2) in pixel space, plus a "kind"
string ("face" | "person" | "full_frame") used by crop_resize.py to pick a
padding multiplier.
"""
import cv2
import numpy as np
_mp_detector = None
_mp_unavailable = False
_haar_cascade = None
_yolo_model = None
def _get_mediapipe_detector(min_conf):
global _mp_detector, _mp_unavailable
if _mp_unavailable:
return None
if _mp_detector is None:
try:
import mediapipe as mp
_mp_detector = mp.solutions.face_detection.FaceDetection(
model_selection=1, min_detection_confidence=min_conf
)
except Exception:
_mp_unavailable = True
return None
return _mp_detector
def _get_haar_cascade():
global _haar_cascade
if _haar_cascade is None:
path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml"
_haar_cascade = cv2.CascadeClassifier(path)
return _haar_cascade
def _get_yolo():
global _yolo_model
if _yolo_model is None:
from ultralytics import YOLO
_yolo_model = YOLO("yolov8n.pt")
return _yolo_model
def _largest_bbox(candidates):
"""candidates: list of (x1, y1, x2, y2). Returns the largest-area one."""
if not candidates:
return None
return max(candidates, key=lambda b: max(0, b[2] - b[0]) * max(0, b[3] - b[1]))
def _detect_face_mediapipe(image_bgr, min_conf):
detector = _get_mediapipe_detector(min_conf)
if detector is None:
return None
h, w = image_bgr.shape[:2]
rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB)
try:
results = detector.process(rgb)
except Exception:
return None
if not results.detections:
return None
boxes = []
for det in results.detections:
rb = det.location_data.relative_bounding_box
x1, y1 = rb.xmin * w, rb.ymin * h
x2, y2 = x1 + rb.width * w, y1 + rb.height * h
boxes.append((x1, y1, x2, y2))
return _largest_bbox(boxes)
def _detect_face_haar(image_bgr):
cascade = _get_haar_cascade()
gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY)
faces = cascade.detectMultiScale(gray, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40))
if len(faces) == 0:
return None
boxes = [(float(x), float(y), float(x + fw), float(y + fh)) for x, y, fw, fh in faces]
return _largest_bbox(boxes)
def _detect_person_yolo(image_bgr, min_conf):
try:
model = _get_yolo()
results = model.predict(image_bgr, classes=[0], conf=min_conf, verbose=False)
except Exception:
return None
boxes = []
for r in results:
if r.boxes is None:
continue
for b in r.boxes:
x1, y1, x2, y2 = b.xyxy[0].tolist()
boxes.append((x1, y1, x2, y2))
return _largest_bbox(boxes)
def detect_subject_bbox(image_bgr: np.ndarray, min_face_conf=0.5, min_person_conf=0.35):
"""Returns (bbox, kind) where kind is 'face', 'person', or 'full_frame'."""
h, w = image_bgr.shape[:2]
bbox = _detect_face_mediapipe(image_bgr, min_face_conf)
if bbox:
return bbox, "face"
bbox = _detect_face_haar(image_bgr)
if bbox:
return bbox, "face"
bbox = _detect_person_yolo(image_bgr, min_person_conf)
if bbox:
return bbox, "person"
return (0.0, 0.0, float(w), float(h)), "full_frame"
def subdivide_bbox(bbox, region, frac=0.45):
"""Narrows a bbox to its top or bottom portion (by height), same width.
Used to bias person-kind crops toward a specific body region (e.g. most
of the frame is a torso/lower-body shot rather than a full figure) without
needing a specialized body-part detector -- just a fraction of the
already-detected person bbox. `region` is "upper", "lower", or anything
else (returned unchanged, e.g. "auto"/"full").
"""
x1, y1, x2, y2 = bbox
h = y2 - y1
if region == "lower":
return (x1, y2 - h * frac, x2, y2)
if region == "upper":
return (x1, y1, x2, y1 + h * frac)
return bbox
|