"""Subject bbox detection for smart cropping. Priority: mediapipe face detection -> OpenCV Haar cascade face detection (no model download required, always available with opencv-python) -> ultralytics YOLOv8n person detection -> full frame (no crop). All bboxes are returned as (x1, y1, x2, y2) in pixel space, plus a "kind" string ("face" | "person" | "full_frame") used by crop_resize.py to pick a padding multiplier. """ import cv2 import numpy as np _mp_detector = None _mp_unavailable = False _haar_cascade = None _yolo_model = None def _get_mediapipe_detector(min_conf): global _mp_detector, _mp_unavailable if _mp_unavailable: return None if _mp_detector is None: try: import mediapipe as mp _mp_detector = mp.solutions.face_detection.FaceDetection( model_selection=1, min_detection_confidence=min_conf ) except Exception: _mp_unavailable = True return None return _mp_detector def _get_haar_cascade(): global _haar_cascade if _haar_cascade is None: path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml" _haar_cascade = cv2.CascadeClassifier(path) return _haar_cascade def _get_yolo(): global _yolo_model if _yolo_model is None: from ultralytics import YOLO _yolo_model = YOLO("yolov8n.pt") return _yolo_model def _largest_bbox(candidates): """candidates: list of (x1, y1, x2, y2). Returns the largest-area one.""" if not candidates: return None return max(candidates, key=lambda b: max(0, b[2] - b[0]) * max(0, b[3] - b[1])) def _detect_face_mediapipe(image_bgr, min_conf): detector = _get_mediapipe_detector(min_conf) if detector is None: return None h, w = image_bgr.shape[:2] rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB) try: results = detector.process(rgb) except Exception: return None if not results.detections: return None boxes = [] for det in results.detections: rb = det.location_data.relative_bounding_box x1, y1 = rb.xmin * w, rb.ymin * h x2, y2 = x1 + rb.width * w, y1 + rb.height * h boxes.append((x1, y1, x2, y2)) return _largest_bbox(boxes) def _detect_face_haar(image_bgr): cascade = _get_haar_cascade() gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY) faces = cascade.detectMultiScale(gray, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40)) if len(faces) == 0: return None boxes = [(float(x), float(y), float(x + fw), float(y + fh)) for x, y, fw, fh in faces] return _largest_bbox(boxes) def _detect_person_yolo(image_bgr, min_conf): try: model = _get_yolo() results = model.predict(image_bgr, classes=[0], conf=min_conf, verbose=False) except Exception: return None boxes = [] for r in results: if r.boxes is None: continue for b in r.boxes: x1, y1, x2, y2 = b.xyxy[0].tolist() boxes.append((x1, y1, x2, y2)) return _largest_bbox(boxes) def detect_subject_bbox(image_bgr: np.ndarray, min_face_conf=0.5, min_person_conf=0.35): """Returns (bbox, kind) where kind is 'face', 'person', or 'full_frame'.""" h, w = image_bgr.shape[:2] bbox = _detect_face_mediapipe(image_bgr, min_face_conf) if bbox: return bbox, "face" bbox = _detect_face_haar(image_bgr) if bbox: return bbox, "face" bbox = _detect_person_yolo(image_bgr, min_person_conf) if bbox: return bbox, "person" return (0.0, 0.0, float(w), float(h)), "full_frame" def subdivide_bbox(bbox, region, frac=0.45): """Narrows a bbox to its top or bottom portion (by height), same width. Used to bias person-kind crops toward a specific body region (e.g. most of the frame is a torso/lower-body shot rather than a full figure) without needing a specialized body-part detector -- just a fraction of the already-detected person bbox. `region` is "upper", "lower", or anything else (returned unchanged, e.g. "auto"/"full"). """ x1, y1, x2, y2 = bbox h = y2 - y1 if region == "lower": return (x1, y2 - h * frac, x2, y2) if region == "upper": return (x1, y1, x2, y1 + h * frac) return bbox