bjooo's picture
Upload folder using huggingface_hub
9c98083 verified
Raw History Blame Contribute Delete
4.35 kB
"""Subject bbox detection for smart cropping.
Priority: mediapipe face detection -> OpenCV Haar cascade face detection
(no model download required, always available with opencv-python) ->
ultralytics YOLOv8n person detection -> full frame (no crop).
All bboxes are returned as (x1, y1, x2, y2) in pixel space, plus a "kind"
string ("face" | "person" | "full_frame") used by crop_resize.py to pick a
padding multiplier.
"""
import cv2
import numpy as np
_mp_detector = None
_mp_unavailable = False
_haar_cascade = None
_yolo_model = None
def _get_mediapipe_detector(min_conf):
global _mp_detector, _mp_unavailable
if _mp_unavailable:
return None
if _mp_detector is None:
try:
import mediapipe as mp
_mp_detector = mp.solutions.face_detection.FaceDetection(
model_selection=1, min_detection_confidence=min_conf
)
except Exception:
_mp_unavailable = True
return None
return _mp_detector
def _get_haar_cascade():
global _haar_cascade
if _haar_cascade is None:
path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml"
_haar_cascade = cv2.CascadeClassifier(path)
return _haar_cascade
def _get_yolo():
global _yolo_model
if _yolo_model is None:
from ultralytics import YOLO
_yolo_model = YOLO("yolov8n.pt")
return _yolo_model
def _largest_bbox(candidates):
"""candidates: list of (x1, y1, x2, y2). Returns the largest-area one."""
if not candidates:
return None
return max(candidates, key=lambda b: max(0, b[2] - b[0]) * max(0, b[3] - b[1]))
def _detect_face_mediapipe(image_bgr, min_conf):
detector = _get_mediapipe_detector(min_conf)
if detector is None:
return None
h, w = image_bgr.shape[:2]
rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB)
try:
results = detector.process(rgb)
except Exception:
return None
if not results.detections:
return None
boxes = []
for det in results.detections:
rb = det.location_data.relative_bounding_box
x1, y1 = rb.xmin * w, rb.ymin * h
x2, y2 = x1 + rb.width * w, y1 + rb.height * h
boxes.append((x1, y1, x2, y2))
return _largest_bbox(boxes)
def _detect_face_haar(image_bgr):
cascade = _get_haar_cascade()
gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY)
faces = cascade.detectMultiScale(gray, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40))
if len(faces) == 0:
return None
boxes = [(float(x), float(y), float(x + fw), float(y + fh)) for x, y, fw, fh in faces]
return _largest_bbox(boxes)
def _detect_person_yolo(image_bgr, min_conf):
try:
model = _get_yolo()
results = model.predict(image_bgr, classes=[0], conf=min_conf, verbose=False)
except Exception:
return None
boxes = []
for r in results:
if r.boxes is None:
continue
for b in r.boxes:
x1, y1, x2, y2 = b.xyxy[0].tolist()
boxes.append((x1, y1, x2, y2))
return _largest_bbox(boxes)
def detect_subject_bbox(image_bgr: np.ndarray, min_face_conf=0.5, min_person_conf=0.35):
"""Returns (bbox, kind) where kind is 'face', 'person', or 'full_frame'."""
h, w = image_bgr.shape[:2]
bbox = _detect_face_mediapipe(image_bgr, min_face_conf)
if bbox:
return bbox, "face"
bbox = _detect_face_haar(image_bgr)
if bbox:
return bbox, "face"
bbox = _detect_person_yolo(image_bgr, min_person_conf)
if bbox:
return bbox, "person"
return (0.0, 0.0, float(w), float(h)), "full_frame"
def subdivide_bbox(bbox, region, frac=0.45):
"""Narrows a bbox to its top or bottom portion (by height), same width.
Used to bias person-kind crops toward a specific body region (e.g. most
of the frame is a torso/lower-body shot rather than a full figure) without
needing a specialized body-part detector -- just a fraction of the
already-detected person bbox. `region` is "upper", "lower", or anything
else (returned unchanged, e.g. "auto"/"full").
"""
x1, y1, x2, y2 = bbox
h = y2 - y1
if region == "lower":
return (x1, y2 - h * frac, x2, y2)
if region == "upper":
return (x1, y1, x2, y1 + h * frac)
return bbox