File size: 4,353 Bytes
9c98083
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
"""Subject bbox detection for smart cropping.

Priority: mediapipe face detection -> OpenCV Haar cascade face detection
(no model download required, always available with opencv-python) ->
ultralytics YOLOv8n person detection -> full frame (no crop).

All bboxes are returned as (x1, y1, x2, y2) in pixel space, plus a "kind"
string ("face" | "person" | "full_frame") used by crop_resize.py to pick a
padding multiplier.
"""
import cv2
import numpy as np

_mp_detector = None
_mp_unavailable = False
_haar_cascade = None
_yolo_model = None


def _get_mediapipe_detector(min_conf):
    global _mp_detector, _mp_unavailable
    if _mp_unavailable:
        return None
    if _mp_detector is None:
        try:
            import mediapipe as mp
            _mp_detector = mp.solutions.face_detection.FaceDetection(
                model_selection=1, min_detection_confidence=min_conf
            )
        except Exception:
            _mp_unavailable = True
            return None
    return _mp_detector


def _get_haar_cascade():
    global _haar_cascade
    if _haar_cascade is None:
        path = cv2.data.haarcascades + "haarcascade_frontalface_default.xml"
        _haar_cascade = cv2.CascadeClassifier(path)
    return _haar_cascade


def _get_yolo():
    global _yolo_model
    if _yolo_model is None:
        from ultralytics import YOLO
        _yolo_model = YOLO("yolov8n.pt")
    return _yolo_model


def _largest_bbox(candidates):
    """candidates: list of (x1, y1, x2, y2). Returns the largest-area one."""
    if not candidates:
        return None
    return max(candidates, key=lambda b: max(0, b[2] - b[0]) * max(0, b[3] - b[1]))


def _detect_face_mediapipe(image_bgr, min_conf):
    detector = _get_mediapipe_detector(min_conf)
    if detector is None:
        return None
    h, w = image_bgr.shape[:2]
    rgb = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2RGB)
    try:
        results = detector.process(rgb)
    except Exception:
        return None
    if not results.detections:
        return None
    boxes = []
    for det in results.detections:
        rb = det.location_data.relative_bounding_box
        x1, y1 = rb.xmin * w, rb.ymin * h
        x2, y2 = x1 + rb.width * w, y1 + rb.height * h
        boxes.append((x1, y1, x2, y2))
    return _largest_bbox(boxes)


def _detect_face_haar(image_bgr):
    cascade = _get_haar_cascade()
    gray = cv2.cvtColor(image_bgr, cv2.COLOR_BGR2GRAY)
    faces = cascade.detectMultiScale(gray, scaleFactor=1.1, minNeighbors=5, minSize=(40, 40))
    if len(faces) == 0:
        return None
    boxes = [(float(x), float(y), float(x + fw), float(y + fh)) for x, y, fw, fh in faces]
    return _largest_bbox(boxes)


def _detect_person_yolo(image_bgr, min_conf):
    try:
        model = _get_yolo()
        results = model.predict(image_bgr, classes=[0], conf=min_conf, verbose=False)
    except Exception:
        return None
    boxes = []
    for r in results:
        if r.boxes is None:
            continue
        for b in r.boxes:
            x1, y1, x2, y2 = b.xyxy[0].tolist()
            boxes.append((x1, y1, x2, y2))
    return _largest_bbox(boxes)


def detect_subject_bbox(image_bgr: np.ndarray, min_face_conf=0.5, min_person_conf=0.35):
    """Returns (bbox, kind) where kind is 'face', 'person', or 'full_frame'."""
    h, w = image_bgr.shape[:2]

    bbox = _detect_face_mediapipe(image_bgr, min_face_conf)
    if bbox:
        return bbox, "face"

    bbox = _detect_face_haar(image_bgr)
    if bbox:
        return bbox, "face"

    bbox = _detect_person_yolo(image_bgr, min_person_conf)
    if bbox:
        return bbox, "person"

    return (0.0, 0.0, float(w), float(h)), "full_frame"


def subdivide_bbox(bbox, region, frac=0.45):
    """Narrows a bbox to its top or bottom portion (by height), same width.

    Used to bias person-kind crops toward a specific body region (e.g. most
    of the frame is a torso/lower-body shot rather than a full figure) without
    needing a specialized body-part detector -- just a fraction of the
    already-detected person bbox. `region` is "upper", "lower", or anything
    else (returned unchanged, e.g. "auto"/"full").
    """
    x1, y1, x2, y2 = bbox
    h = y2 - y1
    if region == "lower":
        return (x1, y2 - h * frac, x2, y2)
    if region == "upper":
        return (x1, y1, x2, y1 + h * frac)
    return bbox