#!/usr/bin/env python3 """ Run YOLO26x-Pose human pose estimation on one image. The script loads the pretrained YOLO26x-Pose checkpoint, performs human pose estimation, and returns detected person bounding boxes (x1, y1, x2, y2), 17 human keypoints, confidence scores, and a 51-feature vector per person (17 keypoints x 3). Downstream consumers assemble box_xyxy + feature_vector into a pandas DataFrame. """ from __future__ import annotations import argparse import json from pathlib import Path from typing import Any import cv2 import numpy as np from ultralytics import YOLO REPO_ROOT = Path(__file__).resolve().parents[1] DEFAULT_MODEL = ( REPO_ROOT / "models" / "yolo26x-pose.pt" ) DEFAULT_IMAGE_SIZE = 960 DEFAULT_CONFIDENCE = 0.25 DEFAULT_IOU = 0.50 DEFAULT_DEVICE = 0 IMAGE_SUFFIXES = { ".jpg", ".jpeg", ".png", ".bmp", ".webp", } KEYPOINT_NAMES = [ "nose", "left_eye", "right_eye", "left_ear", "right_ear", "left_shoulder", "right_shoulder", "left_elbow", "right_elbow", "left_wrist", "right_wrist", "left_hip", "right_hip", "left_knee", "right_knee", "left_ankle", "right_ankle", ] NUM_KEYPOINTS = 17 KEYPOINT_FEATURE_COUNT = NUM_KEYPOINTS * 3 TOTAL_FEATURE_COUNT = KEYPOINT_FEATURE_COUNT BBOX_FEATURE_COUNT = 4 BBOX_FIELDS = ["x1", "y1", "x2", "y2"] def load_model(model_path: Path) -> YOLO: """ Load and validate the YOLO26x-Pose checkpoint. """ if not model_path.exists(): raise FileNotFoundError( f"YOLO26x-Pose checkpoint not found: {model_path}" ) model = YOLO(str(model_path)) if model.task != "pose": raise RuntimeError( f"Expected a pose model, but loaded task={model.task!r}" ) if model.names.get(0) != "person": raise RuntimeError( f"Expected class 0 to be 'person', " f"but found {model.names}" ) return model def validate_image(image_path: Path) -> np.ndarray: """ Load one input image and validate that it can be decoded. """ if not image_path.exists(): raise FileNotFoundError( f"Input image not found: {image_path}" ) if image_path.suffix.lower() not in IMAGE_SUFFIXES: raise ValueError( f"Unsupported image format: {image_path.suffix}" ) image = cv2.imread( str(image_path), cv2.IMREAD_COLOR, ) if image is None: raise ValueError( f"Could not decode image: {image_path}" ) if image.size == 0: raise ValueError( f"Input image is empty: {image_path}" ) return image def _keypoint_records( keypoints_xy: np.ndarray, keypoints_conf: np.ndarray | None, ) -> list[dict[str, Any]]: """ Convert one person's 17 keypoints into JSON-friendly records. """ records: list[dict[str, Any]] = [] for index, point in enumerate(keypoints_xy): x = float(point[0]) y = float(point[1]) record: dict[str, Any] = { "index": index, "name": KEYPOINT_NAMES[index], "x": round(x, 3), "y": round(y, 3), } if keypoints_conf is not None: record["confidence"] = round( float(keypoints_conf[index]), 6, ) else: record["confidence"] = None records.append(record) return records def feature_column_names() -> list[str]: """ Column names for the 51 keypoint features (COCO order). """ columns: list[str] = [] for name in KEYPOINT_NAMES: columns.extend([f"{name}_x", f"{name}_y", f"{name}_conf"]) return columns def build_feature_vector( keypoints_xy: np.ndarray, keypoints_conf: np.ndarray | None, ) -> list[float]: """ Build the 51-feature vector for one detected person. Layout: indices 0-50 : 17 keypoints x (x, y, confidence) """ features: list[float] = [] for index in range(NUM_KEYPOINTS): features.append(round(float(keypoints_xy[index][0]), 3)) features.append(round(float(keypoints_xy[index][1]), 3)) if keypoints_conf is not None: features.append( round(float(keypoints_conf[index]), 6) ) else: features.append(0.0) if len(features) != TOTAL_FEATURE_COUNT: raise RuntimeError( f"Expected {TOTAL_FEATURE_COUNT} features, " f"got {len(features)}." ) return features def build_feature_frame( predictions: list[dict[str, Any]], ) -> "Any": """ Assemble person predictions into a pandas DataFrame. Columns: x1, y1, x2, y2 + 51 keypoint features. """ import pandas as pd columns = BBOX_FIELDS + feature_column_names() rows: list[list[float]] = [] for person in predictions: box = [float(v) for v in person["box_xyxy"]] vector = [float(v) for v in person["feature_vector"]] rows.append(box + vector) return pd.DataFrame(rows, columns=columns) def extract_predictions( result: Any, ) -> list[dict[str, Any]]: """ Convert an Ultralytics pose result into a JSON-friendly structure. """ predictions: list[dict[str, Any]] = [] if result.boxes is None: return predictions if len(result.boxes) == 0: return predictions boxes = ( result.boxes.xyxy .detach() .cpu() .numpy() .astype(np.float32) ) scores = ( result.boxes.conf .detach() .cpu() .numpy() .astype(np.float32) ) classes = ( result.boxes.cls .detach() .cpu() .numpy() .astype(np.int32) ) keypoints_xy = None keypoints_conf = None if result.keypoints is not None: if len(result.keypoints): keypoints_xy = ( result.keypoints.xy .detach() .cpu() .numpy() .astype(np.float32) ) if result.keypoints.conf is not None: keypoints_conf = ( result.keypoints.conf .detach() .cpu() .numpy() .astype(np.float32) ) for index in range(len(boxes)): class_id = int(classes[index]) box = boxes[index] x1 = float(box[0]) y1 = float(box[1]) x2 = float(box[2]) y2 = float(box[3]) person: dict[str, Any] = { "class_id": class_id, "class_name": "person", "confidence": round( float(scores[index]), 6, ), "box_xyxy": [ round(x1, 3), round(y1, 3), round(x2, 3), round(y2, 3), ], "keypoints": [], "feature_count": TOTAL_FEATURE_COUNT, "feature_vector": [], } if ( keypoints_xy is not None and index < len(keypoints_xy) ): confidence = None if ( keypoints_conf is not None and index < len(keypoints_conf) ): confidence = keypoints_conf[index] person["keypoints"] = _keypoint_records( keypoints_xy[index], confidence, ) person["feature_vector"] = build_feature_vector( keypoints_xy=keypoints_xy[index], keypoints_conf=confidence, ) predictions.append(person) return predictions def run_pose( model: YOLO, image: np.ndarray, image_size: int, confidence: float, iou: float, device: Any, max_det: int, ) -> tuple[Any, float]: """ Run YOLO26x-Pose inference and return the result and elapsed time. """ import time started = time.perf_counter() results = model.predict( source=image, imgsz=image_size, conf=confidence, iou=iou, device=device, max_det=max_det, verbose=False, save=False, ) elapsed = time.perf_counter() - started if not results: raise RuntimeError( "YOLO26x-Pose returned no inference result." ) return results[0], elapsed def save_annotated_result( result: Any, output_path: Path, ) -> None: """ Save the Ultralytics annotated pose visualization. """ output_path.parent.mkdir( parents=True, exist_ok=True, ) annotated = result.plot() if annotated is None: raise RuntimeError( "Could not generate annotated pose output." ) success = cv2.imwrite( str(output_path), annotated, ) if not success: raise RuntimeError( f"Could not write output image: {output_path}" ) def build_payload( image_path: Path, image: np.ndarray, model_path: Path, result: Any, elapsed: float, image_size: int, confidence: float, iou: float, device: Any, output_path: Path | None, ) -> dict[str, Any]: """ Build the JSON response for one pose inference. """ predictions = extract_predictions(result) height, width = image.shape[:2] return { "model": "YOLO26x-Pose", "task": "human pose estimation", "checkpoint": str(model_path), "image": str(image_path), "image_shape": [ int(width), int(height), ], "input_size": [ int(image_size), int(image_size), ], "device": str(device), "confidence_threshold": float(confidence), "iou_threshold": float(iou), "class": { "id": 0, "name": "person", }, "keypoint_count": NUM_KEYPOINTS, "keypoint_names": KEYPOINT_NAMES, "feature_count": TOTAL_FEATURE_COUNT, "keypoint_feature_count": KEYPOINT_FEATURE_COUNT, "bbox_fields": BBOX_FIELDS, "bbox_feature_count": BBOX_FEATURE_COUNT, "person_count": len(predictions), "inference_seconds": round( float(elapsed), 6, ), "inference_ms": round( float(elapsed * 1000.0), 3, ), "predictions": predictions, "annotated_output": ( str(output_path) if output_path is not None else None ), } def main() -> int: parser = argparse.ArgumentParser( description=__doc__ ) parser.add_argument( "--image", required=True, type=Path, help="Input image for pose estimation.", ) parser.add_argument( "--model", type=Path, default=DEFAULT_MODEL, help="Path to the YOLO26x-Pose checkpoint.", ) parser.add_argument( "--imgsz", type=int, default=DEFAULT_IMAGE_SIZE, help="Inference image size.", ) parser.add_argument( "--conf", type=float, default=DEFAULT_CONFIDENCE, help="Person detection confidence threshold.", ) parser.add_argument( "--iou", type=float, default=DEFAULT_IOU, help="NMS IoU threshold.", ) parser.add_argument( "--device", default=DEFAULT_DEVICE, help="Inference device, e.g. 0, 1, cpu.", ) parser.add_argument( "--max-det", type=int, default=100, help="Maximum number of detections.", ) parser.add_argument( "--output", type=Path, help="Optional path for the annotated pose image.", ) args = parser.parse_args() if args.imgsz <= 0: parser.error("--imgsz must be greater than zero.") if not 0.0 <= args.conf <= 1.0: parser.error("--conf must be between 0 and 1.") if not 0.0 <= args.iou <= 1.0: parser.error("--iou must be between 0 and 1.") if args.max_det <= 0: parser.error("--max-det must be greater than zero.") image = validate_image( args.image ) model = load_model( args.model ) result, elapsed = run_pose( model=model, image=image, image_size=args.imgsz, confidence=args.conf, iou=args.iou, device=args.device, max_det=args.max_det, ) output_path = args.output if output_path is not None: save_annotated_result( result=result, output_path=output_path, ) payload = build_payload( image_path=args.image, image=image, model_path=args.model, result=result, elapsed=elapsed, image_size=args.imgsz, confidence=args.conf, iou=args.iou, device=args.device, output_path=output_path, ) print( json.dumps( payload, indent=2, ) ) return 0 if __name__ == "__main__": raise SystemExit( main() )