Keypoint Detection
ultralytics
ONNX
TensorRT
human pose estimation
pose-estimation
yolo26
yolo26x-pose
human-pose
Instructions to use select-ai/pose-detection with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- ultralytics
How to use select-ai/pose-detection with ultralytics:
# Couldn't find a valid YOLO version tag. # Replace XX with the correct version. from ultralytics import YOLOvXX model = YOLOvXX.from_pretrained("select-ai/pose-detection") source = 'http://images.cocodataset.org/val2017/000000039769.jpg' model.predict(source=source, save=True) - TensorRT
How to use select-ai/pose-detection with TensorRT:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
Download scripts/run.py from select-ai/pose-detection: direct link, hf CLI and curl.
- Browser
- Download file 13.4 kB
-
https://huggingface.co/select-ai/pose-detection/resolve/main/scripts/run.py
- Command line
-
hf download hf://select-ai/pose-detection/scripts/run.py
-
curl -L -o run.py https://huggingface.co/select-ai/pose-detection/resolve/main/scripts/run.py
13.4 kB
| #!/usr/bin/env python3 | |
| """ | |
| Run YOLO26x-Pose human pose estimation on one image. | |
| The script loads the pretrained YOLO26x-Pose checkpoint, performs | |
| human pose estimation, and returns detected person bounding boxes | |
| (x1, y1, x2, y2), 17 human keypoints, confidence scores, and a | |
| 51-feature vector per person (17 keypoints x 3). Downstream | |
| consumers assemble box_xyxy + feature_vector into a pandas | |
| DataFrame. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import json | |
| from pathlib import Path | |
| from typing import Any | |
| import cv2 | |
| import numpy as np | |
| from ultralytics import YOLO | |
| REPO_ROOT = Path(__file__).resolve().parents[1] | |
| DEFAULT_MODEL = ( | |
| REPO_ROOT | |
| / "models" | |
| / "yolo26x-pose.pt" | |
| ) | |
| DEFAULT_IMAGE_SIZE = 960 | |
| DEFAULT_CONFIDENCE = 0.25 | |
| DEFAULT_IOU = 0.50 | |
| DEFAULT_DEVICE = 0 | |
| IMAGE_SUFFIXES = { | |
| ".jpg", | |
| ".jpeg", | |
| ".png", | |
| ".bmp", | |
| ".webp", | |
| } | |
| KEYPOINT_NAMES = [ | |
| "nose", | |
| "left_eye", | |
| "right_eye", | |
| "left_ear", | |
| "right_ear", | |
| "left_shoulder", | |
| "right_shoulder", | |
| "left_elbow", | |
| "right_elbow", | |
| "left_wrist", | |
| "right_wrist", | |
| "left_hip", | |
| "right_hip", | |
| "left_knee", | |
| "right_knee", | |
| "left_ankle", | |
| "right_ankle", | |
| ] | |
| NUM_KEYPOINTS = 17 | |
| KEYPOINT_FEATURE_COUNT = NUM_KEYPOINTS * 3 | |
| TOTAL_FEATURE_COUNT = KEYPOINT_FEATURE_COUNT | |
| BBOX_FEATURE_COUNT = 4 | |
| BBOX_FIELDS = ["x1", "y1", "x2", "y2"] | |
| def load_model(model_path: Path) -> YOLO: | |
| """ | |
| Load and validate the YOLO26x-Pose checkpoint. | |
| """ | |
| if not model_path.exists(): | |
| raise FileNotFoundError( | |
| f"YOLO26x-Pose checkpoint not found: {model_path}" | |
| ) | |
| model = YOLO(str(model_path)) | |
| if model.task != "pose": | |
| raise RuntimeError( | |
| f"Expected a pose model, but loaded task={model.task!r}" | |
| ) | |
| if model.names.get(0) != "person": | |
| raise RuntimeError( | |
| f"Expected class 0 to be 'person', " | |
| f"but found {model.names}" | |
| ) | |
| return model | |
| def validate_image(image_path: Path) -> np.ndarray: | |
| """ | |
| Load one input image and validate that it can be decoded. | |
| """ | |
| if not image_path.exists(): | |
| raise FileNotFoundError( | |
| f"Input image not found: {image_path}" | |
| ) | |
| if image_path.suffix.lower() not in IMAGE_SUFFIXES: | |
| raise ValueError( | |
| f"Unsupported image format: {image_path.suffix}" | |
| ) | |
| image = cv2.imread( | |
| str(image_path), | |
| cv2.IMREAD_COLOR, | |
| ) | |
| if image is None: | |
| raise ValueError( | |
| f"Could not decode image: {image_path}" | |
| ) | |
| if image.size == 0: | |
| raise ValueError( | |
| f"Input image is empty: {image_path}" | |
| ) | |
| return image | |
| def _keypoint_records( | |
| keypoints_xy: np.ndarray, | |
| keypoints_conf: np.ndarray | None, | |
| ) -> list[dict[str, Any]]: | |
| """ | |
| Convert one person's 17 keypoints into JSON-friendly records. | |
| """ | |
| records: list[dict[str, Any]] = [] | |
| for index, point in enumerate(keypoints_xy): | |
| x = float(point[0]) | |
| y = float(point[1]) | |
| record: dict[str, Any] = { | |
| "index": index, | |
| "name": KEYPOINT_NAMES[index], | |
| "x": round(x, 3), | |
| "y": round(y, 3), | |
| } | |
| if keypoints_conf is not None: | |
| record["confidence"] = round( | |
| float(keypoints_conf[index]), | |
| 6, | |
| ) | |
| else: | |
| record["confidence"] = None | |
| records.append(record) | |
| return records | |
| def feature_column_names() -> list[str]: | |
| """ | |
| Column names for the 51 keypoint features (COCO order). | |
| """ | |
| columns: list[str] = [] | |
| for name in KEYPOINT_NAMES: | |
| columns.extend([f"{name}_x", f"{name}_y", f"{name}_conf"]) | |
| return columns | |
| def build_feature_vector( | |
| keypoints_xy: np.ndarray, | |
| keypoints_conf: np.ndarray | None, | |
| ) -> list[float]: | |
| """ | |
| Build the 51-feature vector for one detected person. | |
| Layout: | |
| indices 0-50 : 17 keypoints x (x, y, confidence) | |
| """ | |
| features: list[float] = [] | |
| for index in range(NUM_KEYPOINTS): | |
| features.append(round(float(keypoints_xy[index][0]), 3)) | |
| features.append(round(float(keypoints_xy[index][1]), 3)) | |
| if keypoints_conf is not None: | |
| features.append( | |
| round(float(keypoints_conf[index]), 6) | |
| ) | |
| else: | |
| features.append(0.0) | |
| if len(features) != TOTAL_FEATURE_COUNT: | |
| raise RuntimeError( | |
| f"Expected {TOTAL_FEATURE_COUNT} features, " | |
| f"got {len(features)}." | |
| ) | |
| return features | |
| def build_feature_frame( | |
| predictions: list[dict[str, Any]], | |
| ) -> "Any": | |
| """ | |
| Assemble person predictions into a pandas DataFrame. | |
| Columns: x1, y1, x2, y2 + 51 keypoint features. | |
| """ | |
| import pandas as pd | |
| columns = BBOX_FIELDS + feature_column_names() | |
| rows: list[list[float]] = [] | |
| for person in predictions: | |
| box = [float(v) for v in person["box_xyxy"]] | |
| vector = [float(v) for v in person["feature_vector"]] | |
| rows.append(box + vector) | |
| return pd.DataFrame(rows, columns=columns) | |
| def extract_predictions( | |
| result: Any, | |
| ) -> list[dict[str, Any]]: | |
| """ | |
| Convert an Ultralytics pose result into a JSON-friendly structure. | |
| """ | |
| predictions: list[dict[str, Any]] = [] | |
| if result.boxes is None: | |
| return predictions | |
| if len(result.boxes) == 0: | |
| return predictions | |
| boxes = ( | |
| result.boxes.xyxy | |
| .detach() | |
| .cpu() | |
| .numpy() | |
| .astype(np.float32) | |
| ) | |
| scores = ( | |
| result.boxes.conf | |
| .detach() | |
| .cpu() | |
| .numpy() | |
| .astype(np.float32) | |
| ) | |
| classes = ( | |
| result.boxes.cls | |
| .detach() | |
| .cpu() | |
| .numpy() | |
| .astype(np.int32) | |
| ) | |
| keypoints_xy = None | |
| keypoints_conf = None | |
| if result.keypoints is not None: | |
| if len(result.keypoints): | |
| keypoints_xy = ( | |
| result.keypoints.xy | |
| .detach() | |
| .cpu() | |
| .numpy() | |
| .astype(np.float32) | |
| ) | |
| if result.keypoints.conf is not None: | |
| keypoints_conf = ( | |
| result.keypoints.conf | |
| .detach() | |
| .cpu() | |
| .numpy() | |
| .astype(np.float32) | |
| ) | |
| for index in range(len(boxes)): | |
| class_id = int(classes[index]) | |
| box = boxes[index] | |
| x1 = float(box[0]) | |
| y1 = float(box[1]) | |
| x2 = float(box[2]) | |
| y2 = float(box[3]) | |
| person: dict[str, Any] = { | |
| "class_id": class_id, | |
| "class_name": "person", | |
| "confidence": round( | |
| float(scores[index]), | |
| 6, | |
| ), | |
| "box_xyxy": [ | |
| round(x1, 3), | |
| round(y1, 3), | |
| round(x2, 3), | |
| round(y2, 3), | |
| ], | |
| "keypoints": [], | |
| "feature_count": TOTAL_FEATURE_COUNT, | |
| "feature_vector": [], | |
| } | |
| if ( | |
| keypoints_xy is not None | |
| and index < len(keypoints_xy) | |
| ): | |
| confidence = None | |
| if ( | |
| keypoints_conf is not None | |
| and index < len(keypoints_conf) | |
| ): | |
| confidence = keypoints_conf[index] | |
| person["keypoints"] = _keypoint_records( | |
| keypoints_xy[index], | |
| confidence, | |
| ) | |
| person["feature_vector"] = build_feature_vector( | |
| keypoints_xy=keypoints_xy[index], | |
| keypoints_conf=confidence, | |
| ) | |
| predictions.append(person) | |
| return predictions | |
| def run_pose( | |
| model: YOLO, | |
| image: np.ndarray, | |
| image_size: int, | |
| confidence: float, | |
| iou: float, | |
| device: Any, | |
| max_det: int, | |
| ) -> tuple[Any, float]: | |
| """ | |
| Run YOLO26x-Pose inference and return the result and elapsed time. | |
| """ | |
| import time | |
| started = time.perf_counter() | |
| results = model.predict( | |
| source=image, | |
| imgsz=image_size, | |
| conf=confidence, | |
| iou=iou, | |
| device=device, | |
| max_det=max_det, | |
| verbose=False, | |
| save=False, | |
| ) | |
| elapsed = time.perf_counter() - started | |
| if not results: | |
| raise RuntimeError( | |
| "YOLO26x-Pose returned no inference result." | |
| ) | |
| return results[0], elapsed | |
| def save_annotated_result( | |
| result: Any, | |
| output_path: Path, | |
| ) -> None: | |
| """ | |
| Save the Ultralytics annotated pose visualization. | |
| """ | |
| output_path.parent.mkdir( | |
| parents=True, | |
| exist_ok=True, | |
| ) | |
| annotated = result.plot() | |
| if annotated is None: | |
| raise RuntimeError( | |
| "Could not generate annotated pose output." | |
| ) | |
| success = cv2.imwrite( | |
| str(output_path), | |
| annotated, | |
| ) | |
| if not success: | |
| raise RuntimeError( | |
| f"Could not write output image: {output_path}" | |
| ) | |
| def build_payload( | |
| image_path: Path, | |
| image: np.ndarray, | |
| model_path: Path, | |
| result: Any, | |
| elapsed: float, | |
| image_size: int, | |
| confidence: float, | |
| iou: float, | |
| device: Any, | |
| output_path: Path | None, | |
| ) -> dict[str, Any]: | |
| """ | |
| Build the JSON response for one pose inference. | |
| """ | |
| predictions = extract_predictions(result) | |
| height, width = image.shape[:2] | |
| return { | |
| "model": "YOLO26x-Pose", | |
| "task": "human pose estimation", | |
| "checkpoint": str(model_path), | |
| "image": str(image_path), | |
| "image_shape": [ | |
| int(width), | |
| int(height), | |
| ], | |
| "input_size": [ | |
| int(image_size), | |
| int(image_size), | |
| ], | |
| "device": str(device), | |
| "confidence_threshold": float(confidence), | |
| "iou_threshold": float(iou), | |
| "class": { | |
| "id": 0, | |
| "name": "person", | |
| }, | |
| "keypoint_count": NUM_KEYPOINTS, | |
| "keypoint_names": KEYPOINT_NAMES, | |
| "feature_count": TOTAL_FEATURE_COUNT, | |
| "keypoint_feature_count": KEYPOINT_FEATURE_COUNT, | |
| "bbox_fields": BBOX_FIELDS, | |
| "bbox_feature_count": BBOX_FEATURE_COUNT, | |
| "person_count": len(predictions), | |
| "inference_seconds": round( | |
| float(elapsed), | |
| 6, | |
| ), | |
| "inference_ms": round( | |
| float(elapsed * 1000.0), | |
| 3, | |
| ), | |
| "predictions": predictions, | |
| "annotated_output": ( | |
| str(output_path) | |
| if output_path is not None | |
| else None | |
| ), | |
| } | |
| def main() -> int: | |
| parser = argparse.ArgumentParser( | |
| description=__doc__ | |
| ) | |
| parser.add_argument( | |
| "--image", | |
| required=True, | |
| type=Path, | |
| help="Input image for pose estimation.", | |
| ) | |
| parser.add_argument( | |
| "--model", | |
| type=Path, | |
| default=DEFAULT_MODEL, | |
| help="Path to the YOLO26x-Pose checkpoint.", | |
| ) | |
| parser.add_argument( | |
| "--imgsz", | |
| type=int, | |
| default=DEFAULT_IMAGE_SIZE, | |
| help="Inference image size.", | |
| ) | |
| parser.add_argument( | |
| "--conf", | |
| type=float, | |
| default=DEFAULT_CONFIDENCE, | |
| help="Person detection confidence threshold.", | |
| ) | |
| parser.add_argument( | |
| "--iou", | |
| type=float, | |
| default=DEFAULT_IOU, | |
| help="NMS IoU threshold.", | |
| ) | |
| parser.add_argument( | |
| "--device", | |
| default=DEFAULT_DEVICE, | |
| help="Inference device, e.g. 0, 1, cpu.", | |
| ) | |
| parser.add_argument( | |
| "--max-det", | |
| type=int, | |
| default=100, | |
| help="Maximum number of detections.", | |
| ) | |
| parser.add_argument( | |
| "--output", | |
| type=Path, | |
| help="Optional path for the annotated pose image.", | |
| ) | |
| args = parser.parse_args() | |
| if args.imgsz <= 0: | |
| parser.error("--imgsz must be greater than zero.") | |
| if not 0.0 <= args.conf <= 1.0: | |
| parser.error("--conf must be between 0 and 1.") | |
| if not 0.0 <= args.iou <= 1.0: | |
| parser.error("--iou must be between 0 and 1.") | |
| if args.max_det <= 0: | |
| parser.error("--max-det must be greater than zero.") | |
| image = validate_image( | |
| args.image | |
| ) | |
| model = load_model( | |
| args.model | |
| ) | |
| result, elapsed = run_pose( | |
| model=model, | |
| image=image, | |
| image_size=args.imgsz, | |
| confidence=args.conf, | |
| iou=args.iou, | |
| device=args.device, | |
| max_det=args.max_det, | |
| ) | |
| output_path = args.output | |
| if output_path is not None: | |
| save_annotated_result( | |
| result=result, | |
| output_path=output_path, | |
| ) | |
| payload = build_payload( | |
| image_path=args.image, | |
| image=image, | |
| model_path=args.model, | |
| result=result, | |
| elapsed=elapsed, | |
| image_size=args.imgsz, | |
| confidence=args.conf, | |
| iou=args.iou, | |
| device=args.device, | |
| output_path=output_path, | |
| ) | |
| print( | |
| json.dumps( | |
| payload, | |
| indent=2, | |
| ) | |
| ) | |
| return 0 | |
| if __name__ == "__main__": | |
| raise SystemExit( | |
| main() | |
| ) |