nishant2401's picture
Upload folder using huggingface_hub
9da727c verified
Raw History Blame Contribute Delete
13.4 kB
#!/usr/bin/env python3
"""
Run YOLO26x-Pose human pose estimation on one image.
The script loads the pretrained YOLO26x-Pose checkpoint, performs
human pose estimation, and returns detected person bounding boxes
(x1, y1, x2, y2), 17 human keypoints, confidence scores, and a
51-feature vector per person (17 keypoints x 3). Downstream
consumers assemble box_xyxy + feature_vector into a pandas
DataFrame.
"""
from __future__ import annotations
import argparse
import json
from pathlib import Path
from typing import Any
import cv2
import numpy as np
from ultralytics import YOLO
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_MODEL = (
REPO_ROOT
/ "models"
/ "yolo26x-pose.pt"
)
DEFAULT_IMAGE_SIZE = 960
DEFAULT_CONFIDENCE = 0.25
DEFAULT_IOU = 0.50
DEFAULT_DEVICE = 0
IMAGE_SUFFIXES = {
".jpg",
".jpeg",
".png",
".bmp",
".webp",
}
KEYPOINT_NAMES = [
"nose",
"left_eye",
"right_eye",
"left_ear",
"right_ear",
"left_shoulder",
"right_shoulder",
"left_elbow",
"right_elbow",
"left_wrist",
"right_wrist",
"left_hip",
"right_hip",
"left_knee",
"right_knee",
"left_ankle",
"right_ankle",
]
NUM_KEYPOINTS = 17
KEYPOINT_FEATURE_COUNT = NUM_KEYPOINTS * 3
TOTAL_FEATURE_COUNT = KEYPOINT_FEATURE_COUNT
BBOX_FEATURE_COUNT = 4
BBOX_FIELDS = ["x1", "y1", "x2", "y2"]
def load_model(model_path: Path) -> YOLO:
"""
Load and validate the YOLO26x-Pose checkpoint.
"""
if not model_path.exists():
raise FileNotFoundError(
f"YOLO26x-Pose checkpoint not found: {model_path}"
)
model = YOLO(str(model_path))
if model.task != "pose":
raise RuntimeError(
f"Expected a pose model, but loaded task={model.task!r}"
)
if model.names.get(0) != "person":
raise RuntimeError(
f"Expected class 0 to be 'person', "
f"but found {model.names}"
)
return model
def validate_image(image_path: Path) -> np.ndarray:
"""
Load one input image and validate that it can be decoded.
"""
if not image_path.exists():
raise FileNotFoundError(
f"Input image not found: {image_path}"
)
if image_path.suffix.lower() not in IMAGE_SUFFIXES:
raise ValueError(
f"Unsupported image format: {image_path.suffix}"
)
image = cv2.imread(
str(image_path),
cv2.IMREAD_COLOR,
)
if image is None:
raise ValueError(
f"Could not decode image: {image_path}"
)
if image.size == 0:
raise ValueError(
f"Input image is empty: {image_path}"
)
return image
def _keypoint_records(
keypoints_xy: np.ndarray,
keypoints_conf: np.ndarray | None,
) -> list[dict[str, Any]]:
"""
Convert one person's 17 keypoints into JSON-friendly records.
"""
records: list[dict[str, Any]] = []
for index, point in enumerate(keypoints_xy):
x = float(point[0])
y = float(point[1])
record: dict[str, Any] = {
"index": index,
"name": KEYPOINT_NAMES[index],
"x": round(x, 3),
"y": round(y, 3),
}
if keypoints_conf is not None:
record["confidence"] = round(
float(keypoints_conf[index]),
6,
)
else:
record["confidence"] = None
records.append(record)
return records
def feature_column_names() -> list[str]:
"""
Column names for the 51 keypoint features (COCO order).
"""
columns: list[str] = []
for name in KEYPOINT_NAMES:
columns.extend([f"{name}_x", f"{name}_y", f"{name}_conf"])
return columns
def build_feature_vector(
keypoints_xy: np.ndarray,
keypoints_conf: np.ndarray | None,
) -> list[float]:
"""
Build the 51-feature vector for one detected person.
Layout:
indices 0-50 : 17 keypoints x (x, y, confidence)
"""
features: list[float] = []
for index in range(NUM_KEYPOINTS):
features.append(round(float(keypoints_xy[index][0]), 3))
features.append(round(float(keypoints_xy[index][1]), 3))
if keypoints_conf is not None:
features.append(
round(float(keypoints_conf[index]), 6)
)
else:
features.append(0.0)
if len(features) != TOTAL_FEATURE_COUNT:
raise RuntimeError(
f"Expected {TOTAL_FEATURE_COUNT} features, "
f"got {len(features)}."
)
return features
def build_feature_frame(
predictions: list[dict[str, Any]],
) -> "Any":
"""
Assemble person predictions into a pandas DataFrame.
Columns: x1, y1, x2, y2 + 51 keypoint features.
"""
import pandas as pd
columns = BBOX_FIELDS + feature_column_names()
rows: list[list[float]] = []
for person in predictions:
box = [float(v) for v in person["box_xyxy"]]
vector = [float(v) for v in person["feature_vector"]]
rows.append(box + vector)
return pd.DataFrame(rows, columns=columns)
def extract_predictions(
result: Any,
) -> list[dict[str, Any]]:
"""
Convert an Ultralytics pose result into a JSON-friendly structure.
"""
predictions: list[dict[str, Any]] = []
if result.boxes is None:
return predictions
if len(result.boxes) == 0:
return predictions
boxes = (
result.boxes.xyxy
.detach()
.cpu()
.numpy()
.astype(np.float32)
)
scores = (
result.boxes.conf
.detach()
.cpu()
.numpy()
.astype(np.float32)
)
classes = (
result.boxes.cls
.detach()
.cpu()
.numpy()
.astype(np.int32)
)
keypoints_xy = None
keypoints_conf = None
if result.keypoints is not None:
if len(result.keypoints):
keypoints_xy = (
result.keypoints.xy
.detach()
.cpu()
.numpy()
.astype(np.float32)
)
if result.keypoints.conf is not None:
keypoints_conf = (
result.keypoints.conf
.detach()
.cpu()
.numpy()
.astype(np.float32)
)
for index in range(len(boxes)):
class_id = int(classes[index])
box = boxes[index]
x1 = float(box[0])
y1 = float(box[1])
x2 = float(box[2])
y2 = float(box[3])
person: dict[str, Any] = {
"class_id": class_id,
"class_name": "person",
"confidence": round(
float(scores[index]),
6,
),
"box_xyxy": [
round(x1, 3),
round(y1, 3),
round(x2, 3),
round(y2, 3),
],
"keypoints": [],
"feature_count": TOTAL_FEATURE_COUNT,
"feature_vector": [],
}
if (
keypoints_xy is not None
and index < len(keypoints_xy)
):
confidence = None
if (
keypoints_conf is not None
and index < len(keypoints_conf)
):
confidence = keypoints_conf[index]
person["keypoints"] = _keypoint_records(
keypoints_xy[index],
confidence,
)
person["feature_vector"] = build_feature_vector(
keypoints_xy=keypoints_xy[index],
keypoints_conf=confidence,
)
predictions.append(person)
return predictions
def run_pose(
model: YOLO,
image: np.ndarray,
image_size: int,
confidence: float,
iou: float,
device: Any,
max_det: int,
) -> tuple[Any, float]:
"""
Run YOLO26x-Pose inference and return the result and elapsed time.
"""
import time
started = time.perf_counter()
results = model.predict(
source=image,
imgsz=image_size,
conf=confidence,
iou=iou,
device=device,
max_det=max_det,
verbose=False,
save=False,
)
elapsed = time.perf_counter() - started
if not results:
raise RuntimeError(
"YOLO26x-Pose returned no inference result."
)
return results[0], elapsed
def save_annotated_result(
result: Any,
output_path: Path,
) -> None:
"""
Save the Ultralytics annotated pose visualization.
"""
output_path.parent.mkdir(
parents=True,
exist_ok=True,
)
annotated = result.plot()
if annotated is None:
raise RuntimeError(
"Could not generate annotated pose output."
)
success = cv2.imwrite(
str(output_path),
annotated,
)
if not success:
raise RuntimeError(
f"Could not write output image: {output_path}"
)
def build_payload(
image_path: Path,
image: np.ndarray,
model_path: Path,
result: Any,
elapsed: float,
image_size: int,
confidence: float,
iou: float,
device: Any,
output_path: Path | None,
) -> dict[str, Any]:
"""
Build the JSON response for one pose inference.
"""
predictions = extract_predictions(result)
height, width = image.shape[:2]
return {
"model": "YOLO26x-Pose",
"task": "human pose estimation",
"checkpoint": str(model_path),
"image": str(image_path),
"image_shape": [
int(width),
int(height),
],
"input_size": [
int(image_size),
int(image_size),
],
"device": str(device),
"confidence_threshold": float(confidence),
"iou_threshold": float(iou),
"class": {
"id": 0,
"name": "person",
},
"keypoint_count": NUM_KEYPOINTS,
"keypoint_names": KEYPOINT_NAMES,
"feature_count": TOTAL_FEATURE_COUNT,
"keypoint_feature_count": KEYPOINT_FEATURE_COUNT,
"bbox_fields": BBOX_FIELDS,
"bbox_feature_count": BBOX_FEATURE_COUNT,
"person_count": len(predictions),
"inference_seconds": round(
float(elapsed),
6,
),
"inference_ms": round(
float(elapsed * 1000.0),
3,
),
"predictions": predictions,
"annotated_output": (
str(output_path)
if output_path is not None
else None
),
}
def main() -> int:
parser = argparse.ArgumentParser(
description=__doc__
)
parser.add_argument(
"--image",
required=True,
type=Path,
help="Input image for pose estimation.",
)
parser.add_argument(
"--model",
type=Path,
default=DEFAULT_MODEL,
help="Path to the YOLO26x-Pose checkpoint.",
)
parser.add_argument(
"--imgsz",
type=int,
default=DEFAULT_IMAGE_SIZE,
help="Inference image size.",
)
parser.add_argument(
"--conf",
type=float,
default=DEFAULT_CONFIDENCE,
help="Person detection confidence threshold.",
)
parser.add_argument(
"--iou",
type=float,
default=DEFAULT_IOU,
help="NMS IoU threshold.",
)
parser.add_argument(
"--device",
default=DEFAULT_DEVICE,
help="Inference device, e.g. 0, 1, cpu.",
)
parser.add_argument(
"--max-det",
type=int,
default=100,
help="Maximum number of detections.",
)
parser.add_argument(
"--output",
type=Path,
help="Optional path for the annotated pose image.",
)
args = parser.parse_args()
if args.imgsz <= 0:
parser.error("--imgsz must be greater than zero.")
if not 0.0 <= args.conf <= 1.0:
parser.error("--conf must be between 0 and 1.")
if not 0.0 <= args.iou <= 1.0:
parser.error("--iou must be between 0 and 1.")
if args.max_det <= 0:
parser.error("--max-det must be greater than zero.")
image = validate_image(
args.image
)
model = load_model(
args.model
)
result, elapsed = run_pose(
model=model,
image=image,
image_size=args.imgsz,
confidence=args.conf,
iou=args.iou,
device=args.device,
max_det=args.max_det,
)
output_path = args.output
if output_path is not None:
save_annotated_result(
result=result,
output_path=output_path,
)
payload = build_payload(
image_path=args.image,
image=image,
model_path=args.model,
result=result,
elapsed=elapsed,
image_size=args.imgsz,
confidence=args.conf,
iou=args.iou,
device=args.device,
output_path=output_path,
)
print(
json.dumps(
payload,
indent=2,
)
)
return 0
if __name__ == "__main__":
raise SystemExit(
main()
)