from __future__ import annotations import numpy as np # MediaPipe Pose landmark indices. NOSE = 0 LEFT_SHOULDER, RIGHT_SHOULDER = 11, 12 LEFT_HIP, RIGHT_HIP = 23, 24 LEFT_ANKLE, RIGHT_ANKLE = 27, 28 N_LANDMARKS = 33 N_RAW_CHANNELS = 4 # x, y, z, visibility N_ENGINEERED = 8 FEATURE_DIM = N_LANDMARKS * N_RAW_CHANNELS + N_ENGINEERED CORE_LANDMARKS = [ NOSE, LEFT_SHOULDER, RIGHT_SHOULDER, 13, # left elbow 14, # right elbow 15, # left wrist 16, # right wrist LEFT_HIP, RIGHT_HIP, 25, # left knee 26, # right knee LEFT_ANKLE, RIGHT_ANKLE, ] CORE_FEATURE_INDICES = np.asarray( [ landmark * N_RAW_CHANNELS + channel for landmark in CORE_LANDMARKS for channel in range(N_RAW_CHANNELS) ] + list(range(N_LANDMARKS * N_RAW_CHANNELS, FEATURE_DIM)), dtype=np.int64, ) def _interpolate_missing(sequence: np.ndarray, visibility_threshold: float) -> np.ndarray: """Interpolate coordinates over time while retaining original visibility.""" result = sequence.astype(np.float32, copy=True) timesteps = np.arange(result.shape[0]) for landmark in range(N_LANDMARKS): valid = result[:, landmark, 3] >= visibility_threshold for channel in range(3): values = result[:, landmark, channel] if valid.any(): result[:, landmark, channel] = np.interp( timesteps, timesteps[valid], values[valid] ) else: result[:, landmark, channel] = 0.0 return result def normalize_pose_sequence( sequence: np.ndarray, visibility_threshold: float = 0.35 ) -> np.ndarray: """Convert [T, 33, 4] landmarks into translation/scale-normalized features. Coordinates are centered at the midpoint of the hips and divided by torso length. Eight interpretable geometry/motion features are appended per frame. """ sequence = np.asarray(sequence, dtype=np.float32) if sequence.ndim != 3 or sequence.shape[1:] != (N_LANDMARKS, N_RAW_CHANNELS): raise ValueError(f"Expected [T, 33, 4], received {sequence.shape}") pose = _interpolate_missing(sequence, visibility_threshold) xyz = pose[:, :, :3] visibility = np.clip(pose[:, :, 3:4], 0.0, 1.0) hip_center = (xyz[:, LEFT_HIP] + xyz[:, RIGHT_HIP]) / 2.0 shoulder_center = (xyz[:, LEFT_SHOULDER] + xyz[:, RIGHT_SHOULDER]) / 2.0 torso_length = np.linalg.norm(shoulder_center - hip_center, axis=1) shoulder_width = np.linalg.norm( xyz[:, LEFT_SHOULDER] - xyz[:, RIGHT_SHOULDER], axis=1 ) scale = np.maximum(torso_length, shoulder_width) scale = np.maximum(scale, 1e-3) normalized_xyz = (xyz - hip_center[:, None, :]) / scale[:, None, None] visible_mask = visibility[:, :, 0] >= visibility_threshold x = normalized_xyz[:, :, 0] y = normalized_xyz[:, :, 1] min_x = np.where(visible_mask, x, np.inf).min(axis=1) max_x = np.where(visible_mask, x, -np.inf).max(axis=1) min_y = np.where(visible_mask, y, np.inf).min(axis=1) max_y = np.where(visible_mask, y, -np.inf).max(axis=1) width = np.where(np.isfinite(min_x) & np.isfinite(max_x), max_x - min_x, 0.0) height = np.where(np.isfinite(min_y) & np.isfinite(max_y), max_y - min_y, 0.0) torso_vector = shoulder_center - hip_center torso_angle = np.arctan2(np.abs(torso_vector[:, 0]), np.abs(torso_vector[:, 1]) + 1e-6) bbox_ratio = height / (width + 1e-3) shoulder_slope = np.abs( normalized_xyz[:, LEFT_SHOULDER, 1] - normalized_xyz[:, RIGHT_SHOULDER, 1] ) ankle_y = ( normalized_xyz[:, LEFT_ANKLE, 1] + normalized_xyz[:, RIGHT_ANKLE, 1] ) / 2.0 head_to_ankle = np.abs(normalized_xyz[:, NOSE, 1] - ankle_y) mean_visibility = visibility[:, :, 0].mean(axis=1) # Motion remains in image coordinates because centering removes global motion. hip_y = hip_center[:, 1] hip_velocity = np.gradient(hip_y) hip_acceleration = np.gradient(hip_velocity) nose_velocity = np.gradient(xyz[:, NOSE, 1]) engineered = np.column_stack( [ torso_angle, bbox_ratio, shoulder_slope, head_to_ankle, mean_visibility, hip_velocity, hip_acceleration, nose_velocity, ] ).astype(np.float32) flattened = np.concatenate([normalized_xyz, visibility], axis=2).reshape( sequence.shape[0], -1 ) return np.nan_to_num(np.concatenate([flattened, engineered], axis=1)).astype( np.float32 ) def featurize_dataset( poses: np.ndarray, visibility_threshold: float = 0.35 ) -> np.ndarray: """Featurize a batch shaped [N, T, 33, 4].""" return np.stack( [normalize_pose_sequence(seq, visibility_threshold) for seq in poses] ) def summarize_sequence_features(features: np.ndarray) -> np.ndarray: """Aggregate temporal features for classical ML baselines.""" features = np.asarray(features, dtype=np.float32) means = features.mean(axis=1) stds = features.std(axis=1) mins = features.min(axis=1) maxs = features.max(axis=1) first_last_delta = features[:, -1] - features[:, 0] return np.concatenate([means, stds, mins, maxs, first_last_delta], axis=1)