File size: 5,292 Bytes
9313a90
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
from __future__ import annotations

import numpy as np

# MediaPipe Pose landmark indices.
NOSE = 0
LEFT_SHOULDER, RIGHT_SHOULDER = 11, 12
LEFT_HIP, RIGHT_HIP = 23, 24
LEFT_ANKLE, RIGHT_ANKLE = 27, 28
N_LANDMARKS = 33
N_RAW_CHANNELS = 4  # x, y, z, visibility
N_ENGINEERED = 8
FEATURE_DIM = N_LANDMARKS * N_RAW_CHANNELS + N_ENGINEERED
CORE_LANDMARKS = [
    NOSE,
    LEFT_SHOULDER,
    RIGHT_SHOULDER,
    13,  # left elbow
    14,  # right elbow
    15,  # left wrist
    16,  # right wrist
    LEFT_HIP,
    RIGHT_HIP,
    25,  # left knee
    26,  # right knee
    LEFT_ANKLE,
    RIGHT_ANKLE,
]
CORE_FEATURE_INDICES = np.asarray(
    [
        landmark * N_RAW_CHANNELS + channel
        for landmark in CORE_LANDMARKS
        for channel in range(N_RAW_CHANNELS)
    ]
    + list(range(N_LANDMARKS * N_RAW_CHANNELS, FEATURE_DIM)),
    dtype=np.int64,
)


def _interpolate_missing(sequence: np.ndarray, visibility_threshold: float) -> np.ndarray:
    """Interpolate coordinates over time while retaining original visibility."""
    result = sequence.astype(np.float32, copy=True)
    timesteps = np.arange(result.shape[0])
    for landmark in range(N_LANDMARKS):
        valid = result[:, landmark, 3] >= visibility_threshold
        for channel in range(3):
            values = result[:, landmark, channel]
            if valid.any():
                result[:, landmark, channel] = np.interp(
                    timesteps, timesteps[valid], values[valid]
                )
            else:
                result[:, landmark, channel] = 0.0
    return result


def normalize_pose_sequence(
    sequence: np.ndarray, visibility_threshold: float = 0.35
) -> np.ndarray:
    """Convert [T, 33, 4] landmarks into translation/scale-normalized features.

    Coordinates are centered at the midpoint of the hips and divided by torso
    length. Eight interpretable geometry/motion features are appended per frame.
    """
    sequence = np.asarray(sequence, dtype=np.float32)
    if sequence.ndim != 3 or sequence.shape[1:] != (N_LANDMARKS, N_RAW_CHANNELS):
        raise ValueError(f"Expected [T, 33, 4], received {sequence.shape}")

    pose = _interpolate_missing(sequence, visibility_threshold)
    xyz = pose[:, :, :3]
    visibility = np.clip(pose[:, :, 3:4], 0.0, 1.0)

    hip_center = (xyz[:, LEFT_HIP] + xyz[:, RIGHT_HIP]) / 2.0
    shoulder_center = (xyz[:, LEFT_SHOULDER] + xyz[:, RIGHT_SHOULDER]) / 2.0
    torso_length = np.linalg.norm(shoulder_center - hip_center, axis=1)
    shoulder_width = np.linalg.norm(
        xyz[:, LEFT_SHOULDER] - xyz[:, RIGHT_SHOULDER], axis=1
    )
    scale = np.maximum(torso_length, shoulder_width)
    scale = np.maximum(scale, 1e-3)

    normalized_xyz = (xyz - hip_center[:, None, :]) / scale[:, None, None]

    visible_mask = visibility[:, :, 0] >= visibility_threshold
    x = normalized_xyz[:, :, 0]
    y = normalized_xyz[:, :, 1]
    min_x = np.where(visible_mask, x, np.inf).min(axis=1)
    max_x = np.where(visible_mask, x, -np.inf).max(axis=1)
    min_y = np.where(visible_mask, y, np.inf).min(axis=1)
    max_y = np.where(visible_mask, y, -np.inf).max(axis=1)
    width = np.where(np.isfinite(min_x) & np.isfinite(max_x), max_x - min_x, 0.0)
    height = np.where(np.isfinite(min_y) & np.isfinite(max_y), max_y - min_y, 0.0)

    torso_vector = shoulder_center - hip_center
    torso_angle = np.arctan2(np.abs(torso_vector[:, 0]), np.abs(torso_vector[:, 1]) + 1e-6)
    bbox_ratio = height / (width + 1e-3)
    shoulder_slope = np.abs(
        normalized_xyz[:, LEFT_SHOULDER, 1] - normalized_xyz[:, RIGHT_SHOULDER, 1]
    )
    ankle_y = (
        normalized_xyz[:, LEFT_ANKLE, 1] + normalized_xyz[:, RIGHT_ANKLE, 1]
    ) / 2.0
    head_to_ankle = np.abs(normalized_xyz[:, NOSE, 1] - ankle_y)
    mean_visibility = visibility[:, :, 0].mean(axis=1)

    # Motion remains in image coordinates because centering removes global motion.
    hip_y = hip_center[:, 1]
    hip_velocity = np.gradient(hip_y)
    hip_acceleration = np.gradient(hip_velocity)
    nose_velocity = np.gradient(xyz[:, NOSE, 1])

    engineered = np.column_stack(
        [
            torso_angle,
            bbox_ratio,
            shoulder_slope,
            head_to_ankle,
            mean_visibility,
            hip_velocity,
            hip_acceleration,
            nose_velocity,
        ]
    ).astype(np.float32)

    flattened = np.concatenate([normalized_xyz, visibility], axis=2).reshape(
        sequence.shape[0], -1
    )
    return np.nan_to_num(np.concatenate([flattened, engineered], axis=1)).astype(
        np.float32
    )


def featurize_dataset(
    poses: np.ndarray, visibility_threshold: float = 0.35
) -> np.ndarray:
    """Featurize a batch shaped [N, T, 33, 4]."""
    return np.stack(
        [normalize_pose_sequence(seq, visibility_threshold) for seq in poses]
    )


def summarize_sequence_features(features: np.ndarray) -> np.ndarray:
    """Aggregate temporal features for classical ML baselines."""
    features = np.asarray(features, dtype=np.float32)
    means = features.mean(axis=1)
    stds = features.std(axis=1)
    mins = features.min(axis=1)
    maxs = features.max(axis=1)
    first_last_delta = features[:, -1] - features[:, 0]
    return np.concatenate([means, stds, mins, maxs, first_last_delta], axis=1)