File size: 5,292 Bytes
9313a90 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 | from __future__ import annotations
import numpy as np
# MediaPipe Pose landmark indices.
NOSE = 0
LEFT_SHOULDER, RIGHT_SHOULDER = 11, 12
LEFT_HIP, RIGHT_HIP = 23, 24
LEFT_ANKLE, RIGHT_ANKLE = 27, 28
N_LANDMARKS = 33
N_RAW_CHANNELS = 4 # x, y, z, visibility
N_ENGINEERED = 8
FEATURE_DIM = N_LANDMARKS * N_RAW_CHANNELS + N_ENGINEERED
CORE_LANDMARKS = [
NOSE,
LEFT_SHOULDER,
RIGHT_SHOULDER,
13, # left elbow
14, # right elbow
15, # left wrist
16, # right wrist
LEFT_HIP,
RIGHT_HIP,
25, # left knee
26, # right knee
LEFT_ANKLE,
RIGHT_ANKLE,
]
CORE_FEATURE_INDICES = np.asarray(
[
landmark * N_RAW_CHANNELS + channel
for landmark in CORE_LANDMARKS
for channel in range(N_RAW_CHANNELS)
]
+ list(range(N_LANDMARKS * N_RAW_CHANNELS, FEATURE_DIM)),
dtype=np.int64,
)
def _interpolate_missing(sequence: np.ndarray, visibility_threshold: float) -> np.ndarray:
"""Interpolate coordinates over time while retaining original visibility."""
result = sequence.astype(np.float32, copy=True)
timesteps = np.arange(result.shape[0])
for landmark in range(N_LANDMARKS):
valid = result[:, landmark, 3] >= visibility_threshold
for channel in range(3):
values = result[:, landmark, channel]
if valid.any():
result[:, landmark, channel] = np.interp(
timesteps, timesteps[valid], values[valid]
)
else:
result[:, landmark, channel] = 0.0
return result
def normalize_pose_sequence(
sequence: np.ndarray, visibility_threshold: float = 0.35
) -> np.ndarray:
"""Convert [T, 33, 4] landmarks into translation/scale-normalized features.
Coordinates are centered at the midpoint of the hips and divided by torso
length. Eight interpretable geometry/motion features are appended per frame.
"""
sequence = np.asarray(sequence, dtype=np.float32)
if sequence.ndim != 3 or sequence.shape[1:] != (N_LANDMARKS, N_RAW_CHANNELS):
raise ValueError(f"Expected [T, 33, 4], received {sequence.shape}")
pose = _interpolate_missing(sequence, visibility_threshold)
xyz = pose[:, :, :3]
visibility = np.clip(pose[:, :, 3:4], 0.0, 1.0)
hip_center = (xyz[:, LEFT_HIP] + xyz[:, RIGHT_HIP]) / 2.0
shoulder_center = (xyz[:, LEFT_SHOULDER] + xyz[:, RIGHT_SHOULDER]) / 2.0
torso_length = np.linalg.norm(shoulder_center - hip_center, axis=1)
shoulder_width = np.linalg.norm(
xyz[:, LEFT_SHOULDER] - xyz[:, RIGHT_SHOULDER], axis=1
)
scale = np.maximum(torso_length, shoulder_width)
scale = np.maximum(scale, 1e-3)
normalized_xyz = (xyz - hip_center[:, None, :]) / scale[:, None, None]
visible_mask = visibility[:, :, 0] >= visibility_threshold
x = normalized_xyz[:, :, 0]
y = normalized_xyz[:, :, 1]
min_x = np.where(visible_mask, x, np.inf).min(axis=1)
max_x = np.where(visible_mask, x, -np.inf).max(axis=1)
min_y = np.where(visible_mask, y, np.inf).min(axis=1)
max_y = np.where(visible_mask, y, -np.inf).max(axis=1)
width = np.where(np.isfinite(min_x) & np.isfinite(max_x), max_x - min_x, 0.0)
height = np.where(np.isfinite(min_y) & np.isfinite(max_y), max_y - min_y, 0.0)
torso_vector = shoulder_center - hip_center
torso_angle = np.arctan2(np.abs(torso_vector[:, 0]), np.abs(torso_vector[:, 1]) + 1e-6)
bbox_ratio = height / (width + 1e-3)
shoulder_slope = np.abs(
normalized_xyz[:, LEFT_SHOULDER, 1] - normalized_xyz[:, RIGHT_SHOULDER, 1]
)
ankle_y = (
normalized_xyz[:, LEFT_ANKLE, 1] + normalized_xyz[:, RIGHT_ANKLE, 1]
) / 2.0
head_to_ankle = np.abs(normalized_xyz[:, NOSE, 1] - ankle_y)
mean_visibility = visibility[:, :, 0].mean(axis=1)
# Motion remains in image coordinates because centering removes global motion.
hip_y = hip_center[:, 1]
hip_velocity = np.gradient(hip_y)
hip_acceleration = np.gradient(hip_velocity)
nose_velocity = np.gradient(xyz[:, NOSE, 1])
engineered = np.column_stack(
[
torso_angle,
bbox_ratio,
shoulder_slope,
head_to_ankle,
mean_visibility,
hip_velocity,
hip_acceleration,
nose_velocity,
]
).astype(np.float32)
flattened = np.concatenate([normalized_xyz, visibility], axis=2).reshape(
sequence.shape[0], -1
)
return np.nan_to_num(np.concatenate([flattened, engineered], axis=1)).astype(
np.float32
)
def featurize_dataset(
poses: np.ndarray, visibility_threshold: float = 0.35
) -> np.ndarray:
"""Featurize a batch shaped [N, T, 33, 4]."""
return np.stack(
[normalize_pose_sequence(seq, visibility_threshold) for seq in poses]
)
def summarize_sequence_features(features: np.ndarray) -> np.ndarray:
"""Aggregate temporal features for classical ML baselines."""
features = np.asarray(features, dtype=np.float32)
means = features.mean(axis=1)
stds = features.std(axis=1)
mins = features.min(axis=1)
maxs = features.max(axis=1)
first_last_delta = features[:, -1] - features[:, 0]
return np.concatenate([means, stds, mins, maxs, first_last_delta], axis=1)
|